diff --git a/.gitattributes b/.gitattributes new file mode 100644 index 0000000..0ba4720 --- /dev/null +++ b/.gitattributes @@ -0,0 +1,19 @@ +# Normalize line endings in the repository. Without this the working tree's +# native endings leak into commits: 3 files were converted CRLF->LF inside an +# unrelated behavioural fix, inflating that diff from 51 real lines to 1045 and +# burying the change under formatting noise. 14 other tracked files were still +# CRLF, so the next contributor would have regenerated it. +* text=auto + +# Binary formats git must not touch. +*.png binary +*.jpg binary +*.jpeg binary +*.gif binary +*.ico binary +*.pdf binary +*.woff binary +*.woff2 binary +*.mp3 binary +*.mp4 binary +*.wav binary diff --git a/.github/workflows/brand-sync.yml b/.github/workflows/brand-sync.yml new file mode 100644 index 0000000..43926f8 --- /dev/null +++ b/.github/workflows/brand-sync.yml @@ -0,0 +1,51 @@ +# Vendored into every repo listed in blockrun's brand/consumers.json — the +# fan-out half of the brand-numbers system. The vendored sync script keeps PR +# CI deterministic and offline (--check never fetches); THIS workflow is where +# freshness comes from: it re-fetches the canonical artifact weekly and lands +# the marker rewrites, so numbers can't silently rot behind green CI again +# (they did: 14 of 15 consumers sat stale, one 70→71 sweep touched 8 files in +# this repo alone). +# +# Direct push to the default branch on purpose: these are docs/marker rewrites +# generated from blockrun.ai/brand/numbers.json by the audited script in +# scripts/. A PR queue nobody tends is how the staleness happened. If the +# branch is protected the push fails and the fallback opens a PR instead. +# No third-party actions beyond actions/checkout — supply-chain surface stays +# at one GitHub-owned action. +name: brand-sync +on: + schedule: + - cron: "17 6 * * 1" # Mondays 06:17 UTC, offset from the top-of-hour crunch + workflow_dispatch: +permissions: + contents: write + pull-requests: write +jobs: + sync: + runs-on: ubuntu-latest + steps: + - uses: actions/checkout@v4 + - name: Refresh brand numbers from the canonical artifact + run: node scripts/sync-brand-numbers.mjs --refresh + - name: Land the rewrite + env: + GH_TOKEN: ${{ github.token }} + run: | + if git diff --quiet; then + echo "already in sync" + exit 0 + fi + git config user.name "github-actions[bot]" + git config user.email "41898282+github-actions[bot]@users.noreply.github.com" + git add -A + git commit -m "chore: sync brand numbers from blockrun.ai/brand/numbers.json" + if git push origin "HEAD:${GITHUB_REF_NAME}"; then + echo "pushed to ${GITHUB_REF_NAME}" + else + BR="brand-sync/$(date +%Y%m%d)" + git push -f origin "HEAD:${BR}" + gh pr create --head "$BR" \ + --title "chore: sync brand numbers" \ + --body "Automated marker refresh from https://blockrun.ai/brand/numbers.json (scripts/sync-brand-numbers.mjs --refresh). Opened as a PR because the default branch is protected." \ + || echo "PR creation unavailable — branch ${BR} pushed, needs manual PR" + fi diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml new file mode 100644 index 0000000..ad92982 --- /dev/null +++ b/.github/workflows/ci.yml @@ -0,0 +1,58 @@ +name: CI + +on: + push: + branches: [main] + pull_request: + branches: [main] + workflow_dispatch: + +jobs: + test: + runs-on: ubuntu-latest + strategy: + matrix: + python-version: ['3.9', '3.11', '3.12'] + + steps: + - uses: actions/checkout@v4 + + - name: Set up Python ${{ matrix.python-version }} + uses: actions/setup-python@v5 + with: + python-version: ${{ matrix.python-version }} + + - name: Install dependencies + run: | + if python3 -c "import sys; exit(0 if sys.version_info >= (3, 10) else 1)"; then + pip install -e ".[dev,solana,anthropic]" + else + pip install -e ".[dev,anthropic]" + fi + + - name: Check formatting + run: black --check . + + - name: Lint + run: ruff check . + + - name: Run unit tests + run: | + if python3 -c "import sys; exit(0 if sys.version_info >= (3, 10) else 1)"; then + pytest tests/unit + else + pytest tests/unit --ignore=tests/unit/test_solana_client.py --ignore=tests/unit/test_solana_wallet.py -k "not SolanaX402" + fi + + # Fails when a number in the docs disagrees with brand-numbers.json. + # Offline by design: it reads the committed snapshot and never fetches, so a + # blockrun.ai deploy in progress cannot fail this repo's CI. Its own job + # rather than a step, so it does not run three times across the Python matrix. + brand-numbers: + runs-on: ubuntu-latest + steps: + - uses: actions/checkout@v4 + - uses: actions/setup-node@v4 + with: + node-version: 22 + - run: node scripts/sync-brand-numbers.mjs --check diff --git a/.github/workflows/publish.yml b/.github/workflows/publish.yml new file mode 100644 index 0000000..5ee5c8e --- /dev/null +++ b/.github/workflows/publish.yml @@ -0,0 +1,76 @@ +name: Publish to PyPI + +on: + release: + types: [published] + +jobs: + build: + runs-on: ubuntu-latest + steps: + - uses: actions/checkout@v4 + + - name: Set up Python + uses: actions/setup-python@v5 + with: + python-version: '3.11' + + # 1.8.1 shipped with VERSION and __init__.py still reading 1.8.0. The + # guard test caught it, but only on the push-triggered CI run, after the + # release had been cut and published. PyPI does not allow overwriting a + # published file, so that artifact is permanently wrong. + # + # These steps run ci.yml's checks on the 3.11 leg only — black, ruff and + # the unit suite. They are NOT full CI: the 3.9 and 3.12 legs still run + # only on push, so version-incompatible syntax is caught there, not here. + # Extras match the 3.11 CI job so the gate covers the Solana paths; with + # the plain extra those tests importorskip and pass silently. + - name: Install package and test deps + run: pip install -e ".[dev,solana]" build + + # A release tag that disagrees with the package version means the wrong + # tree is being published. PyPI would reject a duplicate version, but + # only after the release is cut; fail here instead. + - name: Verify the release tag matches VERSION + run: | + tag="${{ github.event.release.tag_name }}" + declared="v$(tr -d '[:space:]' < VERSION)" + if [ "$tag" != "$declared" ]; then + echo "Release tag $tag does not match VERSION ($declared)." >&2 + exit 1 + fi + echo "Release tag $tag matches VERSION." + + - name: Check formatting + run: black --check blockrun_llm/ tests/ + + - name: Lint + run: ruff check blockrun_llm/ tests/ + + - name: Verify the release is publishable + run: pytest tests/unit -q + + - name: Build package + run: python -m build + + - name: Upload artifacts + uses: actions/upload-artifact@v4 + with: + name: dist + path: dist/ + + publish: + needs: build + runs-on: ubuntu-latest + environment: pypi + permissions: + id-token: write + steps: + - name: Download artifacts + uses: actions/download-artifact@v4 + with: + name: dist + path: dist/ + + - name: Publish to PyPI + uses: pypa/gh-action-pypi-publish@release/v1 diff --git a/.gitignore b/.gitignore index 9e9b7f0..df3eca7 100644 --- a/.gitignore +++ b/.gitignore @@ -1,56 +1,66 @@ -# Byte-compiled / optimized / DLL files -__pycache__/ -*.py[cod] -*$py.class - -# Distribution / packaging -.Python -build/ -develop-eggs/ -dist/ -downloads/ -eggs/ -.eggs/ -lib/ -lib64/ -parts/ -sdist/ -var/ -wheels/ -*.egg-info/ -.installed.cfg -*.egg - -# Virtual environments -venv/ -env/ -.venv/ -.env/ - -# Environment files -.env -.env.local -.env.*.local - -# IDE -.vscode/ -.idea/ -*.swp -*.swo - -# OS -.DS_Store -Thumbs.db - -# Test / coverage -.coverage -.pytest_cache/ -htmlcov/ -.tox/ -.nox/ - -# mypy -.mypy_cache/ - -# Jupyter -.ipynb_checkpoints/ +# Byte-compiled / optimized / DLL files +__pycache__/ +*.py[cod] +*$py.class + +# Distribution / packaging +.Python +build/ +develop-eggs/ +dist/ +downloads/ +eggs/ +.eggs/ +lib/ +lib64/ +parts/ +sdist/ +var/ +wheels/ +*.egg-info/ +.installed.cfg +*.egg + +# Virtual environments +venv/ +env/ +.venv/ +.env/ + +# Environment files +.env +.env.local +.env.*.local + +# IDE +.vscode/ +.idea/ +*.swp +*.swo + +# OS +.DS_Store +Thumbs.db + +# Test / coverage +.coverage +.pytest_cache/ +htmlcov/ +.tox/ +.nox/ + +# mypy +.mypy_cache/ + +# Jupyter +.ipynb_checkpoints/ + +# Local Claude Code config (machine-specific allowlists) +.claude/settings.local.json + +# Sweep / test-run artifacts +sweep-*.json +sweep-*.log + +# Ruff cache +.ruff_cache/ diff --git a/AGENTS.md b/AGENTS.md new file mode 100644 index 0000000..6fee3be --- /dev/null +++ b/AGENTS.md @@ -0,0 +1,138 @@ +# AGENTS.md + +Guidance for AI coding agents working with the BlockRun Python SDK. + +## Project Overview + +**blockrun-llm** is a Python SDK for pay-per-request access to AI models (GPT, Claude, Gemini, DeepSeek, NVIDIA) using API account credit or x402 wallets on Solana and Base. **Includes 8 fully-free NVIDIA-hosted models** — DeepSeek V4 Flash (1M ctx), Nemotron Nano Omni (vision), Qwen3 Next + Coder, Llama 4 Maverick, Mistral Small 4, plus `gpt-oss-120b/20b` (hidden from `/v1/models` but direct calls still work). Accessible via `routing_profile="free"` or any `nvidia/*` model id. + +**Package:** `blockrun-llm` (PyPI) +**Python:** >=3.9 +**Networks:** Solana and Base (Chain ID: 8453) +**Payment:** API account credit, or USDC via x402 v2; free models are also available. + +## Repository Structure + +``` +blockrun-llm/ +├── blockrun_llm/ +│ ├── __init__.py # Package exports +│ ├── client.py # LLMClient, AsyncLLMClient +│ ├── anthropic_client.py # AnthropicClient (official SDK wrapper) +│ ├── solana_client.py # SolanaLLMClient (Solana payments) +│ ├── router.py # ClawRouter smart routing +│ ├── image.py # Image generation client +│ ├── types.py # Pydantic models and type definitions +│ ├── validation.py # Input validation utilities +│ ├── wallet.py # Wallet operations (signing, address) +│ ├── solana_wallet.py # Solana wallet utilities +│ ├── cache.py # Response caching & cost logging +│ └── x402.py # x402 payment protocol implementation +├── tests/ +│ ├── unit/ # Unit tests (no API calls) +│ └── integration/ # Integration tests (requires funded wallet) +├── examples/ # Usage examples +├── pyproject.toml # Package configuration (hatchling) +└── README.md +``` + +## Development Commands + +```bash +# Setup +python -m venv .venv +source .venv/bin/activate +pip install -e ".[dev]" + +# Testing +pytest tests/unit # Unit tests only (no API key needed) +pytest tests/unit --cov # With coverage +pytest # All tests (requires BLOCKRUN_WALLET_KEY) + +# Code Quality +black blockrun_llm/ # Format code +ruff check blockrun_llm/ # Lint +mypy blockrun_llm/ # Type check +``` + +## Code Conventions + +### Style +- Black formatter (line-length: 100) +- Ruff linter +- Type hints required (mypy strict mode) + +### Architecture +- `LLMClient` - Synchronous client +- `AsyncLLMClient` - Async client with context manager +- Account API calls authenticate with a BlockRun key; wallet calls use x402. +- A configured invalid API key is an error, never permission to use a wallet. + +### Error Handling +- `APIError` - General API errors +- `PaymentError` - Payment-specific errors +- Errors are sanitized to prevent key leakage + +## Key Files + +| File | Purpose | +|------|---------| +| `client.py` | Main client classes with `chat()`, `chat_completion()`, `list_models()` | +| `x402.py` | x402 payment protocol (402 handling, payment signing) | +| `wallet.py` | Private key management, transaction signing | +| `validation.py` | Input validation for keys, URLs, parameters | +| `types.py` | Pydantic models for API requests/responses | + +## Testing + +### Unit Tests +No API key or funded wallet required: +```bash +pytest tests/unit -v +``` + +### Integration Tests +Requires `BLOCKRUN_WALLET_KEY` with funded Base wallet (~$1 USDC): +```bash +export BLOCKRUN_WALLET_KEY=0x... +pytest tests/integration -v +``` + +### Local Billing / Cost Tracking +Every paid call writes to `~/.blockrun/cost_log.jsonl` with model / wallet / +network metadata. To audit spending: +```bash +python -m blockrun_llm.billing summary --group-by model +python -m blockrun_llm.billing export csv --from 2026-05-01 --output may.csv +``` +Programmatic access via `from blockrun_llm import get_cost_log_summary, +export_cost_log_csv, export_cost_log_json`. Per-machine only; for +organization-wide accounting query the gateway's ledger. + +### End-to-End Model Sweeps +Before a release or after router/catalog changes: +```bash +python examples/sweep_all_chat_models.py --output-json sweep-results.json +python examples/sweep_all_media_models.py --output-json sweep-media-results.json +``` +Each script captures per-model status / latency / token counts / per-call +cost and exits non-zero if any expected-to-work model fails. The chat sweep +also runs a forward-compat diff against `/v1/models` to flag new IDs not in +the sweep list. Video is excluded from the media sweep by design. + +## Publishing + +```bash +# Build +python -m build + +# Upload to PyPI +twine upload dist/* +``` + +## Security Notes + +- Private keys never leave the machine (local signing only) +- Validate private key format before use +- HTTPS required for production API URLs +- Never log or expose private keys in errors diff --git a/CHANGELOG.md b/CHANGELOG.md new file mode 100644 index 0000000..6d9e2c2 --- /dev/null +++ b/CHANGELOG.md @@ -0,0 +1,1713 @@ +# Changelog + +All notable changes to blockrun-llm will be documented in this file. + +## 1.17.0 — 2026-09-16 + +### Added +- **Arc.** `LLMClient(api_url="https://arc.blockrun.ai/api")` pays on Circle's + Arc. The chain table knew Base and Base Sepolia and fell back to Base for + anything else, while `asset` and `extra` were taken from the 402 as given — + so against arc.blockrun.ai (`eip155:5042`, USDC at `0x3600…`, domain name + `USDC`) every payment was a signature over chainId 8453 with Arc's contract: + invalid, a 401 from the facilitator, after the SDK had reported a payment. + + `EVM_NETWORKS` in `blockrun_llm.x402` maps a 402's `network` to the SDK's + own chain id, USDC address and EIP-712 domain (Base, Arc, Base Sepolia; the + `base-sepolia` alias still resolves). The 402 selects the network and + supplies nothing else: its `extra` no longer reaches the domain, an unknown + network raises `ValueError` naming what is supported, and a 402 whose + `asset` is not that network's USDC raises before anything is signed. Every + EVM client passes the 402's `asset` through. `accepted.asset` and + `accepted.extra` in the payload now describe the network actually signed. + Mirrors `@blockrun/llm` 3.16.0. + + Verified against arc.blockrun.ai with an unfunded throwaway key: Circle's + `/verify` answers `insufficient_funds` and recovers the throwaway's own + address as `payer` — the signature verifies on Arc's domain; only the + balance is missing. + +## 1.16.0 — 2026-09-08 + +### Fixed +- **Music generation works, on both rails.** MiniMax takes one to three + minutes per track and the gateway answers `202 + poll_url` — since + 2026-09-08, immediately. `MusicClient` treated every non-200 as an error, so + a music request could not succeed at all: the gateway's ledger shows every + music create in the last 30 days answered 202, and this SDK raised + `API error: 202` for each. The client now polls the job the way the image + client already did — replaying the create's `PAYMENT-SIGNATURE` on the + wallet rail, carrying the key on the account rail — and returns the track on + the completed poll. Settlement happens on that poll, so a poll budget that + runs out (`MUSIC_POLL_BUDGET_SECONDS`, 300s) has cost nothing. The loop + itself moved to `blockrun_llm/jobs.py`; images and music share it. (#63) +- **`str(exc)` now carries the gateway's explanation.** Raise sites built the + message from the status code alone and stashed the body on `.response`, so + a free-tier stream 429 printed as `API error: 429` while the body said + "Free tier rate limit reached (30 requests/minute per IP)". The one line + worth reading reaches the string a caller logs. +- **`APIError.retry_after`.** api.blockrun.ai answers a rate limit with + `Retry-After` and it never survived the SDK boundary, so every account-rail + consumer had to guess or spin. The raw header is kept as sent (seconds or an + HTTP-date); `retry_after_seconds` resolves it to a number when it can. (#61) +- **Solana clients honour an explicit `api_url` on the account rail.** The + Solana constructors passed a hard-coded `None` to avoid sending an API key + to sol.blockrun.ai, which also discarded a URL the caller typed. The rule — + API key to api.blockrun.ai, Solana key to sol.blockrun.ai, Base key to + blockrun.ai — is now pinned by a table over all 17 exported clients. +- **A blank `BLOCKRUN_API_KEY=` reads as unset**, not as a key; explicit + payment selection survives client construction; the account clients that + were missing pieces of the wallet clients' surface are complete; Solana + signer initialisation is skipped for API accounts. (#60) + +### Added +- **`ChatUsage.reasoning_tokens`.** Reasoning models report their thinking + tokens under `completion_tokens_details.reasoning_tokens` (OpenAI shape) or + as a flat `reasoning_tokens`; both are read, the nested shape authoritative. + (#42) + +## 1.15.0 — 2026-09-05 + +### Added +- **A BlockRun API key works everywhere a wallet key does.** Every paid path in + this SDK assumed x402: a 402 challenge, a locally signed payment, a retry. + That made a wallet the price of admission, which is a non-starter for a team + whose finance function cannot hold USDC and for any CI runner that should not + carry a signing key. A key from [user.blockrun.ai](https://user.blockrun.ai) + now works in the same place. It is not a new client type and not a new + constructor: the `private_key` parameter every client already takes now + accepts a `brk_` key, and `BLOCKRUN_API_KEY` is read when it is empty — so + fourteen client classes and every skill that calls them gained the rail + without a signature change. Requests go to `api.blockrun.ai` as + `Authorization: Bearer …`, draw prepaid credit, and never sign anything. + `client.payment_mode` reports which rail a client ended up on. + + Four things had to change beyond attaching a header, and each was a silent + failure rather than an import error: + + - `api.blockrun.ai` serves `/v1/...` at the root and answers `/api/v1/...` + with `wrong_host`, so the account rail needs its own base URL. + - `poll_url` is minted by the x402 gateway relative to *its* host, so it + arrives as `/api/v1/...`. Resolved unchanged it would have sent every async + job — video, slow images — polling a 404 until its budget ran out. + - On the account rail the async submit answers **202 on the first POST**. + `ImageClient` raised `API error: 202` and `VideoClient` raised + "Expected 402 on first POST", so both were broken for API keys before they + started. + - `setup_agent_wallet()` minted a keyfile unconditionally. With a key + configured there is nothing to sign with, so it now returns an API-key + client and writes nothing to disk — which is what lets an agent or a skill + call it on either rail. + + `SolanaLLMClient` and `AsyncSolanaLLMClient` take the key too, and no longer + require the optional x402 SDK when one is present: on the account rail there + is no transfer to sign, so the chain stops being a question. + +- **`blockrun_llm.apikey`** holds the rail in one module — precedence, + `auth_headers`, poll-URL resolution, and the two refusals — so it is one + decision made once rather than fourteen copies that can drift. + +### Changed +- **Wallet-only helpers refuse instead of answering wrongly.** `get_balance()`, + `get_balance_testnet()` and `onramp()` raise a `ValueError` naming the helper + and pointing at the dashboard. Returning `0` would have been the worst + available answer — indistinguishable from an empty wallet, and an agent + gating on it would stop calling a well-funded account. + `get_wallet_address()` returns `""`. A 402 on this rail is a credit refusal, + not a challenge, so it raises a `PaymentError` quoting the gateway's own + reason and the top-up page. +- **One "nothing configured" message.** Every client raised its own wording + listing only wallet routes, which stopped being the whole truth the moment a + key became a credential. `missing_credential_error()` lists both. +- **README covers both rails and puts Solana ahead of Base.** It opened with + "No API keys required", which is no longer true. Adds an API-key path to + Quick Start and a full *Option A* section (signup, key minting, top-up — + minimum $5, with the 5.5% + $0.30 fee charged once at purchase rather than + per call — precedence, and what changes). The environment table listed two + variables and omitted `SOLANA_WALLET_KEY` entirely; it now lists nine. + +Wallet users are unaffected: precedence is an explicit argument, then +`BLOCKRUN_API_KEY`, then the wallet variables, so nothing changes until that +variable is set. `BLOCKRUN_API_KEY_URL` is deliberately separate from +`BLOCKRUN_API_URL` — the latter names an x402 gateway, and following it would +send the key to a host configured for another rail. + +## 1.14.0 — 2026-08-31 + +### Changed +- **Router Core re-synced to upstream `5ee7c23`** (was `d7bc10c`, two commits + behind — the same pin the TypeScript SDK bundles). Upstream V3.5 rebuilds + every tier chain around ids the public catalog actually lists: the withheld + `kimi-k2.5/k2.6/k2.7`, both grok-4-fast pairs, `grok-4-0709`, + `claude-opus-4.6` and `gemini-3-pro-preview` are gone from every rung, + including fallbacks, so a routed model is always one a user can find on + blockrun.ai/models. Before this sync every profile carried 3–4 off-catalog + rungs per decision; now it carries none. + + Primaries moved only where `portfolio.py` already holds calibration evidence + for the successor — Gemini 3.5 Flash where Kimi K2.7 was, GPT-5 Mini for + agentic MEDIUM, Sonnet 5 over Sonnet 4.6, DeepSeek Reasoner for the cheap + reasoning head. The newer generation the catalog already sells enters as + fallback rungs: GPT-5.6 Luna/Terra, Gemini 3.6 Flash and 3.5 Flash-Lite, + GLM-5.3 and 5.3-Flash, Grok 4.3 and 4.5, Kimi K3, Qwen 3.7 Plus, MiniMax M3. + Promotion waits for a calibration run, because version recency is not a + quality signal. The expired GLM-5.1 promotion is dropped; the promotions + mechanism stays wired with an empty list. + +- **Routing priors regenerated: 66 model profiles** (was 30) and a **71-model + capability snapshot** (`model_capabilities.py`), both from upstream's probe + and catalog-sync scripts. Two stale capability values had been reaching the + hard filter: Haiku 4.5 at 8K max output (actually 64K) and Sonnet 4.6 at 200K + context (actually 1M). Kimi K3 replaces K2.7 in the Mandarin extraction band, + widened to 0.12 so the auto affinity floor gap (0.10) cannot let price + re-select a non-native model, and K3 joins the extraction evidence pool since + it is no longer on the MEDIUM chain. + +### Fixed +- **`routing_profile="free"` had collapsed to a single model with no + fallbacks.** Four of the five ids in the SDK-only `FREE_TIERS` table — + `nvidia/step-3.7-flash`, `nemotron-nano-9b-v2`, `mistral-nemotron`, + `nemotron-nano-12b-v2-vl` — were retired by NVIDIA, and they were every + primary plus all but one fallback. Nothing looked broken because the gateway + server-redirects a retired free id: a dead rung returns 200 and answers + normally, while quietly serving a different model. That is the shape that + defeats a host's `unavailable_models`, and it is why the table rotted + unnoticed. + + The table is rebuilt from ids verified by a two-pass probe that reads back + the response's own `model` field, since a 200 proves nothing. Two models the + catalog still prices at $0 did not survive that check and are excluded: + `nemotron-3-ultra-550b` and `nemotron-3-nano-omni-30b-a3b-reasoning` both + answer as `nemotron-3-nano-30b`. Every tier is back to four candidates, and + the free tier is no longer NVIDIA-only — `cohere/north-mini-code` and + `poolside/laguna-xs-2.1` serve at $0 and carry the free coding load. + + A new test asserts fallback *depth* per tier, not just membership: the table + stayed internally consistent all through the rot, so membership alone could + never have caught it. + +## 1.13.0 — 2026-08-26 + +### Fixed +- **Solana paid-leg re-sign is gated on payment PHASE, not on the failure + cause.** A settlement failure may already have broadcast the transfer, so a + lost acknowledgement could previously authorize a *second* payment for one + request. Settlement failures are now terminal on every paid path — sync and + async, streaming, non-streaming, raw POST and raw GET. + + Everything the gateway rejects **before** broadcast stays retryable, because + re-signing there costs the payer nothing and is what each rejection's own + message asks for: `PAYMENT_UNDERPAID` ("re-fetch the 402 quote and sign the + amount it specifies"), `PAYMENT_REPLAY` ("sign a new payment for each + request"), and every verification-phase rejection including + `expired_signature`, `verification_unavailable` and the `verification_failed` + catch-all that carries facilitator timeouts. These are exactly the concurrent + single-wallet failures the whole-request retry exists to fix, so the ~100% + success rate under concurrent load is preserved; gating on stale-blockhash + alone would have silently reverted it, since `_should_fallback_solana` refuses + every `PaymentError` and they would reach the caller with no second model + tried. + + `insufficient_funds` and the unrecoverable `invalidMessage` causes (no USDC + token account, bad signing key, denylisted payer) remain terminal — no fresh + signature makes them pass. + +- **`build_payment_rejected_error` preserves the gateway's `code` and `reason`.** + These are gateway-owned enums, length-bounded like `details` and + `invalidMessage`; `debug` stays filtered. Without them the client could only + classify a 402 by prose, and the gateway's two 402 body families disagree on + which fields exist: `/v1/chat/completions` sends `code` + `message` + + `reason`, while the other paid routes send `error` + `reason` only. + +### Added +- **Dead-model kill-switch: `unavailable_models`.** A host that observes a model + answering 400/404/410 at the gateway can pass its id in + `options["unavailable_models"]` and it is hard-removed from every routing + chain before selection — the first surviving rung is promoted to primary, and + an eligibility fail-open can never resurrect it. This is the operational + answer to a dead chain rung: effective on the next request instead of waiting + for a Router Core release plus SDK repins. `apply_unavailable_models` is + exported for hosts that manage tier maps directly. Port of upstream + `d7bc10c`, with the case suite ported 1:1. + +- **Cross-language decision-snapshot parity.** + `tests/unit/test_router_core_snapshot.py` recomputes the upstream frozen + corpus — 88 complete decisions (22 prompts × 4 profiles with rotating + tool/vision/structured-output shapes) — and compares every pinned field + against the TypeScript engine's committed fixture, floats and reasoning + strings included. Parity is now proven decision-by-decision rather than + test-case-by-test-case. + +### Changed +- **Router Core re-synced to upstream `d7bc10c`** (was `18bf4ab`, four commits + behind — the same pin the TypeScript SDK bundles). The visible routing + change: the free rungs retarget from the retired `free/gpt-oss-120b/20b` + pair (400 Unknown model at the gateway, probed 2026-08-21) to the current + NVIDIA free tier — `nvidia/step-3.7-flash` heads eco SIMPLE and the three + ultimate-backstop slots, `nvidia/nemotron-nano-9b-v2` takes the fast rung. + Both verified live by direct gateway calls. eco once again reaches a $0 + model on its first candidate; capability entries updated to match. + +## 1.12.0 — 2026-08-19 + +### Added +- **Smart routing on the Solana clients.** `SolanaLLMClient` and + `AsyncSolanaLLMClient` had no routing at all — `smart_chat`, `route` and the + routing profiles were Base-only, so a Solana user got no model selection while + the TypeScript SDK offered it on both chains. All four Python clients (Base and + Solana, sync and async) now expose the same surface: `route()`, `smart_chat()` + and `smart_chat_completion()`. Both chains run the same Router Core engine + against the same catalog, so an identical request picks an identical model; + only the x402 payment floor in the cost metadata differs ($0.002 Base, + $0.001 Solana). Pinned by `tests/unit/test_routing_parity.py`. + +- **`smart_chat_completion(messages, ...)`** on every client — routing for a full + message list rather than a single prompt. `tools`, `tool_choice` and + `response_format` are inputs to the *decision*, not just the request: a turn + that must call a tool routes to a tool-capable model, a JSON schema forces the + structured-output tier, and image parts route to a vision model. Capacity is + checked against the whole transcript, because an agent conversation can be + 100x its final turn and a context overflow is a non-transient error the + fallback chain cannot rescue. + +- **`blockrun/auto`, `blockrun/eco` and `blockrun/premium` virtual model ids.** + Passing one to `chat()` or `chat_completion()` routes the turn instead of + calling a model of that name, ranked fallback chain included — TypeScript SDK + parity, and it lets OpenAI-compatible code opt into routing by changing one + string. + +- **`fallback_models` on the Solana `chat()` / `chat_completion()`.** The + parameter existed only on the Solana streaming path, so a routed Solana call + had a recovery chain it could not walk. The chain now steps to the next ranked + model on a timeout, network error or 5xx, using the same + `_should_fallback_solana` classifier as the stream path — a settled payment is + never retried, so a second model cannot sign a second transfer for one call. + +### Fixed +- **One scoring dimension was silently weighted zero.** The config transpile that + produced `router_core/config.py` snake_cased key names, and `imperativeVerbs` + is both a keyword-list field *and* a dimension name — so its 0.03 weight + landed under `imperative_verbs` while the classifier emitted `imperativeVerbs`, + and `weights.get(name, 0)` scored it zero. Build/deploy-shaped requests were + under-classified: 3 of 8 sampled imperative prompts ("Create and deploy the + service", "Set up the config and deploy it", "Develop a CLI that generates + reports") landed in SIMPLE where they should have been ambiguous and defaulted + up to MEDIUM. Now guarded by a test asserting the weight keys and the emitted + dimension names are the same set, plus the verbatim upstream weight table. + Cross-SDK parity re-verified at 24/24 after the fix. + +- **A 429 now walks the fallback chain instead of failing the call.** Both + clients treated only 5xx as retriable, so a saturated upstream ended the + request even with capable models left in the chain. Found live: a rate-limited + free model answered 429 and the three remaining free models were never tried. + The TypeScript adapter has always counted 429 as transient — the next model in + the chain is a different upstream. Settled payments and permanent payment + failures are still refused before the status check, so no call can pay twice. + +### Changed +- The `/v1/models` → pricing-map conversion moved to + `router_adapter.build_model_pricing()`, shared by all four clients instead of + being written out per client. Rows the catalog marks `available: false` are + skipped everywhere now (previously only the Base sync client did this, as of + 1.11.0). + +## 1.11.0 — 2026-08-15 + +### Added +- **Router Core lands in the Python SDK.** `blockrun_llm/router_core/` is a + faithful port of [`@blockrun/router-core`](https://github.com/BlockRunAI/router-core) + (upstream commit `18bf4ab`) — the product-neutral routing engine the + TypeScript SDK bundles and the gateway runs. The same request now routes + identically across all three. `blockrun_llm/router_adapter.py` is the host + glue (catalog id resolution, x402 payment floors, capacity filtering), ported + from the TypeScript SDK's `src/router-adapter.ts`. + + What the Python SDK did not have before: + - **Portfolio (V3) ranking**, not just tier lookup: candidates are scored on + task affinity, cost, speed and reliability, so the cheapest *capable* model + wins instead of a hardcoded tier primary. + - **Hard capability filtering.** A model that cannot hold the conversation, + emit the requested `max_tokens`, call tools, or read images is dropped + before scoring — previously `smart_chat` could route to a model the request + would fail on with a non-transient 400. + - **Task classification** (`chat`, `code_edit`, `code_agent`, `tool_agent`, + `tool_agent_parallel`, `reasoning_math`, `reasoning_mcq`, `long_context`, + `extraction`, `vision`, `debug`) with per-task calibrated model evidence. + - **Explainable decisions**: `routing.candidates`, `routing.candidate_scores` + (quality / cost / speed / reliability per model), `routing.task_type`, + `routing.profile` and `routing.router_version` are now on the response. + - **Live tier configuration**, shared with the other products, replacing this + SDK's separately hand-maintained tables. + +- **`client.route(prompt, ...)`** returns the routing decision without making or + paying for a model call (TypeScript SDK parity). The first call may fetch the + public catalog for prices; routing itself is local and free. + +### Fixed +- **The `free` profile pointed at models NVIDIA has retired.** Its tier table + led with `nvidia/deepseek-v4-flash` (EOL 2026-08-12, HTTP 410) and fell back + to `nvidia/llama-4-maverick` and `nvidia/qwen3-coder-480b` (also EOL), so free + routing depended entirely on the gateway's redirect safety net. It now routes + over the live free lineup (Step 3.7 Flash, Mistral Nemotron, Nemotron Nano + Omni / 9B / 12B VL), and the adapter drops any candidate the catalog does not + price at $0 — a paid model can no longer leak into a free-profile call. +- **Models the catalog marks unavailable no longer win routing.** `/v1/models` + rows with `available: false` are skipped when building the pricing map; every + smart call to one would have failed with a non-transient error. + +### Changed +- `routing.method` is now `"portfolio"` for the default strategy (`"rules"` for + the free profile and the config-only V2 rollback). Code that asserted + `method == "rules"` needs updating. +- `blockrun_llm/router.py` is now a thin compatibility shim over the core: + `route()` and `classify_by_rules()` keep working, and `RoutingDecision` keeps + its previous keys plus the new metadata. Its hand-maintained `AUTO_TIERS` / + `ECO_TIERS` / `PREMIUM_TIERS` tables are gone — tier configuration lives in + `router_core.DEFAULT_ROUTING_CONFIG`, and `FREE_TIERS` moved to + `router_adapter`. +- Routing cost estimates now include the server margin and the x402 minimum + payment, so `routing.cost_estimate` matches what the gateway actually + charges. Free models are never floored up to the paid minimum. + +### Tests +- `tests/unit/test_router_core.py` ports all four upstream vitest suites + (88 cases) as the parity guard — the Python port must keep choosing the same + models as the TypeScript SDK. `tests/unit/test_router_adapter.py` covers the + host layer: `free/*` → `nvidia/*` id resolution, the payment floor, capacity + filtering, and the free-profile guarantees. + +## 1.10.0 — 2026-07-28 + +### Added +- **`solana_key_to_bytes` accepts the key formats users actually have on disk.** + Alongside the existing bs58 encodings (64-byte keypair and 32-byte seed), it + now decodes the Solana CLI JSON byte-array format (`~/.config/solana/id.json`) + and 64-byte hex keys with or without `0x`. Previously these failed with + `Invalid Solana private key: Non-base58 character` and no further guidance. + TypeScript SDK parity (@blockrun/llm 3.9.0). + +### Changed +- **An invalid Solana key now says where it was loaded from and what it appears + to be.** `get_or_create_solana_wallet` failures name the source (the + `SOLANA_WALLET_KEY` environment variable or the `~/.blockrun/.solana-session` + path). A 32-byte hex key is identified as the EVM (Base) wallet format rather + than rejected as a character-set error, and unrecognized input lists the + accepted formats. + +## 1.9.0 — 2026-07-21 + +### Added +- **Client-side spend limits.** `max_cost_per_call` and `max_session_cost` on + every client (Base and Solana, sync and async) refuse a quote that costs more + than you allowed: + + ```python + client = LLMClient(max_cost_per_call=0.25, max_session_cost=10.00) + ``` + + Also settable per-deployment without code changes, via + `BLOCKRUN_MAX_COST_PER_CALL` and `BLOCKRUN_MAX_SESSION_COST`. An explicit + argument wins over the env var; a malformed env value is ignored rather than + raising, so a bad deploy variable cannot brick every client. + + The refusal happens **before the paid request is sent**, so nothing settles + and nothing is charged — signing alone moves no money, the gateway submitting + the signed authorization does. The new `SpendLimitError` carries `quoted_usd`, + `limit_usd` and `scope` (`"call"` or `"session"`), and subclasses + `PaymentError` so existing handlers keep working and the model fallback chain + refuses it rather than shopping for a cheaper model. + + Both limits are **opt-in and unset by default**, so nothing changes for + existing callers. Until now the SDK had no ceiling anywhere: it computed + `cost_usd` and signed the quote in the next statement, with nothing compared + against anything — while `chat_completion` documented a + `PaymentError: If budget is set and would be exceeded` for a `budget` + parameter that did not exist. That docstring is now true. + +## 1.8.2 — 2026-07-21 + +Supersedes 1.8.1, which was published from a tree where `VERSION` and +`__init__.py` still read 1.8.0. That wheel reports `__version__ == "1.8.0"` and +cannot be corrected in place, since PyPI does not allow overwriting a published +file. Install 1.8.2 to get a package whose self-reported version is truthful. + +### Security +- **A failed paid request no longer triggers a second payment, on either + chain.** Any error raised after the `PAYMENT-SIGNATURE` went out is now + refused for model fallback, for both Base (`_should_fallback`) and Solana + (`_should_fallback_solana`). Previously a post-settlement failure was + indistinguishable from a transient one, so the chain advanced to the next + model and signed again — `smart_chat` on the premium complex tier could + settle six payments and return no tokens, the "CHARGED BUT REQUEST FAILED" + outcome this file already documents under 1.7.1. This covers every error + escaping the paid leg, not + only timeouts: the dominant post-settlement failure is a paid 5xx, which + arrives as `APIError(503)` — exactly a status the fallback logic treats as + retriable. Callers catching `httpx.TimeoutException` are unaffected; the + marker is an attribute, not a new exception type. +- **The clamp warning cannot break the request it warns about.** Its parse ran + on `resource.description`, a server-controlled string, immediately before + signing. A non-string value raised `TypeError` and aborted the call, and the + pattern backtracked super-linearly on a long digit run (measured on CPython + 3.13: 0.49s at 8k digits, 1.95s at 16k, and it keeps squaring). The number is + now matched by a bounded pattern against a + length-capped slice, an ambiguous description (a per-unit rate alongside the + ceiling) stays silent instead of naming the wrong number, and the whole helper + swallows its own failures. +- **Removed a documented payment guard that does not exist.** + `chat_completion()` advertised `PaymentError: If budget is set and would be + exceeded`. There is no `budget` parameter anywhere in the SDK and no + client-side spend cap — every 402 quote is signed automatically. The + docstring now says that plainly and points at `get_spending()`. + +### Changed +- **`max_tokens` above a model's ceiling is no longer silently absorbed.** The + gateway does not reject an over-ceiling value; it clamps to the model's own + ceiling and quotes payment for the clamped value. Verified against the live + 402 leg on 2026-07-21: `claude-opus-4.8` sent 262144 and 1000000 both quote + the 128000 price, and `gpt-5.2` sent 1e12 returns a quote rather than a 400. + The 402 disclosed the clamp in `resource.description` and the SDK discarded + it while signing. Callers now get a warning naming what they asked for and + what they are being charged for, before the signature goes out. +- The comment and `ValueError` around `MAX_TOKENS_SANITY_LIMIT` claimed the + gateway "enforces the real per-model ceiling and reports it". It does not. + Both now describe clamping. + +### Fixed +- **`max_tokens` is validated on Solana too.** `validate_max_tokens` was called + from the Base client and nowhere else; the Solana client put the caller's + value straight into paid request bodies at all four chat entry points, so + `max_tokens=2_000_000` raised on Base and was signed and sent on Solana. With + the gateway clamping rather than rejecting, nothing on either side caught it. +- **`bool` no longer passes as a number.** `bool` is an `int` subclass, so + `isinstance(True, int)` is `True` and `max_tokens=True`, `temperature=True` + and `top_p=False` all sailed through and reached the wire as JSON `true` / + `false`. All three numeric validators now reject it, naming `bool` rather + than saying "must be a number" — which reads as wrong to anyone who knows + `bool` is one. +- Line endings are normalized repo-wide via `.gitattributes` (`* text=auto`). + 17 tracked files were CRLF against an otherwise-LF tree, which turned a + 51-line change into a 1045-line diff in #27. + +## 1.8.1 — 2026-07-21 + +### Fixed +- **`max_tokens` no longer capped below what models actually serve.** The SDK + rejected anything over 100000 client-side. That was not a model limit and not + the gateway's — it was an undocumented sanity check that quietly became the + binding constraint on every caller. Asking `zai/glm-5.2` for the 262144 it + advertises raised a `ValueError` that never reached the network and named a + limit no provider had set. The bound is now `MAX_TOKENS_SANITY_LIMIT` + (1000000), a typo guard for obviously-wrong values (a byte count, a + timestamp, a stray `1e9`) rather than a ceiling any real request can hit. +- **The rejection message no longer reads like a provider response.** + `"max_tokens too large (maximum: 100000)"` was taken for an upstream model + ceiling during an investigation and recorded as one, on the strength of 19 + identical "rejections" that never left the process. The message now says the + number is the SDK's own. + +## 1.8.0 — 2026-07-18 + +### Added +- **Adopt a wallet you already own, deliberately.** `list_discovered_wallets()` + shows wallets belonging to other applications on your system, and + `import_wallet(address)` makes one of them the active BlockRun wallet. Automatic + selection still never adopts a discovered wallet — this is the opt-in path for + users whose funds live in a wallet another tool created. Solana counterparts: + `list_discovered_solana_wallets()` / `import_solana_wallet(address)`. +- **Adopting backs up the wallet it replaces.** The outgoing key is written to + `~/.blockrun/.session.backup-` (mode 0600) before being overwritten, + so switching wallets can never strand funds in the old one. +- **Matching is on the derived address, never the file's claim.** + `list_discovered_wallets()` returns no private keys, and `import_wallet()` + resolves each candidate's address from its key. A wallet file naming an address + it cannot sign for can neither be displayed as that address nor adopted by it. + +## 1.7.2 — 2026-07-18 + +### Security +- **Keep the canonical BlockRun wallet authoritative.** Automatic wallet + resolution no longer adopts a newer `wallet.json` or `solana-wallet.json` + found in another application's dot-directory. The SDK now uses the user's + own `~/.blockrun/.session` or `~/.blockrun/.solana-session`; provider-wallet + discovery remains available only for an explicit, user-confirmed migration. +- **Migration notice on first run after the lockdown.** When a new wallet is + created and other providers' wallets exist on the system, the SDK now names + those addresses and explains how to import one deliberately, instead of + silently leaving the user on an empty wallet. Addresses are derived from the + discovered key, so a wallet file claiming an address it cannot sign for + cannot trick you into funding it. + +### Fixed +- **`get_or_create_wallet()` again honours the legacy `~/.blockrun/wallet.key`.** + It resolved only `.session`, so a user holding the legacy file was issued a + brand new wallet and lost sight of their funds. It now delegates to + `load_wallet()`, matching the TypeScript SDK. + +## 1.7.1 — 2026-07-16 + +### Fixed + +- **The settlement header was read under a name no gateway sends.** Both + gateways emit `PAYMENT-RESPONSE` (the x402 v2 spec name) — `blockrun` at 36 + call sites, `blockrun-sol` at 25 — and neither emits `X-PAYMENT-RESPONSE` + even once. The SDK read only the legacy name, in four hand-rolled places, so + `_last_settlement` decoded nothing against production: no tx hash, no + settlement on any paid call. The sidecar hit this exact bug and fixed it in + blockrun-litellm 0.6.0, live-verified against a real paid call; the SDK half + was never done. Both names now go through one helper + (`tx_log.read_settlement_header`) so they can't drift apart again. The legacy + name stays accepted for other facilitators. + +- **The paid-request error no longer claims your money is gone.** + `"API error after payment"` reads as *funds are lost*, which is usually false + — a real image-edit 500 was reported as lost USDC by two readers before + anyone checked the gateway. It now reports only what the settlement header + proves: a tx hash means SETTLED and is named; absence means unknown. + + Absence is **not** reported as "payment likely not taken", which the first cut + of this change did. That trades a false alarm for a false all-clear, and the + all-clear lands on precisely the wrong requests: Solana's paid chat path + settles *in parallel* with the upstream call and re-raises immediately + (`logChargedButFailed(...); throw primaryError`), so a request the gateway + logs as `CHARGED BUT REQUEST FAILED — refund manually` answers *before* + settlement lands, and therefore carries no header at all. Absence and "you + were charged" co-occur systematically on the one path where it costs money. + Base settles after the upstream call and does match the optimistic reading, + but a set of headers doesn't tell the SDK which gateway produced it. So the + wording names the usual case without asserting it, and points at wallet + history. + + Gated on `tx_hash`, never the header's `success` field: the gateways hard-code + `success: true` even when settle didn't land, so older clients don't surface a + spurious error. A tx hash is the only field that means money moved — the same + field the gateways gate their own revenue accounting on. + +## 1.7.0 — 2026-07-15 + +### Added +- **`input_type` on video generation** (`VideoClient.generate`, `SolanaLLMClient.video`, + `AsyncSolanaLLMClient.video`). Declares the intended seed mode — `text` / + `image` / `first_last_frame` / `reference`. The gateway infers the mode from + the seed fields and rejects with 400 **before charging** when the declared + value disagrees, turning an expensive silent failure into a loud one: a + dropped `image_url` otherwise yields a text-to-video clip you still pay for. + Accepted on both chains. +- **`quality` on Solana image generation + editing** (`SolanaLLMClient.image` / + `image_edit`, sync and async). `low` / `medium` / `high` / `auto` for + `openai/gpt-image-*`; `low` meaningfully cuts generation time. + + **Solana only, by design.** The Base gateway defines no `quality` field and + strips unknown keys, so a value sent there would be silently dropped — + `ImageClient.generate`/`edit` therefore keep rejecting it, now with a hint + pointing at the Solana client. + +### Notes +- Reference-to-video (`reference_videos` / `reference_audios`) is **not** exposed. + Both gateways gate it behind `R2V_ENABLED`, which is currently off, so every + call would return 503. It slots in once that flips. +- Validation covers spelling only. Whether a declared mode matches the seed + fields, and which models accept `quality`, stay the gateway's call — it + answers both before billing, so a second copy here would only drift. + +## 1.6.1 — 2026-07-15 + +### Fixed +- Fail fast when the payer has no USDC token account (#23). Below this an + unfunded wallet burned all 5 payment retries, each costing the gateway 4 + verify retries — 20 facilitator calls per doomed request. + +## 1.6.0 — 2026-07-08 + +### Added +- Attach the BlockRun builder-code service code to Base-chain x402 payments (#21). + +## 1.5.1 — 2026-07-08 + +### Fixed +- Keep Solana video settlement blockhash fresh via proactive re-sign (#22). + Seedance 2.0 jobs could run long enough to exhaust the older two-retry + settlement loop and surface `transaction_simulation_failed`. + +## 1.5.0 — 2026-07-06 + +### Added +- Solana media surface: video / music / speech / portrait / realface / price / + rpc (#16), plus the `rpc_batch` cache fix (#17), a `solana<0.40` pin (#18), + and media hardening (#19). + +## 1.4.7 — 2026-06-26 + +### Added +- **`ChatCompletionChunk.cost_usd` on streamed calls.** The streaming paths now + attach the real per-call x402 charge to every chunk (`_iter_and_archive` / + `_aiter_and_archive`, Base + Solana), the streaming analogue of + `ChatResponse.cost_usd`. It rides on the per-call chunk object, so it's + **race-free** under shared-client concurrency (unlike `client._last_call_cost`, + which goes stale). Downstream consumers (e.g. `blockrun-litellm`) can report + the actual wallet deduction on streamed calls instead of a token×list-price + estimate. Free / 200-first streams skip the signer and carry no `cost_usd`. + +## 1.4.6 — 2026-06-24 + +### Added +- **`ChatResponse.cost_usd` and `ChatResponse.settlement`** (#11, #12). Every + chat completion now carries the **real per-call x402 charge** (and the decoded + on-chain settlement receipt when present), so downstream consumers (e.g. + `blockrun-litellm`) can report the actual wallet deduction instead of a + token×list-price estimate. The cost is attached **race-free** (set on the + response object itself, not read back off the shared client); the free / + 200-first path reports exactly `0.0` (never a stale prior charge). + +## 1.4.4 — 2026-06-18 + +### Added +- **`zai/glm-5.2` — Z.AI's newest flagship.** 1M-token context, top + open-source on long-horizon coding, billed per-token at $1.40/$4.40 (same + as glm-5.1). Added to the README ZAI table (as the new flagship) and to the + chat-model sweep (including the reasoning set). Available now via direct + call; SmartChat sees it live in `/v1/models`. + +### Changed +- **SmartChat/Eco SIMPLE tier now routes to `moonshot/kimi-k2.7`.** Moonshot's + current flagship (256K context, image+video input, `reasoning_content`) and + the only k2 still visible in `/v1/models` — k2.6 and k2.5 are now + `hidden:true`, so pinning the primary to either would silently degrade the + tier. k2.6 retained as the documented previous-gen fallback. + +## 1.4.3 — 2026-06-16 + +### Fixed +- **Clear error when a Solana key is passed to the Base (EVM) client.** Feeding + a base58 Solana secret key into `LLMClient` / `setup_agent_wallet()` (or any + EVM-chain client) used to fail with the cryptic `Private key must be 66 + characters (0x + 64 hexadecimal characters)`. The SDK now detects the base58 + Solana key shape and raises an actionable error pointing to `SolanaLLMClient` + / `setup_agent_solana_wallet()` and the `[solana]` extra. Valid 64-hex EVM + keys (including malformed ones) are unaffected and still get the hex error. + +## 1.4.2 — 2026-06-14 + +### Fixed +- **Solana clients auto-load the on-disk wallet (parity with Base).** + `SolanaLLMClient` / `AsyncSolanaLLMClient` now resolve the key as + `private_key` → `SOLANA_WALLET_KEY` → on-disk wallet (newest + `~/./solana-wallet.json`, else `~/.blockrun/.solana-session`), + so `SOLANA_WALLET_KEY` is no longer required when a wallet session exists — + matching the Base `LLMClient.load_wallet()` fallback. A malformed key from any + source now raises a clean `ValueError` (instead of a raw base58/solders + exception), and an unreadable session file is treated as "no wallet" rather + than crashing. + +## 1.4.1 — 2026-06-14 + +### Fixed +- **Streamed tool calls no longer crash the SDK** (`'dict' object has no + attribute 'delta'`). OpenAI streams tool calls incrementally — the first frame + carries `id` + `function.name`, later frames only `function.arguments` + fragments — which the strict non-stream `ToolCall` schema rejected, forcing a + `model_construct` fallback that left `choices` as raw dicts and crashed the + stream-archiving loop. Added lenient `ChatChunkToolCall` / + `ChatChunkFunctionCall` types (all fields optional) for the streaming + `delta.tool_calls`, and hardened the four sync/async archive loops + (`client.py`, `solana_client.py`) with dict-tolerant accessors so any future + `model_construct` fallback can't crash the stream. Affects `LLMClient` and + `SolanaLLMClient`, sync and async. + +## 1.4.0 — 2026-06-11 + +### Added +- **`LLMClient.onramp(address)` — Coinbase Onramp (FREE).** Mints a one-time + `pay.coinbase.com` link to fund a wallet with fiat (card/bank, 60+ currencies + → Base USDC). POSTs `{address, network: "base", asset: "USDC"}` to + `/v1/onramp/token`. The x402 signature only authenticates the wallet, so the + funding address must equal the signing wallet — pass + `client.get_wallet_address()`. The returned URL is single-use and expires in + ~5 min, so mint it at click time and never cache it. Base / USDC only; + the address is validated against `^0x[0-9a-fA-F]{40}$` and a non-Coinbase URL + raises `APIError("gateway returned no onramp url")`. Not added to the Solana + client (Base-only). Adds `validation.validate_eth_address`. + +### Docs +- **README payment section rewritten** into an explicit two-phase money flow: + Phase 1 fund your wallet once (buy via `onramp()`, transfer Base USDC, or skip + with free NVIDIA models — `get_balance()` to check); Phase 2 every request pays + itself via automatic x402. Plus per-call pay-as-you-go costs, spend tracking + (`get_spending()` / `blockrun_llm.billing`), BaseScan settlement verification, + and the non-custodial key-never-leaves-your-machine guarantee. + +## 1.3.0 — 2026-06-11 + +### Changed +- **Video poll budget default raised 5min → 15min** + (`DEFAULT_GENERATE_BUDGET_SECONDS = 900`). Generation itself is 1-3min, but + the upstream pipeline can lag the status read-path several minutes behind + actual completion (observed 2026-06-11: video done in 100s, status flipped + ~7.5min later). Jobs stay claimable ~48h, so a patient default beats a + premature give-up. Override per call with `budget_seconds`. + +### Added +- **Automatic mid-poll re-signing.** The x402 authorization window is 600s; on + budgets longer than that a poll eventually 402s. The client now fetches a + fresh challenge from the same poll_url and re-signs with the same wallet + (the gateway enforces wallet binding, not signature equality), capped at 2 + re-signs — a fresh signature that 402s again raises `PaymentError`. +- **Recoverable timeouts.** The budget-exhausted `APIError` now carries + `poll_url` in its details and explains that the job stays claimable for + ~48h — re-GET the poll_url with a fresh same-wallet signature to fetch + (and settle) the finished video. A client timeout is no longer a dead end. + +## 1.2.3 — 2026-06-08 + +### Added +- **`AsyncSolanaLLMClient` now has `image`, `image_edit`, and `get_balance`.** + This completes async-Solana public-method parity with the sync + `SolanaLLMClient` and the async EVM client. `image`/`image_edit` are backed by + a new async `_request_image_with_payment` that handles the gateway's async + `202 + poll` slow path (gpt-image-2, dall-e-3, nano-banana-pro 4K) — signing + once and polling until completion, settling only on the completed poll. + `get_balance` runs the synchronous Solana RPC read in a worker thread + (`asyncio.to_thread`) so it doesn't block the event loop. + +## 1.2.2 — 2026-06-08 + +### Fixed +- **Video poll: terminal success is keyed on `status == "completed"`, not a + literal HTTP 200** (parity with the Go 0.16.2 / TS 3.2.3 fixes). A + completed-but-non-200 poll no longer spins to the budget deadline and raises + "did not complete / no payment taken" for a job the caller was already + charged for. + +## 1.2.1 — 2026-06-08 + +### Added +- **`AsyncSolanaLLMClient.search(...)`** — async standalone search (Grok Live + Search) parity with the sync `SolanaLLMClient` and the async EVM client. Thin + wrapper over the async raw payment helper; same signature + (`query`, `sources`, `max_results`, `from_date`, `to_date`, `timeout`). + +## 1.2.0 — 2026-06-08 + +### Added +- **`AsyncSolanaLLMClient` passthrough parity.** The async Solana client now + mirrors the sync `SolanaLLMClient` (and `AsyncLLMClient`) for the data + passthroughs it previously lacked: prediction markets (`pm` + all `pm_*`), + Exa web search (`exa`, `exa_search`, `exa_find_similar`, `exa_contents`, + `exa_answer`), DefiLlama (`defi` + `defi_*`), 0x DEX (`dex` + `dex_*`), and + Modal sandboxes (`modal` + `modal_sandbox_*`). Added the async raw request + helpers (`_request_with_payment_raw` / `_get_with_payment_raw`) these build + on, with Solana x402 signing, caching, and settlement capture. +- **`VideoClient.generate_from_content(content, …)`** — submits a standard + Seedance `content[]` body to the gateway's `POST /v1/videos` endpoint + (validates unsupported inputs before charging, then delegates to the same + x402 submit+poll pipeline as `generate`). For migrating existing + `content[]`-shaped payloads unchanged; most callers should still prefer + `generate(...)` with structured kwargs. + +## 1.1.0 — 2026-06-07 + +### Added +- **DefiLlama passthrough (`/v1/defillama/*`, live since 2026-05-02 — coverage + backfill).** `defi(path, **params)` plus typed conveniences + `defi_protocols` / `defi_protocol(slug)` / `defi_chains` / `defi_yields` / + `defi_prices(coins)` on `LLMClient`, `AsyncLLMClient` and `SolanaLLMClient`. + $0.005/call ($0.001 for prices). +- **0x DEX passthrough (`/v1/zerox/*`, live since 2026-05-02 — coverage + backfill).** Free (no x402; BlockRun monetizes via on-chain affiliate fee): + `dex(path, ...)` + `dex_price` / `dex_quote` / `dex_gasless_price` / + `dex_gasless_quote` / `dex_gasless_submit` / `dex_gasless_status` / + `dex_chains` / `dex_gasless_chains` on all three clients. +- **Modal sandbox compute (`/v1/modal/*`, live since 2026-04-09 — coverage + backfill).** `modal(path, body)` + `modal_sandbox_create` ($0.01 CPU / + $0.05 GPU) / `modal_sandbox_exec` / `modal_sandbox_status` / + `modal_sandbox_terminate` ($0.001 each) on all three clients. + +## 1.0.0 — 2026-06-07 + +### Removed (BREAKING) +- **`XClient` and the entire X/Twitter (AttentionVC) surface.** The backend + removed the AttentionVC integration on 2026-04-30; every `/v1/x/*` endpoint + has returned HTTP 404 since. Deleted: `x_client.py` (`XClient`), the 15 + `x_*` methods on `LLMClient` / `AsyncLLMClient` / `SolanaLLMClient`, and the + 18 `X*` response types (`XUser`, `XTweet`, `XSearchResponse`, ...). + `XSearchSource` (Grok Live Search `sources:["x"]`) is unrelated and stays. + If you need X/Twitter data, use Grok Live Search (`SearchClient` / + `client.search(...)` with the `x` source) instead. + +## 0.39.0 — 2026-06-07 + +### Added +- **`RpcClient` — Multi-chain JSON-RPC (40+ chains).** Mirrors the new + backend `POST /v1/rpc/{network}` (Tatum gateway passthrough, launched + 2026-06-07). Flat $0.002 per call; a JSON-RPC batch charges per element. + - `call(network, method, params)` — single JSON-RPC 2.0 call. EVM chains + speak `eth_*`; non-EVM (Solana / Bitcoin-family / NEAR / Sui / XRP + Ledger / Polkadot) speak their native JSON-RPC. + - `batch(network, requests)` — JSON-RPC batch, priced per element. + - `SUPPORTED_NETWORKS` (40 curated chains) + `NETWORK_ALIASES` (eth, arb, + op, matic, bnb, avax, sol, btc, xrp, dot, ...). Unknown well-formed slugs + fall through server-side to `{slug}-mainnet`, so new Tatum chains work + without an SDK update. + - New types: `RpcResponse` (JSON-RPC envelope + `network` / `cache_hit` / + `tx_hash` gateway metadata), `RpcError`. +- **`VideoClient.generate()` new Seedance parameters** (backend 2026-06-02): + - `last_frame_url` — first-and-last-frame interpolation: the model tweens + from `image_url` (first frame) to `last_frame_url` (final frame). + Requires `image_url` + a Seedance model. Priced as image-to-video. + - `reference_image_urls` — omni / multi-reference: up to 9 reference images + for character/style consistency (Seedance 2.0 only); cite them as + "image 1", "image 2" in the prompt. Mutually exclusive with `image_url` / + `last_frame_url` / `real_face_asset_id`. + - token360 passthroughs that were already live upstream: `aspect_ratio`, + `seed`, `watermark`, `return_last_frame`. + - Client-side validation mirrors the backend mutual-exclusion rules. + +### Changed +- **Free-tier router table rebuilt from a 2026-06-07 live sweep** (every + visible free model probed): + - `nvidia/qwen3-next-80b-a3b-thinking` hit NVIDIA end-of-life 2026-05-21 + (HTTP 410) — dropped as COMPLEX/REASONING primary. COMPLEX → + `nvidia/qwen3-coder-480b` (871ms probe); REASONING → + `nvidia/nemotron-3-nano-omni-30b-a3b-reasoning` (681ms, explicit + reasoning + vision). + - `nvidia/mistral-small-4-119b` is timing out upstream (3/3 probes >60s) — + dropped as SIMPLE primary and from all fallback chains. + - `nvidia/deepseek-v4-flash` RECOVERED from the 05-09 NIM regression + (896ms probe) — reinstated as SIMPLE primary. +- README free-model tables updated to match (qwen3-next retired, + mistral-small flagged as timing out); sweep example pruned. + +## 0.38.1 — 2026-06-06 + +### Changed +- **GLM flat-rate pricing fully retired.** Z.AI's remaining launch promos ended + 2026-06-06: `zai/glm-5` now bills per-token at $0.60/$1.92 and + `zai/glm-5-turbo` at $1.20/$4.00 (no more flat $0.001/call anywhere in the + family; glm-5.1 stays $1.40/$4.40). README ZAI section rewritten. +- **`zai/glm-5` removed from the ECO COMPLEX router fallback chain** — its slot + existed only for the flat-rate pricing; at $0.60/$1.92 the existing per-token + chain (deepseek-v4-pro $0.435/$0.87 first) is both cheaper and stronger. + +## 0.38.0 — 2026-06-05 + +### Added +- **`SpeechClient` — BlockRun Voice (ElevenLabs TTS + sound effects).** + - `generate()` (alias `speak()`) → `POST /v1/audio/speech` — OpenAI-compatible + text-to-speech. Models: `elevenlabs/flash-v2.5` (default, $0.05/1k chars), + `elevenlabs/turbo-v2.5` ($0.05/1k), `elevenlabs/multilingual-v2` ($0.10/1k), + `elevenlabs/v3` ($0.10/1k). Voice aliases (sarah, george, laura, charlie, + river, roger, callum, harry) or raw ElevenLabs voice_ids; `response_format` + mp3/opus/pcm/wav; optional `speed` 0.7–1.2. Price scales with character + count, minimum $0.001/request. + - `sound_effect()` → `POST /v1/audio/sound-effects` — cinematic sound effects + up to 22s, flat $0.05/generation (`elevenlabs/sound-effects`). + - `list_voices()` → `GET /v1/audio/voices` — free voice discovery + (rate-limited 60 req/min/IP). + - New types: `SpeechResponse`, `SpeechAudio`. +- **xAI catalog additions (resold via OpenRouter credit pool, 2026-06-04):** + `xai/grok-4.3` ($1.50/$4.00, 1M context, reasoning + vision) and + `xai/grok-build-0.1` ($1.50/$3.00, 256K, fast agentic coding). Added to the + chat sweep script and README. Older Grok chat SKUs (grok-3/4/4.1-fast + families) are now hidden from `/v1/models`; direct calls still work. + +### Changed +- **`zai/glm-5.1` launch promo ended (2026-06-05)** — now bills per-token at + $1.40/$4.40 instead of flat $0.001/call. Removed from the ECO COMPLEX router + fallback chain (it became the most expensive option there); `zai/glm-5` + (still flat $0.001/call) takes the cheap long-context fallback slot. +- **`deepseek/deepseek-v4-pro` pricing corrected to $0.435/$0.87** — DeepSeek + made the 75% launch promo the permanent list price after 2026-05-31 (README + and router comments previously said the promo would expire back to list). + +## 0.37.0 — 2026-06-01 + +### Fixed +- **Concurrent Solana payments now reach ~100% success.** Sharing one + `SolanaLLMClient` / `AsyncSolanaLLMClient` across concurrent paid requests from + a single wallet previously hit `invalid_exact_svm_payload_amount_mismatch` and + `authorization already used` (replay) rejections under load (~3-10% failures), + because the underlying x402 client is not concurrency-safe and a rejected + payment couldn't recover. Two fixes: + - A per-client signing lock (`threading.Lock` for sync, lazy `asyncio.Lock` for + async) serialises the fast nonce/signature critical section. + - A **whole-request payment retry**: a non-permanent payment rejection re-runs + the entire request with a fresh 402 probe + fresh signature (new nonce, + correct amount, current blockhash), for sync/async and streaming/non-stream + (streaming only before the first chunk, so output is never replayed). New + `_is_unrecoverable_payment_error` narrows the no-retry set to genuinely + terminal cases (no funds / bad key / denylisted). + - Verified at concurrency 10 on a shared client: opus-4.7, gemini-3.1-pro and + gpt-5.5 all went from ~69-99% to **100/100**. + +## 0.35.0 — 2026-05-31 + +### Added + +- **`response_format` (JSON mode) and `stop` sequences on chat.** The gateway + now honors both OpenAI params on `/v1/chat/completions` — natively for + OpenAI/Azure, and emulated for Anthropic/Bedrock (a raw-JSON system + instruction with code-fence stripping for `{"type": "json_object"}`; `stop` + mapped to `stop_sequences`). Threaded through `chat`, `chat_completion`, and + `chat_completion_stream` on both `LLMClient` and `SolanaLLMClient` (sync and + async). Example: `client.chat("openai/gpt-4o", "...", response_format={"type": "json_object"})`. +- **Genuine `openai/gpt-4o` and `openai/gpt-4o-mini`** documented in the README + pricing table (gpt-4o $2.50/$10.00 · 128K; gpt-4o-mini $0.15/$0.60 · 128K). + The gateway no longer substitutes gpt-5.x for these IDs. + +## 0.34.0 — 2026-05-29 + +### Fixed + +- **`SolanaLLMClient` no longer truncates long chats and slow images at 60s.** + The historical flat `DEFAULT_TIMEOUT = 60.0` applied to every method + on the mega-class — chat, image, music, search, X, exa, pyth — while + the Base SDK splits the same surface across per-use-case clients + (`LLMClient=120s`, `ImageClient=200s`, `MusicClient=210s`, + `VideoClient=360s`). Long chats with high `max_tokens`, slow image + generations, and deep search queries were silently dying inside the + SDK at 60s. Raises the flat `DEFAULT_TIMEOUT` to `120.0` (matches + Base chat) and introduces per-use-case constants + (`DEFAULT_CHAT_TIMEOUT`, `DEFAULT_IMAGE_TIMEOUT`, + `DEFAULT_SEARCH_TIMEOUT`, `DEFAULT_FAST_TIMEOUT`). Each request now + carries the timeout for its *workload* rather than the single client + default: `image()` / `image_edit()` use `DEFAULT_IMAGE_TIMEOUT` (200s), + `search()` and the `exa_*` methods use `DEFAULT_SEARCH_TIMEOUT` (300s), + and chat uses the 120s baseline — sync **and** async. Closes #7. +- **`solana_key_to_bytes()` now wraps every failure in the documented + `ValueError("Invalid Solana private key: …")`.** A bare + `except ValueError: raise` used to let modern `base58`'s raw + "Invalid character" error escape past the wrapper, so callers (and the + `test_invalid_key_raises` test) matching on the documented message + broke. All decode failures are now wrapped consistently. +- **`transaction_simulation_failed` no longer wastes 5+ minutes on + pointless retries.** Adds a `_PERMANENT_PAYMENT_PATTERNS` table + mirroring the gateway-side `blockrun-sol/src/lib/x402-solana.ts` + `PERMANENT_ERRORS` classification. `_should_fallback_solana` now + short-circuits when the exception's reason matches a permanent + pattern — even when the exception type itself is "transient" + (`httpx.Timeout`, `httpx.NetworkError`). Worst-case wall-clock for + a deterministic Solana settlement failure drops from ~5min + (3 generation attempts) to one attempt's worth. Closes #6. + +### Added + +- New module-level helpers: + - `_is_permanent_payment_error(reason: str) -> bool` — case-insensitive + substring match against the permanent classification, used by both + the streaming fallback decision and any future retry classifier so + one policy applies everywhere. + - `DEFAULT_CHAT_TIMEOUT`, `DEFAULT_IMAGE_TIMEOUT`, + `DEFAULT_SEARCH_TIMEOUT`, `DEFAULT_FAST_TIMEOUT` constants + (importable from `blockrun_llm.solana_client`) so callers can use + the same numbers as the SDK does. + +### Added + +- **Per-call `timeout=` override on every long-running public method** + (level 2 of #7) — `chat`, `chat_completion`, `chat_completion_stream`, + `image`, `image_edit`, `search`, sync and async. The kwarg wins over + the per-use-case default and the constructor value, so a single + oversized request can raise (or tighten) its own budget without + reconfiguring the client: + + ```python + client.chat_completion(model, messages, max_tokens=8192, timeout=240) + client.image("...", model="openai/gpt-image-2", timeout=300) + ``` +- **`image_timeout` / `search_timeout` constructor parameters** on both + `SolanaLLMClient` and `AsyncSolanaLLMClient` (defaulting to + `DEFAULT_IMAGE_TIMEOUT` / `DEFAULT_SEARCH_TIMEOUT`) — mirrors the + per-client tuning the Base SDK gets from separate `ImageClient` / + search-aware `LLMClient` classes. + +### Changed + +- **`SolanaLLMClient(..., timeout=)` still works**, but the + default value of the constructor parameter is now + `DEFAULT_CHAT_TIMEOUT` (120s) instead of the old 60s, and it governs + the **chat** baseline specifically; image and search read from their + own constructor parameters / constants. Callers passing an explicit + value are unaffected. + +### Notes + +- 18 Base SDK clients still emit the generic + `PaymentError("Payment was rejected. Check your wallet balance.")` — + see the v0.32.0 follow-up note. Tracked separately. + +## 0.33.0 — 2026-05-29 + +### Added +- **`anthropic/claude-opus-4.8`** ($5/$25 per M, 1M context, 128K output, + agentic coding + adaptive thinking) — Anthropic's most capable Claude. + Promoted to `PREMIUM_TIERS["COMPLEX"]` primary; opus-4.7 and opus-4.5 + retained as fallbacks. Also replaces opus-4.7 in the + `PREMIUM_TIERS["REASONING"]` fallback chain. Added to the README pricing + table and `examples/sweep_all_chat_models.py`. + +## 0.32.0 — 2026-05-28 + +### Fixed +- **Image generation 202 + poll slow path** now handled transparently in both + `ImageClient.generate()` / `.edit()` (Base) and `SolanaLLMClient.image()` / + `.image_edit()` (Solana). Slow models (`openai/gpt-image-2`, + `openai/dall-e-3`, `google/nano-banana-pro` at 4K, etc.) routinely exceed the + gateway's 30s inline window and come back as `202` + `poll_url` instead of + the finished image. The Solana path used to pass the job stub straight to + `ImageResponse(**data)` and crash with a Pydantic ValidationError ("missing + field `data`"); the Base path raised a confusing `APIError 202`. Both now + poll the same `poll_url` with the same PAYMENT-SIGNATURE on `IMAGE_POLL_INTERVAL_SECONDS` + (5s default) until `status: completed`, then return the parsed `ImageResponse`. + Settlement only happens on the completed poll, so timing out the budget + (`IMAGE_POLL_BUDGET_SECONDS`, 300s default) raises `APIError 504` and **no + payment is taken**. +- **PaymentError now preserves the gateway's real failure reason.** On a 402 + retry response, the SDK used to raise a generic + `"Payment rejected. Check your Solana USDC balance."` — losing the + facilitator's actual reason (`transaction_simulation_failed`, + `insufficient_funds`, `payment_expired`, etc.). The new + `PaymentError(message, *, status_code=..., response=...)` keyword args + carry the gateway body so callers and upstream proxies can surface the + real reason. All four `SolanaLLMClient` retry paths (sync raw, sync get, + sync stream, async post, async stream) and the Base `ImageClient` retry + use the shared `validation.build_payment_rejected_error` helper. + +### Changed +- **`PaymentError` constructor is now keyword-extended.** Existing + `PaymentError("...")` calls are unchanged. The two new optional kwargs are + `status_code: Optional[int]` and `response: Optional[dict]`. + +### Notes for sidecar / proxy authors +- `blockrun-litellm >= 0.3.9` surfaces `PaymentError.response.details` on + the 402 HTTP body. If you wrap `PaymentError` yourself, pull + `exc.response.get("details")` for the structured facilitator reason. +- Follow-up: 18 other Base SDK clients (`client.py`, `phone.py`, + `realface.py`, `surf.py`, `voice.py`, etc.) still inline the legacy + `raise PaymentError("Payment was rejected. Check your wallet balance.")` + pattern. They should migrate to `build_payment_rejected_error` in a + follow-up PR — not blocking, but customers debugging settlement + failures on those endpoints still lose context until then. + +## 0.31.0 — 2026-05-27 + +### Added +- **`google/gemini-3.5-flash`** — Google's newest-generation Flash with built-in + thinking mode: frontier-class quality at Flash speed and pricing ($0.50/M in, + $3.00/M out, 1M context). Now live in production. Added to the README model + pricing table and wired into the smart router's COMPLEX tier as the leading + fallback (ahead of `google/gemini-3-flash-preview`, which remains available). + +## 0.30.1 — 2026-05-26 + +### Changed +- **Default image-edit model is now `openai/gpt-image-2`** (was `openai/gpt-image-1`) + across `ImageClient.edit()`, `LLMClient.image_edit()` (sync + async), and + `SolanaLLMClient.image_edit()`. Matches the production `/v1/images/image2image` + schema default and aligns Python, TypeScript, and Go SDKs. Pass `model=` explicitly + to keep using the cheaper `gpt-image-1`. + +## 0.30.0 — 2026-05-26 + +### Added +- **Multi-image fusion across all edit entry points.** The `image` parameter + now accepts `Union[str, List[str]]` on `ImageClient.edit()`, + `LLMClient.image_edit()` (sync + async), and `SolanaLLMClient.image_edit()` + — pass a single base64 `data:image/...` data URI to edit one image, or a list + of 2–4 URIs to fuse them (e.g. a subject photo + a brand logo). Matches the + now-live `/v1/images/image2image` contract, which previously rejected arrays + with `400 "expected string, received array"`. Single-string calls are + unchanged and fully backward compatible. Fusion caps mirror the server: + `openai/*` up to 4 source images, `google/*` (Nano Banana) up to 3; a `mask` + cannot be combined with multiple source images. + +### Fixed +- Documented the full set of edit-capable models (`openai/gpt-image-1`, + `openai/gpt-image-2`, `google/nano-banana`, `google/nano-banana-pro`) and + corrected the `edit()`/`image_edit()` docs, which incorrectly claimed a plain + URL was accepted — the route requires a base64 `data:image/...` data URI. + +## 0.29.0 — 2026-05-25 + +### Added +- **`RealFaceClient` — real-person face enrollment via x402.** RealFace + registers a *real person's* likeness (vs. `PortraitClient`, which is for + AI-generated characters). The asset works exactly like a Virtual Portrait + on Seedance 2.0 / 2.0-fast — both return a `ta_xxxxxxxx` id you pass as + `real_face_asset_id` on `VideoClient.generate()` — but enrollment proves + the rights-holder is the person in the photo via a brief on-phone liveness + check. **No KYC.** Three-step flow: + - `init(name)` — *free*, rate-limited. Returns a `group_id` + an `h5_link` + the real person scans on their phone. + - `status(group_id)` / `wait_for_active(group_id)` — *free*. Poll until the + person finishes the liveness check. + - `enroll(name, image_url, group_id)` — **$0.01 USDC**, one-time. Settles + only after the face matches the live capture, so `425` (group not active), + `422` (face mismatch), and `502` (upstream failure) return errors with no + charge. + + Plus `list_realfaces()` over the free `GET /v1/wallet/
/realfaces` + endpoint. + + ```python + from blockrun_llm import RealFaceClient + faces = RealFaceClient() + init = faces.init(name="Jane — spokesperson") # show init.h5_link as a QR + faces.wait_for_active(init.group_id) # they do the phone check + rf = faces.enroll(name="Jane — spokesperson", + image_url="https://example.com/jane.jpg", + group_id=init.group_id) + print(rf.asset_id) # ta_… → pass as real_face_asset_id on Seedance 2.0 + ``` + +- **`RealFaceInit`, `RealFaceStatus`, `RealFaceEnrollment`, `RealFaceList`, + `RealFaceListItem`** exported from the package root. + +### Changed +- **Reversed the v0.28.1 "real-person video is unsupported" stance.** + Real-person likeness is now supported through the no-KYC RealFace liveness + flow above (KYC is no longer required). The `VideoClient` class/parameter + docstrings, the `real_face_asset_id` validator message, and the README now + describe `real_face_asset_id` as accepting **either** a Virtual Portrait + (`PortraitClient`, $0.01) **or** a RealFace (`RealFaceClient`, $0.01). No + wire-format change — both still pass the same `ta_` id. `seedance-1.5-pro` + does not support either asset type. + +## 0.28.1 — 2026-05-23 + +### Added +- **`PortraitClient` — Virtual Portrait enrollment via x402.** Wraps + `POST /v1/portrait/enroll` ($0.01 USDC, one-time, no KYC) and the + free `GET /v1/wallet/
/portraits` listing endpoint. Enroll an + AI character image, get back a `ta_xxxxxxxx` asset id, then reuse it + as `real_face_asset_id` on `VideoClient.generate()` for Seedance 2.0 / + 2.0-fast to keep the same character across multiple videos. Settlement + is held until upstream registration succeeds, so failed enrollments + (content filter, image too large) return 502 with no charge. + + ```python + from blockrun_llm import PortraitClient + p = PortraitClient().enroll( + name="My Spokesperson", + image_url="https://example.com/character.jpg", + ) + print(p.asset_id) # ta_abcdef1234567890 + print(p.settlement.tx_hash) # 0x9f3a… + ``` + +- **`PortraitEnrollment`, `PortraitUsage`, `PortraitSettlement`, + `PortraitList`, `PortraitListItem`** exported from the package root. + +### Changed +- **`VideoClient` Seedance docs realigned with the (then-)dropped + RealFace path.** _(Reversed in 0.29.0 — real-person video is now + supported via the no-KYC RealFace liveness flow.)_ At the time, the + `VideoClient` class docstring, the `real_face_asset_id` parameter + docstring, the validator error message, and the README example were + changed to describe `real_face_asset_id` exclusively as a Virtual + Portrait (`POST /v1/portrait/enroll`, $0.01, no KYC). No behavior + change — the wire format (the `ta_` id) is unchanged. + +## 0.28.0 — 2026-05-22 + +### Added +- **`VideoClient.generate()` — face-reference, resolution, and audio + controls** to align with the documented `/v1/videos/generations` schema: + - `real_face_asset_id="ta_xxxxxx"` — condition Seedance 2.0 fast/pro on + a Virtual Portrait or Token360 RealFace asset. Validates the `ta_` + prefix and is mutually exclusive with `image_url`. + - `resolution="360p" | "480p" | "720p" | "1080p" | "4K"` — drop to 480p + for ~half the per-clip Seedance cost; bump to 1080p / 4K for higher + fidelity. Grok ignores this field. + - `generate_audio=True/False` — override Seedance's default (audio on + for text-to-video, off for image- or face-conditioned). Grok ignores. + +### Changed +- Refreshed Seedance pricing in the `VideoClient` docstring and README + to match the live per-M-token billing (token360 charges by tokens at + ~20,256 tok/sec at 720p), replacing the old per-second figures: + - `bytedance/seedance-1.5-pro` — $4.32/M (flat) ≈ $0.46 / 5s 720p + - `bytedance/seedance-2.0-fast` — $11.20/M text · $6.60/M image + - `bytedance/seedance-2.0` — $14.00/M text · $8.60/M image + - `xai/grok-imagine-video` unchanged at $0.050/sec. + +## 0.27.0 — 2026-05-22 + +### Added +- **Opt-in per-transaction log to a project-local folder.** Pass + `transaction_log=True` to `LLMClient`, `AsyncLLMClient`, `SolanaLLMClient`, + or `AsyncSolanaLLMClient` (or set `BLOCKRUN_TX_LOG=1`) and every paid call + appends one plain-text row to `./log/transactions.log`: + + ``` + 2026-05-21 15:44:46 chat anthropic/claude-sonnet-4.6 in= 3 out=4 $0.034137 0x6513d128… + ``` + + Columns: timestamp, endpoint tag, model (left-padded 30), prompt/completion + tokens, USD cost (6 decimals), and the first 10 chars of the on-chain + settlement hash (Base tx hash or Solana signature). The hash is decoded + from the `X-PAYMENT-RESPONSE` header the facilitator returns after + settlement, so each row is verifiable against BaseScan / Solscan with one + click — the row matches what hit the ledger. + + Pass a string/Path instead of `True` to choose a different directory. + Disabled by default; no impact on the existing `~/.blockrun/cache`, + `~/.blockrun/data/`, or `~/.blockrun/cost_log.jsonl` layers — this lives + in its own folder next to your code. + +- **`TransactionLogger`, `decode_settlement_header`, `format_row`** are + exported from the package root for callers who want to build their own + reconciliation tooling on top of the same primitives. + +## 0.26.0 — 2026-05-18 + +### Added +- **`PhoneClient` — Twilio-backed phone lookup + number provisioning via x402.** + New module `blockrun_llm/phone.py` wraps the backend's `/v1/phone/*` partner + endpoints. Methods: + - `lookup(phone_number)` — carrier + line-type ($0.01) + - `lookup_fraud(phone_number)` — adds SIM-swap / call-forwarding signals ($0.05) + - `buy_number(country="US", area_code=None)` — provision a US/CA number with a + 30-day lease bound to your wallet ($5.00). Settlement is held until Twilio + confirms the purchase, so failed buys never charge your wallet. + - `renew_number(phone_number)` — extend by 30 days ($5.00) + - `list_numbers()` — list your active numbers ($0.001) + - `release_number(phone_number)` — return a number to the pool (free, still + flows through x402 for wallet-identity verification) + Use the provisioned number as the `from_` caller ID in `VoiceClient.call()`. + +- **`SurfClient` — asksurf.ai crypto-data gateway via x402.** New module + `blockrun_llm/surf.py` wraps `/v1/surf/*` and exposes ~83 endpoints covering + exchange data, on-chain SQL, prediction markets (Polymarket + Kalshi), + wallet/social analytics, and project intelligence. Tiered pricing matches + the backend: tier 1 / 2 / 3 → $0.001 / $0.005 / $0.020. API: + - `SurfClient.endpoints()` — full discovery catalog + - `SurfClient.endpoint_info(path)` / `SurfClient.price(path)` — single-endpoint metadata + - `client.get(path, params)` / `client.post(path, body)` — direct callers + - `client.call(path, params=…, body=…)` — auto-routes GET vs POST from the catalog + Required-param validation runs client-side before the network round trip. + +### Changed +- **`VoiceClient.call()` docs reflect new `from` resolution** on the backend: + if `from_` is omitted and your wallet owns exactly one active number, the + backend auto-picks it; 0 owned → 403 `no_active_number`; 2+ owned → 400 + `ambiguous_from` with the candidate list in the error body. No code change + was needed — the SDK already forwarded `from_` correctly — but the docstring + was stale. + +## 0.25.0 — 2026-05-16 + +### Added +- **`VoiceClient` — AI-powered outbound phone calls via x402.** New module + `blockrun_llm/voice.py` wraps the backend's `POST /v1/voice/call` (paid, + $0.54/call) and `GET /v1/voice/call/{call_id}` (free polling). The AI agent + dials a US/Canada E.164 number and conducts a real-time conversation + following your `task` instructions; STT + LLM + TTS are handled upstream by + Bland.ai. Full pass-through for `from`, `voice` (7 presets + custom Bland + IDs), `max_duration` (1–30 min), `language`, `first_sentence`, + `wait_for_greeting`, `interruption_threshold`, and `model` tier (base / + enhanced / turbo). Status polling returns the full Bland call record + (status, transcript, recording URL, ended_reason). Exported as `VoiceClient` + from `blockrun_llm`. See README "Voice Calls" section for usage. + +## 0.24.0 — 2026-05-14 + +### Changed +- **Default Solana RPC is now BlockRun's proxy** — + `SolanaLLMClient` / `AsyncSolanaLLMClient` resolve their RPC + endpoint to ``https://sol.blockrun.ai/api/v1/solana/rpc`` when no + ``SOLANA_RPC_URL`` env var or explicit ``rpc_url`` arg is set. + This is BlockRun's own multi-region, Tatum-backed Solana JSON-RPC + proxy. It is free for anyone using the SDK — the cost is bundled + into LLM inference fees you already pay. Method-aware caching on + the server (``getLatestBlockhash`` at 30s TTL) collapses bursty + signing traffic to a handful of upstream RPC calls, so partners + no longer need to register Helius / Tatum / QuickNode for typical + loads. + + The previous default ``https://api.mainnet-beta.solana.com`` is + still reachable via ``SOLANA_RPC_URL=...`` but is no longer the + default — its public rate limit (~10-40 RPS) is too aggressive + for any real concurrency. + + No code change required to opt in: upgrade and you're using it. + To stay on a private Helius / Tatum / QuickNode RPC, set + ``SOLANA_RPC_URL`` (the 0.23.0 env-var mechanism is unchanged). + +### Deprecated +- **`XClient` (BlockRun `/v1/x/*` AttentionVC integration)** — the + backend ``/v1/x/*`` endpoints were removed on 2026-04-30. All + ``XClient`` method calls now return HTTP 404 until a replacement + X/Twitter data upstream is reintroduced. The class is kept in the + SDK so existing imports do not break; instantiation now emits a + ``DeprecationWarning`` so callers can migrate cleanly when a + replacement ships. + +## 0.23.0 — 2026-05-14 + +### New +- **Custom Solana RPC support via env vars** — Solana clients + (`SolanaLLMClient` + `AsyncSolanaLLMClient`) now resolve their RPC + endpoint from explicit args, then these env vars, then the public + default: + - ``SOLANA_RPC_URL`` — the JSON-RPC endpoint URL. Use this when + your provider embeds auth in the URL (Helius style: + ``https://mainnet.helius-rpc.com/?api-key=...``). + - ``SOLANA_RPC_API_KEY`` — convenience shortcut for the common + ``x-api-key: `` header style (Tatum, some Triton tiers). + Internally becomes ``SOLANA_RPC_HEADERS='{"x-api-key":"..."}'``. + - ``SOLANA_RPC_HEADERS`` — JSON dict for arbitrary header auth + (``'{"x-api-key":"...","x-rate-tier":"pro"}'``). + + This unblocks production traffic — the public + ``api.mainnet-beta.solana.com`` rate-limits aggressively + (~10-40 RPS) and a partner deploying behind a free-tier Helius + key was seeing failures at 30-100 concurrent requests. + + Previously the only way to switch RPCs was to edit + ``_adapter.py`` source; that change is lost on every upgrade. + Env vars make this idempotent across releases. + +- **Header-auth Solana gateways (Tatum, header-only Triton) now + work** — the upstream x402 SDK's + ``register_exact_svm_client`` only takes ``rpc_url``, not custom + headers, so the underlying ``solana.rpc.api.Client`` was always + built without ``extra_headers``. We now pre-populate the SVM + scheme's client cache with a properly-configured ``SolanaClient`` + before any payment payload is constructed. + +### Configuration example + +For Tatum (header-auth): +```bash +export SOLANA_RPC_URL=https://solana-mainnet.gateway.tatum.io +export SOLANA_RPC_API_KEY=t-... +``` + +For Helius (URL-embedded auth): +```bash +export SOLANA_RPC_URL='https://mainnet.helius-rpc.com/?api-key=...' +``` + +For arbitrary header schemes: +```bash +export SOLANA_RPC_URL=https://your.gateway/... +export SOLANA_RPC_HEADERS='{"x-api-key":"...","x-rate-tier":"pro"}' +``` + +### Verified e2e +- Live test against ``solana-mainnet.gateway.tatum.io`` with + ``x-api-key`` header — the signing pipeline (blockhash fetch + + TransferChecked tx construction + signature) completed + end-to-end through Tatum and submitted the payment to BlockRun's + gateway. (Final on-chain settlement failed for an unrelated + reason: the test wallet was empty.) + +### Notes for partners hitting RPC rate limits +- Helius free tier is 10 RPS — adequate for low QPS, not for + bursty 50-100 concurrent. Move to Helius Developer ($99/mo, + 25 RPS) or Tatum (200 RPS). +- A separate ``0.24.0`` will add client-side blockhash caching so + ~10 RPS of paid traffic resolves to <1 RPS of upstream RPC calls + — at that point Helius free becomes viable for most production + loads. Tracked separately because the change touches the x402 + scheme cache more invasively. + +## 0.22.1 — 2026-05-12 + +### Fixed +- **Tool calling on Solana.** `SolanaLLMClient.chat_completion`, + `SolanaLLMClient.chat_completion_stream`, + `AsyncSolanaLLMClient.chat_completion`, and + `AsyncSolanaLLMClient.chat_completion_stream` now accept ``tools`` / + ``tool_choice`` kwargs and forward them to the upstream model. + Previously the parameters were missing from the Solana SDK methods so + partners couldn't use function calling on the Solana chain — but the + BlockRun backend always supported the field uniformly; the SDK was + the bottleneck. + + Live-verified: ``client.chat_completion("nvidia/deepseek-v4-flash", + [...], tools=[get_weather], tool_choice="auto")`` returned + ``tool_call: get_weather('{"city": "Tokyo"}')`` against + ``sol.blockrun.ai``. + +## 0.22.0 — 2026-05-12 + +### New +- **``AsyncSolanaLLMClient``** — async counterpart of + ``SolanaLLMClient``. Mirrors the sync API for chat completions (both + non-streaming and streaming) so ``asyncio`` callers don't need to + thread-pool around blocking I/O. Built on the async ``x402Client`` + (instead of ``x402ClientSync``) + ``httpx.AsyncClient``. Public + surface for the first release: ``chat()``, ``chat_completion()``, + ``chat_completion_stream()``, ``list_models()``, ``close()`` plus + ``__aenter__`` / ``__aexit__``. Image / Exa / Predexon / Music + endpoints are still sync-only on Solana (they'll follow if there's + demand). Same retry policy and ``fallback_models`` semantics as + every other streaming client. +- **Paid streaming now writes to ``~/.blockrun/cost_log.jsonl`` and + ``~/.blockrun/data/``** — closing the audit-trail gap that 0.20.x + introduced. ``LLMClient`` (sync + async) and ``SolanaLLMClient`` + (sync + new async) all accumulate streamed content during the SSE + iteration, then call ``save_to_cache`` once ``data: [DONE]`` arrives, + building a synthetic ``chat.completion`` response so the local + archive matches the non-stream paid path one-for-one. Free models + skip the archive (``cost_usd == 0``). Failures during the stream do + not produce a partial archive row. + +### Verified e2e +- Async Solana streaming via ``AsyncSolanaLLMClient.chat_completion_stream`` + against ``sol.blockrun.ai`` with the free + ``nvidia/deepseek-v4-flash`` model: 2 content chunks, + ``"Hello! How can I"``, on the second attempt (first hit a + transient NVIDIA NIM upstream timeout that resolved itself). +- 12/12 Base streaming unit tests + 6/6 Solana streaming unit tests + still pass — the archive-on-completion change is additive and + doesn't touch the existing assertions. + +## 0.21.0 — 2026-05-12 + +### New +- **Streaming on Solana.** `SolanaLLMClient.chat_completion_stream(...)` + is now a thing, mirroring the Base `LLMClient` API one-for-one: + yields `ChatCompletionChunk` per SSE `data:` line, does the 402 → + sign-locally-with-SVM-x402 → retry-with-PAYMENT-SIGNATURE dance + before the first chunk, supports the same retry policy (5xx ×3 with + 1s/2s/4s backoff) and `fallback_models` chain walking. +- Constraint: like Base, fallback can only fire **before** the first + chunk is yielded — once any chunk has reached the caller, switching + models would concatenate two distinct responses. +- Async is not yet implemented for the Solana client (consistent with + the rest of `SolanaLLMClient` which is sync-only today). + +### Tests +- 6 new mock-based unit tests in `tests/unit/test_streaming_solana.py`: + free-model direct streaming, paid-model sign-and-retry, recovery + after 2× 503, raising after exhausted retries, fallback-chain + walking, and payment-rejected → `PaymentError`. + +### Verified e2e +- Live call against `sol.blockrun.ai` with the free + `nvidia/deepseek-v4-flash` model: 2 content chunks, content + "Silence.", 0.8s. + +## 0.20.1 — 2026-05-12 + +### Improved +- **Streaming 5xx retry policy.** `_stream_with_payment` now retries + transient upstream errors (500 / 502 / 503 / 504) up to three times per + phase with exponential backoff (1s / 2s / 4s), instead of the single + retry shipped in 0.20.0. Both the unauthenticated probe and the + paid retry honor the same policy. Tuned for NVIDIA NIM upstream + flakiness on free models — most transient hiccups now self-heal + before bubbling up to the caller. Exposed as + `LLMClient._STREAM_5XX_STATUSES` / `_STREAM_5XX_BACKOFFS` so callers + can monkey-patch the policy in tests or override at runtime. +- **`fallback_models` parameter on `chat_completion_stream`** (sync + + async). Walks the chain when the primary upstream produces a retriable + error (timeouts, network errors, 5xx after exhausting in-band retries). + **Constraint:** fallback only triggers *before the first chunk is + yielded* — once any byte has reached the caller, switching upstreams + would concatenate two distinct responses. After-first-chunk failures + propagate to the caller as before. + +### Tests +- Six new unit tests in `tests/unit/test_streaming.py` covering: + recovery after two 503s, raising after exhausting retries, retry on + the paid (post-402) retry leg, fallback to a healthy model after a + primary 503-storm, no fallback after a chunk has been yielded, and + no fallback on a non-retriable 4xx. + +## 0.20.0 — 2026-05-11 + +### New + +- **Server-Sent Events streaming for chat completions.** New methods + `LLMClient.chat_completion_stream(...)` and + `AsyncLLMClient.chat_completion_stream(...)` return an iterator of + :class:`ChatCompletionChunk` objects, yielding one chunk per SSE event + until the upstream emits `data: [DONE]`. The 402 → sign-locally → + retry flow is identical to the non-streaming path; free models + (e.g. `nvidia/deepseek-v4-flash`) stream directly without a payment + dance. New types exported: `ChatCompletionChunk`, `ChatChunkChoice`, + `ChatChunkDelta`. Validated end-to-end against the production + `blockrun.ai` gateway (sync + async, free model). Caveats: + `search_parameters` and the Responses-API models (`codex`, + `gpt-5.4-pro`) reject streaming server-side with 400 — same constraint + as the gateway. Six new unit tests cover the free path, paid 402-sign- + retry path, payment rejection, and tolerance for malformed chunks. +- **Local billing / cost-tracking surface.** Every paid call now writes a + `{ts, endpoint, cost_usd, model, wallet, network, client_kind}` row to + `~/.blockrun/cost_log.jsonl`. New helpers on top: + - `get_cost_log_summary(*, from_date, to_date, wallet, network, group_by)` + — aggregate by `endpoint` / `model` / `wallet` / `network` / + `client_kind` / `day` / `month`. + - `export_cost_log_csv(...)` and `export_cost_log_json(...)` — render + filtered per-call records, optionally to a file. + - `python -m blockrun_llm.billing summary | export {csv|json}` CLI with + `--from / --to / --wallet / --network / --group-by / --output` flags. + Older 3-field cost-log rows remain readable; `by_endpoint` is still + emitted as a backwards-compat alias when grouping by endpoint. +- **Predexon v2 typed helpers — full coverage across sync, async, Solana.** + All three clients now expose the same 17 `pm_*` methods: + - Canonical cross-venue (Tier 1): `pm_markets`, `pm_listings`, `pm_outcome` + - Polymarket (Tier 1): `pm_polymarket_markets`, `pm_polymarket_events`, + `pm_polymarket_markets_keyset`, `pm_polymarket_events_keyset`, + `pm_polymarket_positions`, `pm_polymarket_trades`, + `pm_polymarket_leaderboard` + - Kalshi / Limitless (Tier 1): `pm_kalshi_markets`, `pm_limitless_markets` + - Sports (Tier 1): `pm_sports_categories`, `pm_sports_markets` + - Wallet identity (Tier 2): `pm_wallet_identity`, `pm_wallet_identities`, + `pm_wallet_cluster` +- **`exa_*` methods on `LLMClient` (Base USDC).** `exa()`, `exa_search()`, + `exa_find_similar()`, `exa_contents()`, `exa_answer()` — same surface and + pricing as the existing `SolanaLLMClient` versions ($0.01/request for + search / find-similar / answer, $0.002/URL for contents). +- **`fallback_models=[...]` on `chat()` and `chat_completion()`** (sync + + async). On timeout, network error, or 5xx, the SDK transparently walks + the list before raising. 4xx and `PaymentError` propagate immediately. + Each fallback hop logs one line to stderr so the caller can see which + model actually served the response. +- **`smart_chat()` uses the tier's fallback chain automatically.** + `RoutingDecision` gained a `fallbacks: List[str]` field populated from + the chosen tier; `smart_chat()` plumbs it through to `chat()`. +- **`examples/sweep_all_chat_models.py`** — runnable end-to-end sweep over + every chat model the SDK exposes, with a forward-compat diff against + `/v1/models`, async smoke, budget guard, and optional JSON output. +- **`examples/sweep_all_media_models.py`** — sister script for image and + music models. Video is excluded by design (long polling, expensive). +- **New chat models in router / pricing tables:** + - `anthropic/claude-opus-4.7` ($5/$25 per M, 1M context, 128K output, + agentic coding + adaptive thinking) — promoted to + `PREMIUM_TIERS["COMPLEX"]` primary; opus-4.5 retained as fallback. + - `zai/glm-5.1` (flat $0.001/call, 200K context) — added to + `ECO_TIERS["COMPLEX"]` fallback chain for long-context work. + +### Changed + +- **`/v1/images/models` is deprecated; image models live in `/v1/models` + with `categories: ["image"]`.** `list_image_models()` (module-level, + sync, async) and `list_all_models()` now read the unified catalog with + the same return shape, so existing callers keep working without an + extra request. +- **Pricing reads aligned with the current `/v1/models` schema.** + `_get_model_pricing()` now reads nested `pricing.input` / `pricing.output` + for paid models and `pricing.flat` for flat-billed models, falling back + to the legacy top-level keys. Router cost estimates and savings % + reflect the right numbers again, and flat-billed models compete in + routing decisions on the right basis. +- **`FREE_TIERS["MEDIUM"]` primary** moved from `nvidia/deepseek-v4-flash` + to `nvidia/llama-4-maverick`; v4-flash references in `AUTO_TIERS` / + `ECO_TIERS` / `FREE_TIERS` fallback chains likewise redirected so the + safety net hits a working model when the primary is unavailable. +- **ZAI GLM-5 family pricing** corrected from per-token to flat + $0.001/call across the README pricing tables to match the catalog. +- **OpenAI dated-version responses** (e.g. `gpt-5.5-2026-04-20` for a + request to `openai/gpt-5.5`) are no longer flagged as redirects — only + base-id mismatches count. + +### Removed + +- `black-forest/flux-1.1-pro` — dropped from the README image table and + from the media-sweep target list. Not in the live catalog. + +## 0.19.0 + +- **Predexon v2 endpoints exposed via typed helpers.** All v2 endpoints went live in production on 2026-05-07 (`blockrun-web-00451-cnw`). The generic `pm()` / `pm_query()` passthrough already handled them, but agents can now discover the new shape from method names + docstrings. Ten new convenience methods on `LLMClient` — each is a thin wrapper, no breaking changes to the existing `pm()` API: + - **Canonical cross-venue (Tier 1):** `pm_markets(**filters)`, `pm_listings(**filters)`, `pm_outcome(predexon_id)`. Predexon's unified data layer with cross-venue IDs across Polymarket, Kalshi, Limitless, Opinion, Predict.Fun. + - **Polymarket keyset pagination (Tier 1):** `pm_polymarket_markets_keyset(**filters)`, `pm_polymarket_events_keyset(**filters)` — cursor-based for stable traversal of large result sets. + - **Sports markets (Tier 1):** `pm_sports_categories()`, `pm_sports_markets(**filters)`. + - **Wallet identity & clustering (Tier 2):** `pm_wallet_identity(wallet)` (GET), `pm_wallet_identities(addresses)` (POST, up to 200), `pm_wallet_cluster(address)` (GET on-chain relationship graph). +- `pm()` / `pm_query()` docstrings updated to advertise v2 examples and surface the Tier 1 / Tier 2 split inline. + +## 0.18.0 + +- **DeepSeek V4 family in paid catalog.** Backend added `deepseek/deepseek-v4-pro` (1.6T MoE / 49B active, 1M context — strongest open-weight reasoner; MMLU-Pro 87.5, GPQA 90.1, SWE-bench 80.6, LiveCodeBench 93.5; **$0.50 in / $1.00 out per 1M under the 75% promo through 2026-05-31**, list $2.00/$4.00). The legacy `deepseek/deepseek-chat` and `deepseek/deepseek-reasoner` IDs are now V4 Flash non-thinking / thinking modes — repriced to **$0.20 in / $0.40 out per 1M, 1M context** (was $0.28/$0.42, 128K). Same upstream as `nvidia/deepseek-v4-flash` but on the paid endpoint with higher reliability and 5MB request bodies. +- **Smart router: free tier primaries repointed to visible models.** `FREE_TIERS["SIMPLE"]` was pinned to `nvidia/gpt-oss-120b` (now `hidden: true` in catalog — privacy-delisted from `/v1/models` though `available: true` for direct callers) and `FREE_TIERS["MEDIUM"]` to `nvidia/deepseek-v3.2` (hidden — NVIDIA NIM hung, backend redirects to v4-flash). Both are absent from `/v1/models`, so Python's pricing dict (built from that endpoint) could not resolve them and SmartChat silently fell through. Repointed primaries to visible IDs: `SIMPLE` → `nvidia/mistral-small-4-119b`, `MEDIUM` → `nvidia/deepseek-v4-flash`. Direct calls by full ID (`client.chat("nvidia/gpt-oss-120b", ...)`) still work — only auto-routing changed. +- **Smart router: V4 Pro promoted into reasoning fallbacks.** `AUTO_TIERS["REASONING"]` and `ECO_TIERS["REASONING"]` now list `deepseek/deepseek-v4-pro` as the first fallback after `deepseek-reasoner` (V4 Flash thinking stays primary because it's cheaper). `ECO_TIERS["COMPLEX"]` adds V4 Pro to fallbacks for harder reasoning tasks. +- README refresh: DeepSeek pricing table shows V4 Pro / V4 Flash chat / V4 Flash reasoner with correct prices and 1M context. NVIDIA free table notes that `gpt-oss-120b/20b` are hidden from `/v1/models` but still callable by direct ID (re-enabled 2026-04-30 after a brief privacy delisting). +- **`XClient` deprecated.** BlockRun's `/v1/x/*` (AttentionVC-partnered) integration was removed from the backend on 2026-04-30 (commit 80dcf52). The class is kept in the SDK so existing imports do not break, but instantiation now emits a `DeprecationWarning` — all calls return HTTP 404 until a replacement upstream is wired up. +- **DeepSeek V4 thinking + tool-call multi-turn now works.** Backend commit `f8a2d44` (2026-05-03) preserves `reasoning_content` on assistant messages with `tool_calls` for DeepSeek V4 thinking-mode (`deepseek-reasoner` / `deepseek-v4-pro`) — previously the streaming `/v1/messages` path stripped it, causing upstream 400 "reasoning_content in the thinking mode must be passed back" on tool-using multi-turn sessions, which the route then mis-classified as transient 503 → 5 retries with backoff on a deterministic failure. SDK `ChatMessage` already carried `reasoning_content` and `thinking` fields, so the fix is purely server-side; this entry exists so users seeing past failures know they're resolved. + +## 0.17.1 + +- **Smart router: AUTO/ECO `SIMPLE` primaries promoted from `moonshot/kimi-k2.5` → `moonshot/kimi-k2.6`** (Moonshot's flagship — 256K context, vision + `reasoning_content`, $0.95 in / $4.00 out per 1M). The catalog now hides `kimi-k2.5` as superseded, so it no longer appears in `/v1/models` and the SDK could not resolve its pricing — routing was silently falling through to the next fallback. `kimi-k2.5` retained as the first fallback for clients explicitly pinned to its pricing. +- Doc refresh: README Smart Routing example output and SIMPLE tier table now reference `moonshot/kimi-k2.6`. + +## 0.17.0 + +- **New flagship model: `openai/gpt-5.5`** (released 2026-04-23, first fully retrained base since GPT-4.5). 1M context, 128K output, native agent + computer use. Pricing $5.00 / $30.00 per 1M tokens. +- **Smart router: `PREMIUM_TIERS["MEDIUM"]` now points at `openai/gpt-5.5`**; `gpt-5.4` demoted to first fallback. The cost-savings baseline in `estimate_cost` was rebased from GPT-5.4 ($2.50/$15) to GPT-5.5 ($5.00/$30) so reported savings stay meaningful against the current flagship. +- Doc-example refresh: `AnthropicClient` cross-provider example and `examples/arbitrage_analyzer.py` `frontier` tier now reference `openai/gpt-5.5`. +- Reconciles `__version__` and `VERSION` (previously drifted at 0.16.1 vs 0.15.0); both now 0.17.0. + +## 0.16.1 + +- **`ImageClient` default timeout 120s → 200s.** The gateway's per-call OpenAI + timeout for `gpt-image-2` was bumped to 180s server-side (it routinely takes + ~120-180s at 1536x1024 and larger), so the SDK's old 120s default was cutting + the request before the server had a chance to return. New default leaves + ~20s of buffer above the server cap. Existing users passing an explicit + `timeout=` are unaffected. + +## 0.16.0 + +- **VideoClient switches to async submit+poll**. Upstream `/v1/videos/generations` + moved from sync to async on 2026-04-23 (submit returns a job id; client polls + until completion). Public signature of `VideoClient.generate(...)` is unchanged + — still blocks until the video is ready and returns `VideoResponse` with the + MP4 URL and tx hash. Internally the client now signs once, submits, and + replays the same signature on GET polls every 5s until upstream completes. + Settlement only fires on the first completed poll, so upstream failure or + budget exhaustion = zero charge. +- Added `budget_seconds` parameter to `generate()` (default 300s) to cap the + polling window. +- Bumped advertised `max_timeout_seconds` on video requests from 300s to 600s + so the signed auth stays valid across the full polling window. + +## 0.15.0 + +- **New image model: `openai/gpt-image-2`** (ChatGPT Images 2.0). Reasoning-driven generation with multilingual text rendering + character consistency. Pricing: $0.06 for 1024² / $0.12 for 1536×1024 or 1024×1536. Supports both `client.generate()` and `client.edit()` via the `/v1/images/image2image` endpoint. +- **New video models: 3 ByteDance Seedance variants** on `VideoClient`: + - `bytedance/seedance-1.5-pro` — $0.03/sec, 720p, 5s default (up to 10s). + - `bytedance/seedance-2.0-fast` — $0.15/sec, ~60-80s generation, sweet-spot price/quality. + - `bytedance/seedance-2.0` — $0.30/sec, 720p Pro quality. + All support text-to-video and image-to-video. Pass the model ID to `VideoClient.generate(..., model=...)`. +- README Image/Video sections list new models; image editing section notes `gpt-image-1` and `gpt-image-2` as supported. +- Also: `pyproject.toml` version was stuck at 0.13.0 despite `__version__` saying 0.14.1 (prevented PyPI publishes from shipping the NVIDIA refresh). Both now aligned at 0.15.0. + +## 0.14.1 + +- **NVIDIA free-tier refresh (backend 2026-04-21).** Router updated to point at the current survivors + the two new models: `nvidia/qwen3-next-80b-a3b-thinking` (reasoning flagship, 116 tok/s) and `nvidia/mistral-small-4-119b` (fastest free chat, 114 tok/s). +- Retired IDs no longer referenced by `router.py`: `nvidia/nemotron-super-49b`, `nvidia/nemotron-ultra-253b`, `nvidia/mistral-large-3-675b`. The backend still redirects them, but offline routing now points at the canonical successors (`nvidia/qwen3-next-80b-a3b-thinking`, `nvidia/mistral-small-4-119b`, `nvidia/llama-4-maverick`, `nvidia/glm-4.7`). +- AUTO / ECO `SIMPLE` primaries switched from `nvidia/kimi-k2.5` (retired) to `moonshot/kimi-k2.5` — backend redirect still works, but the router now references the canonical target. +- README NVIDIA table refreshed (8 visible models + `moonshot/kimi-k2.5`). + +## 0.14.0 + +- **New `SearchClient`** — wraps `POST /v1/search` (standalone Grok Live Search). $0.025 per source + margin, 1–50 sources per call. +- **New `XClient`** — 13 methods mapping the `/v1/x/*` endpoints (user lookup/info/followers/following/verified-followers/tweets/mentions, tweet lookup/replies/thread, search, trending, articles/rising). Replaces orphaned `X*` types that had no caller. +- **New `PriceClient`** — Pyth-backed market data with `.price()`, `.history()`, `.list_symbols()`. Crypto, FX and commodity are fully free (price + history + list); stocks across 12 markets (us/hk/jp/kr/gb/de/fr/nl/ie/lu/cn/ca) and the `usstock` legacy alias charge for price + history, list stays free. The client handles both paths transparently. +- `ChatMessage` gains optional `reasoning_content` and `thinking` fields for reasoning-capable models (DeepSeek Reasoner, Grok 4 / 4.20 reasoning). +- `ChatUsage` gains optional `cache_read_input_tokens` / `cache_creation_input_tokens` for Anthropic prompt caching telemetry. +- `Model` gains optional `billing_mode` (`paid`/`flat`/`free`), `flat_price`, `categories`, `hidden` so `list_models()` can surface full backend metadata. +- New market-data types: `PricePoint`, `PriceBar`, `PriceHistoryResponse`, `SymbolListResponse`. +- `VERSION` file synced to match `__init__.py`. + +## 0.13.0 + +- **New `VideoClient`** — generate AI videos via `xai/grok-imagine-video` ($0.05/sec, 8s default). +- `VideoResponse`, `VideoClip`, `VideoModel` types added. +- Text-to-video and image-to-video supported; client blocks until polling completes (~30-120s). +- `ImageData` now exposes `source_url` and `backed_up` for gateway-mirrored assets. +- Grok Imagine image models (`xai/grok-imagine-image`, `-pro`) routable via `ImageClient`. +- Grok 4.20 chat models (`xai/grok-4.20-reasoning`, `-non-reasoning`, `-multi-agent`) routable via the chat API. + +## 0.11.0 + +- 43+ models supported +- Base and Solana chain payments +- x402 v2 protocol +- Image generation support +- Anthropic-compatible client +- Smart model routing +- Response caching diff --git a/CLAUDE.md b/CLAUDE.md new file mode 100644 index 0000000..45f43e3 --- /dev/null +++ b/CLAUDE.md @@ -0,0 +1,67 @@ +# BlockRun LLM SDK (Python) + +Python SDK for 78 LLMs plus image/video/music/speech generation, standalone search, multi-chain RPC, and Pyth-backed market data — paid two ways: a BlockRun API key drawing on prepaid account credit, or USDC micropayments via x402 where the wallet signature is the authentication and the key never leaves your machine. + +## Commands + +```bash +pip install -e ".[dev]" # install in dev mode +pip install -e ".[dev,solana]" # with Solana support +pytest # run tests +black blockrun_llm/ # format code +ruff check blockrun_llm/ # lint +mypy blockrun_llm/ # type check +``` + +## Project structure + +``` +blockrun_llm/ +├── __init__.py # Package exports +├── client.py # LLMClient (EVM: Base, Arc — the 402's network picks the chain) +├── solana_client.py # SolanaLLMClient +├── wallet.py # EVM wallet management +├── solana_wallet.py # Solana wallet management +├── x402.py # x402 payment protocol +├── router_core/ # Port of @blockrun/router-core (shared with the TS SDK + gateway) +├── router_adapter.py # Host glue: catalog ids, payment floors, free profile +├── router.py # Back-compat shim over router_core +├── types.py # Type definitions +├── validation.py # Input validation +├── cache.py # Response caching +├── image.py # Image generation (+ image-to-image) +├── music.py # Music generation +├── speech.py # Text-to-speech + sound effects (BlockRun Voice / ElevenLabs) +├── video.py # Video generation +├── portrait.py # Virtual Portrait enrollment (AI characters) +├── realface.py # RealFace enrollment (real-person likeness) +├── search.py # Standalone Grok Live Search +├── price.py # Pyth market data (crypto/fx/commodity/stocks) +├── rpc.py # Multi-chain JSON-RPC (Tatum gateway, 40+ chains) +└── anthropic_client.py # Anthropic-compatible client +``` + +## Key dependencies + +- `httpx` — HTTP client +- `eth-account` — Ethereum wallet +- `pydantic` — Data validation +- `x402[svm]` — Solana x402 payments (optional) + +## Supported chains + +- Base Mainnet (primary) — USDC +- Arc (Circle, chain 5042) — USDC, via `api_url="https://arc.blockrun.ai/api"`; same `LLMClient` and key +- Base Sepolia (testnet) — Testnet USDC +- Solana Mainnet — USDC SPL + +The EVM domain signed follows the 402's `network` through `EVM_NETWORKS` in `blockrun_llm/x402.py`; a 402's `extra` is never trusted for it, an unknown network and a non-USDC `asset` are refused. + +## Conventions + +- Python >= 3.9 +- Format with Black (line-length 100) +- Lint with Ruff (line-length 100) +- Type check with mypy (strict) +- MIT license +- PyPI: `blockrun-llm` diff --git a/CONTRIBUTING.md b/CONTRIBUTING.md new file mode 100644 index 0000000..3ca8ce1 --- /dev/null +++ b/CONTRIBUTING.md @@ -0,0 +1,37 @@ +# Contributing to blockrun-llm + +## Setup + +```bash +git clone https://github.com/BlockRunAI/blockrun-llm +cd blockrun-llm +pip install -e ".[dev,solana]" +``` + +## Development + +```bash +pytest # Run tests +black blockrun_llm/ # Format +ruff check blockrun_llm/ # Lint +mypy blockrun_llm/ # Type check +``` + +## Code Standards + +- Python >= 3.9 +- Black formatting (line-length 100) +- Ruff linting (line-length 100) +- mypy strict mode +- All tests must pass + +## Pull Requests + +1. Fork the repo +2. Create a feature branch +3. Run `pytest`, `black`, `ruff`, `mypy` +4. Submit PR with clear description + +## License + +MIT diff --git a/LICENSE b/LICENSE index 3f30420..7812798 100644 --- a/LICENSE +++ b/LICENSE @@ -1,21 +1,21 @@ -MIT License - -Copyright (c) 2025 BlockRun - -Permission is hereby granted, free of charge, to any person obtaining a copy -of this software and associated documentation files (the "Software"), to deal -in the Software without restriction, including without limitation the rights -to use, copy, modify, merge, publish, distribute, sublicense, and/or sell -copies of the Software, and to permit persons to whom the Software is -furnished to do so, subject to the following conditions: - -The above copyright notice and this permission notice shall be included in all -copies or substantial portions of the Software. - -THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR -IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, -FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE -AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER -LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, -OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE -SOFTWARE. +MIT License + +Copyright (c) 2025 BlockRun + +Permission is hereby granted, free of charge, to any person obtaining a copy +of this software and associated documentation files (the "Software"), to deal +in the Software without restriction, including without limitation the rights +to use, copy, modify, merge, publish, distribute, sublicense, and/or sell +copies of the Software, and to permit persons to whom the Software is +furnished to do so, subject to the following conditions: + +The above copyright notice and this permission notice shall be included in all +copies or substantial portions of the Software. + +THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR +IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, +FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE +AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER +LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, +OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE +SOFTWARE. diff --git a/README.md b/README.md index 8fd6989..428634c 100644 --- a/README.md +++ b/README.md @@ -1,269 +1,1895 @@ -# BlockRun LLM SDK - -Pay-per-request access to GPT-4o, Claude 4, Gemini 2.5, and more via x402 micropayments on Base. - -**Network:** Base (Chain ID: 8453) -**Payment:** USDC -**Protocol:** x402 v2 (CDP Facilitator) - -## Installation - -```bash -pip install blockrun-llm -``` - -## Quick Start - -```python -from blockrun_llm import LLMClient - -client = LLMClient() # Uses BLOCKRUN_WALLET_KEY (never sent to server) -response = client.chat("openai/gpt-4o", "Hello!") -``` - -That's it. The SDK handles x402 payment automatically. - -## How It Works - -1. You send a request to BlockRun's API -2. The API returns a 402 Payment Required with the price -3. The SDK automatically signs a USDC payment on Base -4. The request is retried with the payment proof -5. You receive the AI response - -**Your private key never leaves your machine** - it's only used for local signing. - -## Available Models - -### OpenAI GPT-5 Family -| Model | Input Price | Output Price | -|-------|-------------|--------------| -| `openai/gpt-5.2` | $1.75/M | $14.00/M | -| `openai/gpt-5.1` | $1.25/M | $10.00/M | -| `openai/gpt-5` | $1.25/M | $10.00/M | -| `openai/gpt-5-mini` | $0.25/M | $2.00/M | -| `openai/gpt-5-nano` | $0.05/M | $0.40/M | -| `openai/gpt-5.2-pro` | $21.00/M | $168.00/M | -| `openai/gpt-5-pro` | $15.00/M | $120.00/M | - -### OpenAI GPT-4 Family -| Model | Input Price | Output Price | -|-------|-------------|--------------| -| `openai/gpt-4.1` | $2.00/M | $8.00/M | -| `openai/gpt-4.1-mini` | $0.40/M | $1.60/M | -| `openai/gpt-4.1-nano` | $0.10/M | $0.40/M | -| `openai/gpt-4o` | $2.50/M | $10.00/M | -| `openai/gpt-4o-mini` | $0.15/M | $0.60/M | - -### OpenAI O-Series (Reasoning) -| Model | Input Price | Output Price | -|-------|-------------|--------------| -| `openai/o1` | $15.00/M | $60.00/M | -| `openai/o1-mini` | $1.10/M | $4.40/M | -| `openai/o3` | $2.00/M | $8.00/M | -| `openai/o3-mini` | $1.10/M | $4.40/M | -| `openai/o4-mini` | $1.10/M | $4.40/M | - -### Anthropic Claude -| Model | Input Price | Output Price | -|-------|-------------|--------------| -| `anthropic/claude-opus-4` | $15.00/M | $75.00/M | -| `anthropic/claude-sonnet-4` | $3.00/M | $15.00/M | -| `anthropic/claude-haiku-4.5` | $1.00/M | $5.00/M | - -### Google Gemini -| Model | Input Price | Output Price | -|-------|-------------|--------------| -| `google/gemini-3-pro-preview` | $2.00/M | $12.00/M | -| `google/gemini-2.5-pro` | $1.25/M | $10.00/M | -| `google/gemini-2.5-flash` | $0.15/M | $0.60/M | -| `google/gemini-2.5-flash-lite` | **Free** | **Free** | - -### Image Generation -| Model | Price | -|-------|-------| -| `openai/dall-e-3` | $0.04-0.08/image | -| `openai/gpt-image-1` | $0.02-0.04/image | - -## Usage Examples - -### Simple Chat - -```python -from blockrun_llm import LLMClient - -client = LLMClient() # Uses BLOCKRUN_WALLET_KEY (never sent to server) - -response = client.chat("openai/gpt-4o", "Explain quantum computing") -print(response) - -# With system prompt -response = client.chat( - "anthropic/claude-sonnet-4", - "Write a haiku", - system="You are a creative poet." -) -``` - -### Full Chat Completion - -```python -from blockrun_llm import LLMClient - -client = LLMClient() # Uses BLOCKRUN_WALLET_KEY (never sent to server) - -messages = [ - {"role": "system", "content": "You are a helpful assistant."}, - {"role": "user", "content": "How do I read a file in Python?"} -] - -result = client.chat_completion("openai/gpt-4o", messages) -print(result.choices[0].message.content) -``` - -### Async Usage - -```python -import asyncio -from blockrun_llm import AsyncLLMClient - -async def main(): - async with AsyncLLMClient() as client: - # Simple chat - response = await client.chat("openai/gpt-4o", "Hello!") - print(response) - - # Multiple requests concurrently - tasks = [ - client.chat("openai/gpt-4o", "What is 2+2?"), - client.chat("anthropic/claude-sonnet-4", "What is 3+3?"), - client.chat("google/gemini-2.5-flash", "What is 4+4?"), - ] - responses = await asyncio.gather(*tasks) - for r in responses: - print(r) - -asyncio.run(main()) -``` - -### List Available Models - -```python -from blockrun_llm import LLMClient - -client = LLMClient() -models = client.list_models() - -for model in models: - print(f"{model['id']}: ${model['inputPrice']}/M input, ${model['outputPrice']}/M output") -``` - -## Environment Variables - -| Variable | Description | Required | -|----------|-------------|----------| -| `BLOCKRUN_WALLET_KEY` | Your EVM wallet private key | Yes (or pass to constructor) | -| `BLOCKRUN_API_URL` | API endpoint | No (default: https://blockrun.ai/api) | - -## Setting Up Your Wallet - -1. Create a wallet on Base network (Coinbase Wallet, MetaMask, etc.) -2. Get some ETH on Base for gas (small amount, ~$1) -3. Get USDC on Base for API payments -4. Export your private key and set it as `BLOCKRUN_WALLET_KEY` - -```bash -# .env file -BLOCKRUN_WALLET_KEY=0x...your_private_key_here -``` - -## Error Handling - -```python -from blockrun_llm import LLMClient, APIError, PaymentError - -client = LLMClient() - -try: - response = client.chat("openai/gpt-4o", "Hello!") -except PaymentError as e: - print(f"Payment failed: {e}") - # Check your USDC balance -except APIError as e: - print(f"API error ({e.status_code}): {e}") -``` - -## Testing - -### Running Unit Tests - -Unit tests do not require API access or funded wallets: - -```bash -pytest tests/unit # Run unit tests only -pytest tests/unit --cov # Run with coverage report -pytest tests/unit -v # Verbose output -``` - -### Running Integration Tests - -Integration tests call the production API and require: -- A funded Base wallet with USDC ($1+ recommended) -- `BLOCKRUN_WALLET_KEY` environment variable set -- Estimated cost: ~$0.05 per test run - -```bash -export BLOCKRUN_WALLET_KEY=0x... -pytest tests/integration # Run integration tests only -pytest # Run all tests -``` - -Integration tests are automatically skipped if `BLOCKRUN_WALLET_KEY` is not set. - -## Security - -### Private Key Safety - -- **Private key stays local**: Your key is only used for signing on your machine -- **No custody**: BlockRun never holds your funds -- **Verify transactions**: All payments are on-chain and verifiable - -### Best Practices - -**Private Key Management:** -- Use environment variables, never hard-code keys -- Use dedicated wallets for API payments (separate from main holdings) -- Set spending limits by only funding payment wallets with small amounts -- Never commit `.env` files to version control -- Rotate keys periodically - -**Input Validation:** -The SDK validates all inputs before API requests: -- Private keys (format, length, valid hex) -- API URLs (HTTPS required for production, HTTP allowed for localhost) -- Model names and parameters (ranges for max\_tokens, temperature, top\_p) - -**Error Sanitization:** -API errors are automatically sanitized to prevent sensitive information leaks. - -**Monitoring:** -```python -address = client.get_wallet_address() -print(f"View transactions: https://basescan.org/address/{address}") -``` - -**Keep Updated:** -```bash -pip install --upgrade blockrun-llm # Get security patches -``` - -## Links - -- [Website](https://blockrun.ai) -- [Documentation](https://docs.blockrun.ai) -- [GitHub](https://github.com/blockrun/blockrun-llm) -- [Discord](https://discord.gg/blockrun) - -## License - -MIT +# BlockRun LLM SDK (Python) + +> **blockrun-llm** is a Python SDK for accessing 80+ large language models (GPT-5.x, Claude 4.x, Gemini 3.x, DeepSeek, Grok 4.x, GLM, MiniMax, Moonshot and more) plus image / video / music generation, Grok Live Search, prediction-market data (Predexon), Exa neural web search, and Pyth-backed market data. Every call is paid per request — no subscription, no seats, no minimum. Built for AI agents that need to operate autonomously. +> +> **Two ways to pay, same SDK, same catalogue.** Sign up at +> **[user.blockrun.ai](https://user.blockrun.ai)** for an API key and prepaid +> credit (top up with a card), or hold USDC in your own wallet and let each +> request settle itself over x402 — on **Solana or Base**. Every client takes +> either credential in the same first argument. +> +> 🆓 **Includes 8 fully-free NVIDIA-hosted models** — DeepSeek V4 Flash (1M context), Nemotron Nano Omni (vision), Qwen3 Next + Coder, Llama 4 Maverick, Mistral Small 4, plus `gpt-oss-120b/20b` (hidden from `/v1/models` but direct calls still work). Zero USDC, no rate-limit gimmicks. Use `routing_profile="free"` or call any `nvidia/*` model directly. + +[![PyPI](https://img.shields.io/pypi/v/blockrun-llm.svg)](https://pypi.org/project/blockrun-llm/) +[![License: MIT](https://img.shields.io/badge/License-MIT-green.svg)](LICENSE) + +**BlockRun assumes Claude Code as the agent runtime.** + +## Supported Chains + +| Rail | Network | Payment | Status | +|------|---------|---------|--------| +| **API key** | none — `api.blockrun.ai` | prepaid credit, topped up with a card | ✅ | +| **Solana** | Solana Mainnet | USDC (SPL), gasless — the facilitator pays the fee | ✅ Recommended for x402 | +| **Base** | Base Mainnet (Chain ID: 8453) | USDC | ✅ | +| **Arc** | Circle Arc (Chain ID: 5042) — `arc.blockrun.ai` | USDC (Arc's native token), settled by Circle — no gas per call | ✅ | +| **Base Testnet** | Base Sepolia (Chain ID: 84532) | Testnet USDC | ✅ Development | + +**Protocol:** x402 v2 on the wallet rails; plain bearer auth on the API-key rail. + +## Installation + +```bash +pip install blockrun-llm # Base chain (EVM/USDC) — includes all core deps +pip install blockrun-llm[solana] # Base + Solana (USDC SPL) payments +pip install blockrun-llm[dev] # Base + dev tools (pytest, black, ruff, mypy) +pip install blockrun-llm[dev,solana] # Everything +``` + +## Quick Start + +```python +from blockrun_llm import LLMClient + +# Reads BLOCKRUN_API_KEY if set, otherwise BLOCKRUN_WALLET_KEY for x402. +client = LLMClient() +response = client.chat("openai/gpt-5.2", "Hello!") +``` + +```bash +# API key — sign up at https://user.blockrun.ai, then: +export BLOCKRUN_API_KEY=brk_live_... + +# …or a wallet, and every call pays itself in USDC: +export SOLANA_WALLET_KEY=... # with SolanaLLMClient +export BLOCKRUN_WALLET_KEY=0x... # with LLMClient (Base) +``` + +That's it. Either credential can also be passed directly — +`LLMClient("brk_live_…")` or `LLMClient("0x…")` — and `client.payment_mode` +reports which rail you ended up on. + +### Try It Free (No Balance Required) + +Want to kick the tires before topping up or funding a wallet? Route to +BlockRun's free NVIDIA tier — it settles $0 on both rails, so an unfunded wallet +or a $0 credit account is enough: + +```python +from blockrun_llm import LLMClient + +client = LLMClient() # a credential is still needed; a balance is not + +# Option 1: call a free model directly +response = client.chat("nvidia/step-3.7-flash", "Explain x402 in 1 sentence") + +# Option 2: let the smart router pick the best free model per request +result = client.smart_chat("What is 2+2?", routing_profile="free") +print(result.model) # e.g. 'nvidia/step-3.7-flash' (cheapest capable for SIMPLE tier) +print(result.response) # '4' +``` + +**Available free models** (input + output both $0, all NVIDIA-hosted): + +| Model ID | Context | Best For | +|----------|---------|----------| +| `nvidia/step-3.7-flash` | 131K | Fast general-purpose chat + reasoning | +| `nvidia/mistral-nemotron` | 131K | Fast free Mistral (Mistral × NVIDIA) | +| `nvidia/nemotron-3-nano-omni-30b-a3b-reasoning` | 256K | Only vision-capable free model — text + images + video (≤2 min) + audio (≤1 hr) | +| `nvidia/nemotron-nano-9b-v2` | 131K | Compact fast chat | +| `nvidia/nemotron-nano-12b-v2-vl` | 131K | Compact vision | +| `nvidia/gpt-oss-120b` | 128K | OpenAI open-weight 120B — the free workhorse. Hidden from `/v1/models` (so SmartChat won't auto-pick it) but direct calls still work | +| `nvidia/gpt-oss-20b` | 128K | OpenAI open-weight 20B — 155 tok/s. Hidden from `/v1/models` but direct calls still work | + +> Need V4-Pro-class reasoning? Use the paid `deepseek/deepseek-v4-pro` ($0.435/$0.87 — the 75% launch promo became the permanent list price after 2026-05-31) — `nvidia/deepseek-v4-pro` is hidden because NVIDIA's NIM deployment is hung; backend MODEL_REDIRECTS forwards calls to V4 Flash. + +> **Privacy note for `gpt-oss-120b/20b`**: NVIDIA's free build.nvidia.com tier reserves the right to use prompts/outputs for service improvement. The models are hidden from `/v1/models` so SmartChat won't auto-route to them, but direct calls still work — use them only when prompts contain no sensitive data. + +> **Retired**: NVIDIA has EOL'd (HTTP 410) most of its early free lineup — the free DeepSeek family (last: `nvidia/deepseek-v4-flash`, 2026-08-12), `llama-4-maverick`, the qwen3 SKUs, free Mistral small/large, and more. The gateway auto-redirects pinned callers to a healthy free model, so old model IDs still return 200. + +## Solana Support + +**Solana is the recommended chain for x402 payments**: settlement is sub-second +and BlockRun's facilitator co-signs as fee payer, so a transfer costs you no SOL +and you hold nothing but USDC. Base works identically and remains what the bare +`LLMClient` uses, so nothing existing changes — but if you are choosing today, +choose Solana. + +Pay for AI calls with Solana USDC via [sol.blockrun.ai](https://sol.blockrun.ai): + +```python +from blockrun_llm import SolanaLLMClient + +# SOLANA_WALLET_KEY env var (bs58-encoded Solana secret key) +client = SolanaLLMClient() + +# Or pass key directly +client = SolanaLLMClient(private_key="your-bs58-solana-key") + +# A BlockRun API key works here too — on the account rail there is no transfer +# to sign, so the chain stops being a question. +client = SolanaLLMClient(private_key="brk_live_...") + +# Same API as LLMClient +response = client.chat("openai/gpt-5.2", "gm Solana") +print(response) + +# DeepSeek on Solana +answer = client.chat("deepseek/deepseek-chat", "Explain Solana consensus", temperature=0.5) +``` + +**Agent setup (auto-loads or creates a wallet):** +```python +from blockrun_llm import setup_agent_solana_wallet + +client = setup_agent_solana_wallet() # uses ~/.blockrun/.solana-session, env, or creates one +client.chat("openai/gpt-5.2", "gm Solana") +``` + +**Setup:** +```bash +pip install blockrun-llm[solana] +export SOLANA_WALLET_KEY="your-bs58-solana-key" +``` + +**Endpoint:** `https://sol.blockrun.ai/api` +**Payment:** Solana USDC (SPL Token, mainnet) + +> **Base vs Solana keys are not interchangeable.** A Solana key is base58 +> (~44 chars for a seed, ~88 for a full keypair); a Base/EVM key is `0x` + 64 +> hex chars. Pass a Solana key to `SolanaLLMClient` — **not** `LLMClient` / +> `setup_agent_wallet()`. If you do mix them up, the SDK now tells you exactly +> what to switch to instead of failing with a cryptic "must be 66 characters" +> error. + +## Arc Support + +The same `LLMClient` pays on [Circle's Arc](https://www.arc.network) via [arc.blockrun.ai](https://arc.blockrun.ai) — point `api_url` at it and hold USDC on Arc in the same EVM wallet: + +```python +from blockrun_llm import LLMClient + +client = LLMClient(api_url="https://arc.blockrun.ai/api") # BLOCKRUN_WALLET_KEY as usual +print(client.chat("openai/gpt-4o", "gm Arc")) +``` + +The 402 from that host names `eip155:5042`, and the SDK signs the EIP-3009 authorization against Arc's USDC (`0x3600…0000`, EIP-712 domain `USDC` v2) — never Base's. Circle's facilitator verifies and settles it on Arc; you pay no gas. Which networks the SDK can sign for is the `EVM_NETWORKS` table in `blockrun_llm.x402` (Base, Arc, Base Sepolia); a 402 naming any other network, or a non-USDC asset, is refused before anything is signed. + +**Setup:** +1. Same wallet key as Base: `export BLOCKRUN_WALLET_KEY="0x..."` +2. Fund it with USDC on Arc (Arc's native token, shown as the ERC-20 at `0x3600…0000`) +3. `api_url="https://arc.blockrun.ai/api"` — payments are automatic via x402 + +## Smart Routing (Router Core) + +Let the SDK automatically pick the cheapest capable model for each request: + +```python +from blockrun_llm import LLMClient + +client = LLMClient() + +# Auto-routes to the cheapest capable model +result = client.smart_chat("Summarize this changelog entry in one line") +print(result.response) +print(result.model) # 'google/gemini-2.5-flash' +print(result.routing.task_type) # 'chat' +print(f"Saved {result.routing.savings * 100:.0f}%") # 'Saved 90%' + +# Complex reasoning task -> routes to a reasoning model +result = client.smart_chat("Prove the Riemann hypothesis step by step") +print(result.model) # 'deepseek/deepseek-v4-pro' +``` + +Routing works the same on every client — `LLMClient`, `AsyncLLMClient`, +`SolanaLLMClient` and `AsyncSolanaLLMClient` all expose `route()`, +`smart_chat()` and `smart_chat_completion()`. Both chains run the same engine +against the same catalog, so an identical request picks an identical model; only +the x402 minimum in the cost estimate differs. + +```python +# Route a full message list — tools and response_format shape the decision, +# not just the request +result = client.smart_chat_completion( + [{"role": "user", "content": "Cancel order B-42"}], + tools=[{"type": "function", "function": {"name": "cancel_order", "parameters": {}}}], + tool_choice="required", +) +print(result.model) # a tool-capable model +print(result.routing.task_type) # 'tool_agent' + +# Or opt in from OpenAI-compatible code by changing one string +response = client.chat_completion("blockrun/auto", messages) +``` + +Want to see the decision without paying for a call? `client.route(...)` runs the +same routing locally and returns the decision only: + +```python +decision = client.route("Prove the Riemann hypothesis step by step") +print(decision.model) # 'deepseek/deepseek-v4-pro' +print(decision.tier) # 'REASONING' +print(decision.task_type) # 'reasoning' +print(decision.candidates) # ordered chain; smart_chat walks it on a 5xx/timeout +print(decision.reasoning) # human-readable explanation of the pick +``` + +### Routing Profiles + +| Profile | Description | Best For | +|---------|-------------|----------| +| `free` | NVIDIA free tier — smart-routes across the 6 $0 models (Step 3.7 Flash, Mistral Nemotron, Nemotron Nano Omni / 9B / 12B VL) | Zero-cost testing, dev, prod | +| `eco` | Cheapest capable model per tier | Cost-sensitive production | +| `auto` | Best balance of cost/quality (default) | General use | +| `premium` | Top-tier models (Anthropic, OpenAI, Moonshot) | Quality-critical tasks | + +```python +# Use premium models for complex tasks +result = client.smart_chat( + "Write production-grade async Python code", + routing_profile="premium" +) +print(result.model) # 'openai/gpt-5.3-codex' +``` + +### How It Works + +Routing runs on [Router Core](https://github.com/BlockRunAI/router-core) — the +same product-neutral engine the TypeScript SDK and the BlockRun gateway use, so +an identical request routes identically across all three. It is 100% local and +takes <1ms; no extra model call is made to decide. + +Three stages: + +1. **Classify** — a 15-dimension weighted + scorer maps the request onto a capability tier (token count, code presence, + reasoning markers, technical terms, creative markers, agentic patterns, and + more), and a task classifier labels the *shape* of the work: `chat`, + `code_edit`, `code_agent`, `tool_agent`, `reasoning_math`, `long_context`, + `extraction`, `vision`, … +2. **Filter** — capability constraints are hard filters, not preferences. A + model that cannot hold the conversation, emit the requested output length, + call tools, or read images is dropped before scoring, so the router never + picks a model the request would fail on. +3. **Rank** — surviving candidates are scored on task affinity, cost, speed and + reliability. The winner serves the request; the rest become the ordered + fallback chain that `smart_chat` walks on a timeout or 5xx. + +The four capability tiers: + +| Tier | Example Tasks | Auto Profile Model | +|------|---------------|-------------------| +| SIMPLE | Short questions, definitions | google/gemini-2.5-flash | +| MEDIUM | Code snippets, explanations | moonshot/kimi-k2.7 | +| COMPLEX | Architecture, long documents | google/gemini-3.1-pro | +| REASONING | Proofs, math, multi-step reasoning | deepseek/deepseek-v4-pro | + +Every decision is explainable — `result.routing` carries the tier, the task +type, the confidence, the ranked `candidates`, the per-candidate +`candidate_scores` (quality / cost / speed / reliability) and a `reasoning` +string describing why that model won. + +## How Payment Works + +Two front doors onto the same gateway, the same catalogue and the same response +shapes. You choose one with the credential you hand the client. + +| | **API key** — `api.blockrun.ai` | **Wallet (x402)** — `sol.blockrun.ai` / `blockrun.ai` | +|---|---|---| +| Authenticates with | `brk_live_…` from [user.blockrun.ai](https://user.blockrun.ai) | a signature from your own wallet | +| Pays from | prepaid credit on your account | USDC you hold, settled on-chain per call | +| Set up by | signing in with Google, minting a key, topping up with a card | funding a wallet with USDC | +| Chain | none — credit is off-chain | **Solana** or **Base** | +| Custody | BlockRun holds the credit you bought | non-custodial; your key never leaves your machine | +| Best for | teams that cannot run wallets, CI, anyone who wants a card receipt | agents, autonomous spend, no-signup access | + +Free models are free on both. + +### Option A — API key (user.blockrun.ai) + +1. **Sign in** at **[user.blockrun.ai](https://user.blockrun.ai)** with Google. +2. **Mint a key** on the *API Keys* page. It is shown once — copy it then. +3. **Top up** on the *Billing* page with a card. Minimum $5. The processing fee + (5.5% + $0.30) is charged **once, at purchase** — never on a call — so $10.85 + buys $10.00 of credit and every model then bills at the published list price, + with no per-call minimum and no per-call fee. +4. **Export it:** + +```bash +export BLOCKRUN_API_KEY=brk_live_... +``` + +```python +from blockrun_llm import LLMClient + +client = LLMClient() # picks up BLOCKRUN_API_KEY +print(client.payment_mode) # 'apikey' +print(client.chat("openai/gpt-5.2", "What is 2+2?")) +``` + +Requests go to `https://api.blockrun.ai/v1` with the key as +`Authorization: Bearer …`. There is no 402 round trip and nothing is signed — +the gateway meters the call at exact usage and draws it from your credit. +Spending, per-call activity and remaining balance are on +[user.blockrun.ai/dashboard](https://user.blockrun.ai/dashboard). + +**Precedence**, since it decides whether a call spends credit or on-chain USDC: + +1. an explicit argument — `LLMClient("brk_live_…")` or `LLMClient("0x…")`; +2. `BLOCKRUN_API_KEY`, which **beats** `BLOCKRUN_WALLET_KEY` / + `BASE_CHAIN_WALLET_KEY` / `SOLANA_WALLET_KEY`; +3. the wallet variables. + +An existing wallet setup is untouched until you set `BLOCKRUN_API_KEY`, and +passing a wallet key explicitly always opts back out. `BLOCKRUN_API_KEY_URL` +overrides the account-rail host; it is deliberately not `BLOCKRUN_API_URL`, +which names an x402 gateway — an API-key client following that would send your +key to a host configured for a different rail. + +**What changes.** `get_balance()`, `get_balance_testnet()` and `onramp()` raise +a `ValueError` pointing at the dashboard rather than answering: returning `0` +is indistinguishable from an empty wallet, and an agent gating on it would stop +calling a well-funded account. `get_wallet_address()` returns `""`. Running out +of credit raises a `PaymentError` naming the top-up page, not a wallet error. +`setup_agent_wallet()` mints nothing and hands back an API-key client, so a +skill can call it unconditionally. Everything else is identical. + +### Option B — wallet + x402 + +You hold USDC in your own wallet — on **Solana** or **Base** — and each request +pays for itself with an on-chain micropayment. No signup, nothing custodial. + +#### Phase 1 — Fund your wallet once + +You only do this when your balance runs low. Three ways: + +```python +from blockrun_llm import LLMClient + +client = LLMClient() # auto-detects wallet from BLOCKRUN_WALLET_KEY + +# (a) Buy USDC with a card / bank (FREE) — mint a one-time Coinbase Onramp link +link = client.onramp(client.get_wallet_address()) +print(link["url"]) # open https://pay.coinbase.com/... to buy USDC on Base +# The link is single-use and expires in ~5 min — mint it at click time, never cache it. + +# (b) Transfer existing Base USDC to your wallet address +print(client.get_wallet_address()) # send USDC on Base to this 0x… address + +# (c) Skip funding entirely — the free NVIDIA models cost $0 +client.chat("nvidia/step-3.7-flash", "Hello!") # routing_profile="free" also works +``` + +`$5` of USDC covers thousands of paid requests. Check your balance any time: + +```python +print(f"Balance: ${client.get_balance():.2f} USDC") +``` + +#### Phase 2 — Every request pays itself (automatic x402) + +```python +reply = client.chat("anthropic/claude-sonnet-4.6", "Explain x402 in one line") +``` + +That single call does all of this under the hood: + +1. You send the request to BlockRun's gateway. +2. The gateway returns `402 Payment Required` with the price. +3. The SDK signs a USDC payment on Base **locally** (EIP-712) — your private + key never leaves your machine. +4. The request is retried with the signed payment proof. +5. The gateway settles on-chain and returns the AI response. + +One call, no separate pay step. + +### What it costs, and how to verify it + +- **Pay-as-you-go, per call.** You pay only the gateway price of each request + (see [Available Models](#available-models)). The free NVIDIA models are `$0`. +- **Track spend.** `client.get_spending()` returns this session's + `{total_usd, calls}`. On the API-key rail the gateway does not tell the client + what a call cost, so treat that total as a floor and + [user.blockrun.ai/dashboard](https://user.blockrun.ai/dashboard) as the + authority. Every paid call also appends a line to + `~/.blockrun/cost_log.jsonl`; summarize/export it with + `blockrun_llm.billing` (`get_cost_log_summary`, `export_cost_log_csv`) — see + [Billing & Cost Tracking](#billing--cost-tracking). +- **Verify settlements on-chain.** Each settlement returns a tx hash you can + inspect on BaseScan — `https://basescan.org/tx/`, or view all activity + for your wallet at `https://basescan.org/address/`. +- **Non-custodial.** Your wallet is yours; the key is only used for local + signing and **never leaves your machine**. No deposits held by BlockRun. + +## Available Models + +### OpenAI GPT-5.5 Family +Released 2026-04-23 — first fully retrained base since GPT-4.5. 1M context, 128K output, native agent + computer use. + +| Model | Input Price | Output Price | Context | +|-------|-------------|--------------|---------| +| `openai/gpt-5.5` | $5.00/M | $30.00/M | 1M | + +### OpenAI GPT-5.4 Family +| Model | Input Price | Output Price | Context | +|-------|-------------|--------------|---------| +| `openai/gpt-5.4` | $2.50/M | $15.00/M | 1M | +| `openai/gpt-5.4-pro` | $30.00/M | $180.00/M | 1M | +| `openai/gpt-5.4-mini` | $0.75/M | $4.50/M | 400K | +| `openai/gpt-5.4-nano` | $0.20/M | $1.25/M | 1M | + +### OpenAI GPT-5 Family +| Model | Input Price | Output Price | Context | +|-------|-------------|--------------|---------| +| `openai/gpt-5.3` | $1.75/M | $14.00/M | 128K | +| `openai/gpt-5.2` | $1.75/M | $14.00/M | 400K | +| `openai/gpt-5-mini` | $0.25/M | $2.00/M | 200K | +| `openai/gpt-5.2-pro` | $21.00/M | $168.00/M | 400K | +| `openai/gpt-5.3-codex` | $1.75/M | $14.00/M | 400K | + +### OpenAI GPT-4o Family +| Model | Input Price | Output Price | Context | +|-------|-------------|--------------|---------| +| `openai/gpt-4o` | $2.50/M | $10.00/M | 128K | +| `openai/gpt-4o-mini` | $0.15/M | $0.60/M | 128K | + +### OpenAI O-Series (Reasoning) +| Model | Input Price | Output Price | Context | +|-------|-------------|--------------|---------| +| `openai/o1` | $15.00/M | $60.00/M | 200K | +| `openai/o3` | $2.00/M | $8.00/M | 200K | +| `openai/o3-mini` | $1.10/M | $4.40/M | 128K | + +### Anthropic Claude +| Model | Input Price | Output Price | Context | Notes | +|-------|-------------|--------------|---------|-------| +| `anthropic/claude-opus-4.8` | $5.00/M | $25.00/M | 1M | Most capable Claude — agentic coding + adaptive thinking, 128K output | +| `anthropic/claude-opus-4.7` | $5.00/M | $25.00/M | 1M | Agentic coding + adaptive thinking, 128K output | +| `anthropic/claude-opus-4.6` | $5.00/M | $25.00/M | 200K | Hidden from `/v1/models` (superseded by 4.7); direct calls still work | +| `anthropic/claude-opus-4.5` | $5.00/M | $25.00/M | 200K | | +| `anthropic/claude-sonnet-4.6` | $3.00/M | $15.00/M | 200K | | +| `anthropic/claude-haiku-4.5` | $1.00/M | $5.00/M | 200K | | + +### Google Gemini +| Model | Input Price | Output Price | Context | +|-------|-------------|--------------|---------| +| `google/gemini-3.1-pro` | $2.00/M | $12.00/M | 1M | +| `google/gemini-3.5-flash` | $0.50/M | $3.00/M | 1M | +| `google/gemini-3-flash-preview` | $0.50/M | $3.00/M | 1M | +| `google/gemini-2.5-pro` | $1.25/M | $10.00/M | 1M | +| `google/gemini-2.5-flash` | $0.30/M | $2.50/M | 1M | +| `google/gemini-3.1-flash-lite` | $0.25/M | $1.50/M | 1M | +| `google/gemini-2.5-flash-lite` | $0.10/M | $0.40/M | 1M | + +### DeepSeek + +V4 family launched 2026-04-24. DeepSeek upstream now serves the legacy +`deepseek-chat` / `deepseek-reasoner` aliases as V4 Flash non-thinking / +thinking modes. V4 Pro is the new flagship paid SKU — 1.6T MoE / 49B active, +1M context, MMLU-Pro 87.5, GPQA 90.1, SWE-bench 80.6, LiveCodeBench 93.5. + +| Model | Input Price | Output Price | Context | Notes | +|-------|-------------|--------------|---------|-------| +| `deepseek/deepseek-v4-pro` | $0.435/M | $0.87/M | 1M | V4 flagship — strongest open-weight reasoner. The 75% launch promo became the permanent list price after 2026-05-31 | +| `deepseek/deepseek-chat` | $0.14/M | $0.28/M | 1M | V4 Flash non-thinking (paid endpoint with 5MB request bodies) | +| `deepseek/deepseek-reasoner` | $0.20/M | $0.40/M | 1M | V4 Flash thinking (same upstream as `deepseek-chat`, thinking enabled by default) | + +### MiniMax +| Model | Input Price | Output Price | Context | Notes | +|-------|-------------|--------------|---------|-------| +| `minimax/minimax-m3` | $0.30/M | $1.20/M | 1M | M3 flagship — strong reasoning + coding, 1M context | +| `minimax/minimax-m2.7` | $0.30/M | $1.20/M | 200K | | + +### xAI Grok + +Grok 4.3 and Grok Build are resold through BlockRun's OpenRouter credit pool +(same pattern as `deepseek/deepseek-v4-pro` and `minimax/minimax-m3`). Older +Grok chat SKUs (grok-3/4/4.1-fast families) are hidden from `/v1/models` but +direct calls by full ID still work. + +| Model | Input Price | Output Price | Context | Notes | +|-------|-------------|--------------|---------|-------| +| `xai/grok-4.3` | $1.25/M | $2.50/M | 1M | Reasoning model, vision-capable, tuned for agentic workflows | +| `xai/grok-build-0.1` | $1.00/M | $2.00/M | 256K | Fast agentic coding model — interactive software-engineering workflows | + +### ZAI + +The GLM flat-rate launch promos have fully ended (glm-5.1 on 2026-06-05; +glm-5 and glm-5-turbo on 2026-06-06) — the whole family now bills per-token. + +| Model | Input Price | Output Price | Context | Notes | +|-------|-------------|--------------|---------|-------| +| `zai/glm-5.2` | $1.40/M | $4.40/M | 1M | Z.AI's newest flagship — 1M-token context, top open-source on long-horizon coding | +| `zai/glm-5.1` | $1.40/M | $4.40/M | 200K | #1 open-source on SWE-Bench Pro, 8-hour autonomous execution | +| `zai/glm-5` | $0.60/M | $1.92/M | 200K | | +| `zai/glm-5-turbo` | $1.20/M | $4.00/M | 200K | | + +### NVIDIA (Free & Hosted) + +Free tier refreshed 2026-08-12. NVIDIA has retired (HTTP 410 end-of-life) +the entire free DeepSeek family — `nvidia/deepseek-v4-flash` was the last to +go — along with `llama-4-maverick`, `qwen3-coder-480b`, the free Mistral +small/large SKUs, and others. Retired models stay callable by ID: the gateway +auto-redirects them to a healthy free model, so pinned callers still get a +200. `nvidia/gpt-oss-120b` and `nvidia/gpt-oss-20b` remain callable by direct +ID but are hidden from `/v1/models` over the NVIDIA free tier's +prompt-retention terms (so SmartChat won't auto-pick them). The live list is +`GET /v1/models` filtered on the free flag. + +| Model | Input Price | Output Price | Context | Notes | +|-------|-------------|--------------|---------|-------| +| `nvidia/step-3.7-flash` | **FREE** | **FREE** | 131K | Fast general-purpose chat + reasoning | +| `nvidia/mistral-nemotron` | **FREE** | **FREE** | 131K | Fast free Mistral | +| `nvidia/nemotron-3-nano-omni-30b-a3b-reasoning` | **FREE** | **FREE** | 256K | First vision-capable free model — RGB images, mp4 video | +| `nvidia/nemotron-nano-9b-v2` | **FREE** | **FREE** | 131K | Compact fast chat | +| `nvidia/gpt-oss-120b` | **FREE** | **FREE** | 128K | OpenAI open-weight 120B — 123 tok/s. Hidden from `/v1/models`; direct calls work | +| `nvidia/gpt-oss-20b` | **FREE** | **FREE** | 128K | OpenAI open-weight 20B — 155 tok/s. Hidden from `/v1/models`; direct calls work | +| `moonshot/kimi-k2.5` | $0.60/M | $3.00/M | 262K | Kimi K2.5 direct from Moonshot (replaces `nvidia/kimi-k2.5`) | +| `moonshot/kimi-k2.6` | $0.95/M | $4.00/M | 256K | Moonshot flagship (vision + reasoning_content) | + +### Testnet Models (Base Sepolia) +| Model | Price | +|-------|-------| +| `openai/gpt-oss-20b` | $0.001/request | +| `openai/gpt-oss-120b` | $0.002/request | + +*Testnet models use flat pricing (no token counting) for simplicity.* + +### Verifying Models End-to-End + +The SDK ships two runnable sweep scripts under `examples/`: + +```bash +# Chat LLMs — every chat model the SDK exposes +python examples/sweep_all_chat_models.py --output-json sweep-results.json + +# Image + music models (video excluded — long polling, expensive per clip) +python examples/sweep_all_media_models.py --output-json sweep-media-results.json +``` + +Each script captures per-model status, latency, token counts, and per-call +cost, prints a grouped report, and exits non-zero if any expected-to-work +model fails. Useful before a release or after router/catalog changes. + +`smart_chat()` and `chat()` accept an optional `fallback_models=[...]` list — +on timeout / 5xx / network error the SDK transparently walks the chain +before raising. `smart_chat()` populates this from the tier's fallback list +automatically. + +### Image Generation + +| Model | Price | +|-------|-------| +| `openai/dall-e-3` | $0.04/image | +| `openai/gpt-image-1` | $0.02/image | +| `openai/gpt-image-2` | $0.06/image (reasoning-driven, multilingual text rendering, character consistency) | +| `google/nano-banana` | $0.05/image | +| `google/nano-banana-pro` | $0.10/image | +| `xai/grok-imagine-image` | $0.02/image | +| `xai/grok-imagine-image-pro` | $0.07/image | +| `zai/cogview-4` | $0.015/image | + +Image editing (`client.edit` / `client.image_edit`) hits the `/v1/images/image2image` endpoint and supports `openai/gpt-image-1`, `openai/gpt-image-2`, `google/nano-banana`, and `google/nano-banana-pro`. Pass a list of source images to fuse multiple inputs (openai/* up to 4, google/* up to 3). + +**`quality` (Solana only).** On Solana, `image` and `image_edit` accept +`quality="low" | "medium" | "high" | "auto"` for `openai/gpt-image-*` — `low` +meaningfully cuts generation time: + +```python +sol = SolanaLLMClient() +result = sol.image("a red apple", model="openai/gpt-image-2", quality="low") +``` + +This is deliberately absent from the Base `ImageClient`: the Base gateway has +no `quality` field and would silently ignore the value, so passing it there +raises `TypeError` rather than quietly doing nothing. + +### Video Generation +| Model | Price | Default 5s 720p | +|-------|-------|-----------------| +| `xai/grok-imagine-video` | $0.050/sec | 8s ≈ $0.40 | +| `bytedance/seedance-1.5-pro` | $4.32 / M tok (flat) | ≈ $0.46 | +| `bytedance/seedance-2.0-fast` | $11.20 / M text · $6.60 / M image | ≈ $1.19 t2v / $0.70 i2v | +| `bytedance/seedance-2.0` | $14.00 / M text · $8.60 / M image | ≈ $1.49 t2v / $0.91 i2v | + +Seedance is billed by token360 in tokens (~20,256 tok/sec at 720p). Drop +`resolution="480p"` for ~half the cost, or bump to `1080p` / `4K`. +Seedance defaults to `720p` with synced audio on text-to-video; image- or +face-conditioned paths default audio off. Grok ignores `resolution` and +`generate_audio`. + +```python +from blockrun_llm import VideoClient + +client = VideoClient() +result = client.generate("a red apple slowly spinning on a wooden table") +print(result.data[0].url) # permanent MP4 URL +print(result.data[0].duration_seconds) # 8 + +# Image-to-video +result = client.generate( + "the subject turns its head and smiles", + image_url="https://example.com/portrait.jpg", +) + +# Character-consistency video (Seedance 2.0 fast/pro). Pass a ta_xxxxxx +# asset to keep the same face across clips — either a Virtual Portrait +# (AI character, PortraitClient, $0.01) or a RealFace (real person, +# RealFaceClient, $0.01, no KYC). Mutually exclusive with image_url. +result = client.generate( + "the subject smiles warmly and waves at the camera", + model="bytedance/seedance-2.0", + real_face_asset_id="ta_abc123xyz", + resolution="1080p", + generate_audio=True, +) + +# First-and-last-frame interpolation (Seedance only): the model tweens +# from image_url (first frame) to last_frame_url (final frame). +# Priced identically to image-to-video. +result = client.generate( + "the flower blooms in golden morning light", + model="bytedance/seedance-1.5-pro", + image_url="https://example.com/bud.jpg", + last_frame_url="https://example.com/bloom.jpg", +) + +# Omni / multi-reference (Seedance 2.0 only): up to 9 reference images +# for character/style consistency. Cite them as "image 1", "image 2" +# in the prompt. Mutually exclusive with image_url / last_frame_url / +# real_face_asset_id. +result = client.generate( + "the character from image 1 walks through the city from image 2", + model="bytedance/seedance-2.0", + reference_image_urls=[ + "https://example.com/character.jpg", + "https://example.com/city.jpg", + ], +) + +# input_type — declare the seed mode you intend, and get an error instead of +# a surprise. The gateway infers the mode from the seed fields above; if your +# declared value disagrees it returns 400 WITHOUT charging. +# +# Worth it when the seed fields are built dynamically: if `image_url` comes +# back empty, the request quietly degrades to text-to-video and you still pay +# for the clip. Declaring input_type="image" turns that into a 400 instead. +result = client.generate( + "the portrait turns to face the camera", + model="bytedance/seedance-2.0", + image_url=maybe_empty_url, # if this is falsy... + input_type="image", # ...you get a 400, not a text-to-video bill +) +``` + +### Text-to-Speech & Sound Effects (`SpeechClient`) + +BlockRun Voice (ElevenLabs) — OpenAI-compatible TTS plus cinematic sound +effects. TTS price scales with character count: `(chars / 1000) × model +rate`, minimum $0.001/request. Synthesis is synchronous (<1s for Flash). + +| Model | Price | Max Input | Notes | +|-------|-------|-----------|-------| +| `elevenlabs/flash-v2.5` | $0.05/1k chars | 40k chars | ~75ms latency, 32 languages (default) | +| `elevenlabs/turbo-v2.5` | $0.05/1k chars | 40k chars | ~250ms latency, balanced quality | +| `elevenlabs/multilingual-v2` | $0.10/1k chars | 10k chars | Long-form narration, audiobooks | +| `elevenlabs/v3` | $0.10/1k chars | 5k chars | Max expressiveness, 70+ languages | +| `elevenlabs/sound-effects` | $0.05/generation | 1k chars | Sound effects up to 22s | + +```python +from blockrun_llm import SpeechClient + +client = SpeechClient() + +# Text-to-speech (voice aliases: sarah, george, laura, charlie, +# river, roger, callum, harry — or any raw ElevenLabs voice_id) +result = client.generate("Welcome to BlockRun.", voice="george") +print(result.data[0].url) # audio URL (mp3 by default) + +# Other formats / speed +result = client.generate( + "Breaking news from the world of micropayments.", + model="elevenlabs/v3", + response_format="wav", + speed=1.1, +) + +# Sound effects (flat $0.05/generation) +result = client.sound_effect("rain on a tin roof, distant thunder") + +# List voices (free, rate-limited) +voices = client.list_voices() +``` + +## Virtual Portraits (`PortraitClient`) + +`PortraitClient` wraps `POST /v1/portrait/enroll` ($0.01 USDC, one-time, +no KYC) and the free `GET /v1/wallet/
/portraits` listing endpoint. +Enroll an AI-generated character image, get back a `ta_xxxxxxxx` asset id, +then reuse it as `real_face_asset_id` on Seedance 2.0 / 2.0-fast to keep +the same character across as many videos as you want. + +> Need a **real person's** likeness instead? Use +> [`RealFaceClient`](#real-person-faces-realfaceclient) below — it +> enrolls a real face for **$0.01** via a quick on-phone liveness check, +> **no KYC**. Virtual Portraits are for AI-generated personas, mascots, +> avatars, and virtual spokespeople; RealFace is for real people. Both +> return a `ta_xxxxxx` id usable as `real_face_asset_id` on Seedance +> 2.0 / 2.0-fast. + +```python +from blockrun_llm import PortraitClient, VideoClient + +portraits = PortraitClient() +portrait = portraits.enroll( + name="My Spokesperson", + image_url="https://example.com/character.jpg", +) +print(portrait.asset_id) # ta_abcdef1234567890 +print(portrait.settlement.tx_hash) # 0x9f3a… (BaseScan-verifiable) + +# Reuse the same ta_ id on any Seedance 2.0 / 2.0-fast call +video = VideoClient() +clip = video.generate( + "the character smiles warmly and waves at the camera", + model="bytedance/seedance-2.0-fast", + real_face_asset_id=portrait.asset_id, +) +print(clip.data[0].url) + +# Browse this wallet's enrolled portraits (free, rate-limited) +listing = portraits.list_portraits() +for p in listing.portraits: + print(p.assetId, p.name, p.enrollmentTxHash) +``` + +Settlement is held until the upstream registration succeeds — if the +image fails the content filter or exceeds 10 MB, the route returns 502 +and **no payment is taken**, safe to retry with a different image. + +## Real-Person Faces (`RealFaceClient`) + +`RealFaceClient` enrolls a **real person's** likeness so you can keep the +same human face across multiple Seedance 2.0 / 2.0-fast videos. Unlike a +Virtual Portrait (an AI-generated character), RealFace proves the enroller +is the person in the photo via a brief **on-phone liveness check** (nod + +blink, ~1 minute) — **no KYC**, no government ID, no account login. + +Enrollment is a three-step flow: + +1. **`init(name)`** — *free*. Returns a `group_id` and an `h5_link` the + real person opens on their phone (render it as a QR code). +2. **phone liveness** — the rights-holder opens the link, allows camera + access, nods + blinks (~60s). Nothing is sent to BlockRun in this step. +3. **`enroll(name, image_url, group_id)`** — **$0.01 USDC**, one-time. + Uploads the face photo, matches it against the live capture, and + returns a `ta_xxxxxxxx` asset id. + +```python +from blockrun_llm import RealFaceClient, VideoClient + +faces = RealFaceClient() + +# 1. Start enrollment (free). Show init.h5_link as a QR for the person. +init = faces.init(name="Jane — Q3 spokesperson") +print(init.h5_link) # they scan + do the liveness check + +# 2. Block until they finish the phone liveness check. +faces.wait_for_active(init.group_id) + +# 3. Finalize ($0.01) with the person's face photo. +rf = faces.enroll( + name="Jane — Q3 spokesperson", + image_url="https://example.com/jane.jpg", + group_id=init.group_id, +) +print(rf.asset_id) # ta_abcdef1234567890 +print(rf.settlement.tx_hash) # 0x9f3a… (BaseScan-verifiable) + +# Reuse the ta_ id on any Seedance 2.0 / 2.0-fast call +video = VideoClient() +clip = video.generate( + "she smiles warmly and waves at the camera", + model="bytedance/seedance-2.0-fast", + real_face_asset_id=rf.asset_id, +) +print(clip.data[0].url) + +# Browse this wallet's enrolled RealFaces (free, rate-limited) +listing = faces.list_realfaces() +for r in listing.realfaces: + print(r.assetId, r.name, r.enrollmentTxHash) +``` + +Settlement happens only *after* the face is successfully matched and +registered, so failed enrollments return an error with **no charge**: +`425` = group not active yet (finish the phone check first), `422` = the +photo did not match the live capture (use a clearer front-facing photo), +`502` = upstream upload failure (safe to retry). The H5 session expires +~120s after each `init`; call `init(group_id=…)` to refresh an expired +link. + +## Voice Calls (`VoiceClient`) + +`VoiceClient` wraps `POST /v1/voice/call` (paid, $0.54/call) and +`GET /v1/voice/call/{call_id}` (free polling) — AI-powered outbound phone +calls powered by Bland.ai. The agent dials the recipient and runs a real-time +conversation based on your `task` instructions. US + Canada destinations. + +```python +from blockrun_llm import VoiceClient + +client = VoiceClient() + +# Initiate (paid $0.54) +result = client.call( + to="+14155552671", + task="You are a friendly assistant calling to confirm a 3pm dentist appointment.", + voice="maya", # nat / josh / maya / june / paige / derek / florian + max_duration=5, # minutes (1–30) +) +print(result["call_id"]) + +# Poll for transcript + recording (free) +status = client.get_status(result["call_id"]) +print(status.get("status"), status.get("recording_url")) +``` + +Bring your own caller-ID: pass `from_="+14155552671"` (must be a BlockRun +phone number you own; buy via `PhoneClient.buy_number()` or +`/v1/phone/numbers/buy`). If you omit `from_` and your wallet owns exactly one +active number, the backend auto-picks it; with multiple active numbers you'll +get a `400 ambiguous_from` and the error body lists your candidates. + +## Phone Numbers (`PhoneClient`) + +`PhoneClient` wraps `/v1/phone/*` — Twilio-backed phone lookup and +wallet-bound number provisioning. Buy a number once to use it as caller ID in +`VoiceClient`; the number is leased for 30 days and tied to your wallet. + +```python +from blockrun_llm import PhoneClient + +client = PhoneClient() + +# Carrier + line-type lookup ($0.01) +info = client.lookup("+14155552671") + +# Carrier + SIM-swap/forwarding fraud signals ($0.05) +fraud = client.lookup_fraud("+14155552671") + +# Buy a number — 30-day lease, wallet-bound ($5.00). +# Payment is held until Twilio confirms the purchase, so failed buys never charge you. +bought = client.buy_number(country="US", area_code="415") +print(bought["phone_number"], bought["expires_at"]) + +# List, renew, release +print(client.list_numbers()) # $0.001 +client.renew_number(bought["phone_number"]) # $5.00, +30 days +client.release_number(bought["phone_number"]) # free +``` + +## Surf — Crypto Intelligence (`SurfClient`) + +`SurfClient` wraps `/v1/surf/*` — the asksurf.ai partner gateway, ~83 crypto +endpoints across exchanges, on-chain SQL, prediction markets (Polymarket + +Kalshi), wallets, social analytics, and project intelligence. Tiered pricing: +$0.001 / $0.005 / $0.020 per call (tier 1 / 2 / 3). + +```python +from blockrun_llm import SurfClient + +client = SurfClient() + +# Discovery +print(SurfClient.endpoints()) # full catalog +print(client.price("market/ranking")) # 0.001 +print(client.endpoint_info("onchain/sql")) # {'method': 'POST', 'tier': 3, ...} + +# GET — pass query params (validated against the catalog) +btc_price = client.get("exchange/price", {"pair": "BTC/USDT"}) +holders = client.get("token/holders", {"address": "0x...", "chain": "ethereum"}) + +# POST — JSON body +rows = client.post("onchain/sql", {"query": "SELECT count() FROM ethereum.blocks"}) + +# Generic helper — auto-routes GET vs POST from the catalog +result = client.call("token/holders", params={"address": "0x...", "chain": "ethereum"}) +``` + +## Standalone Search (`SearchClient`) + +`SearchClient` wraps `POST /v1/search` — standalone Grok Live Search with +automatic x402 payment. Pricing: `$0.025/source + margin` +(10 sources ≈ `$0.26`). + +```python +from blockrun_llm import SearchClient + +client = SearchClient() +result = client.search( + "Latest news on x402 adoption", + sources=["x", "web"], + max_results=10, +) +print(result.summary) +for url in result.citations or []: + print(url) +``` + +## Market Data (`PriceClient`) + +Pyth-backed realtime quotes and OHLC history across crypto, FX, commodities +and 12 global equity markets. Crypto / FX / commodity are **fully free** +across price, history and list; stocks (`stocks/{market}` and the `usstock` +legacy alias) charge `$0.001` per price or history call. Pass +`require_wallet=False` when you only need free endpoints. + +```python +from blockrun_llm import PriceClient + +# Free usage — no wallet +p = PriceClient(require_wallet=False) +btc = p.price("crypto", "BTC-USD") +eur = p.price("fx", "EUR-USD") +symbols = p.list_symbols("crypto", q="sol", limit=20) + +# Paid — requires a wallet +p2 = PriceClient() +aapl = p2.price("stocks", "AAPL", market="us") +bars = p2.history( + "stocks", "AAPL", + market="us", + resolution="D", + from_ts=1_700_000_000, + to_ts=1_710_000_000, +) +``` + +Supported stock markets: `us, hk, jp, kr, gb, de, fr, nl, ie, lu, cn, ca`. + +## Multi-chain RPC (`RpcClient`) + +Standard JSON-RPC 2.0 access to 40 chains through one endpoint — Ethereum, +Base, Solana, Polygon, BSC, Arbitrum, Optimism, Avalanche, Bitcoin, Sui, and +more (powered by Tatum's RPC gateway). No API key, no per-chain endpoints: +flat **$0.002 per call** in USDC; a JSON-RPC batch charges per element. + +```python +from blockrun_llm import RpcClient + +client = RpcClient() + +# EVM chains speak eth_* JSON-RPC +block = client.call("ethereum", "eth_blockNumber") +print(int(block.result, 16)) + +balance = client.call( + "base", "eth_getBalance", + ["0x4200000000000000000000000000000000000006", "latest"], +) + +# Non-EVM chains speak their native JSON-RPC +slot = client.call("solana", "getSlot") +utxo_tip = client.call("bitcoin", "getblockcount") + +# Batch: one payment, per-element pricing ($0.002 x N) +out = client.batch("polygon", [ + {"method": "eth_blockNumber"}, + {"method": "eth_gasPrice"}, +]) + +print(block.network) # "ethereum" (canonical key from X-Network) +print(block.cache_hit) # True if served from the gateway's hot cache +print(block.tx_hash) # x402 settlement tx +``` + +40 curated chains are exported as `blockrun_llm.SUPPORTED_NETWORKS`; common +aliases (`eth`, `arb`, `op`, `matic`, `bnb`, `avax`, `sol`, `btc`, `xrp`, +`dot`, ...) resolve server-side (`blockrun_llm.NETWORK_ALIASES`). Unknown but +well-formed slugs fall through to a generic `{slug}-mainnet` gateway attempt, +so new chains work without an SDK update. Hot, low-volatility reads +(`eth_chainId`, mined blocks/receipts, `getTransaction`, ...) are served from +a method-aware gateway cache — same price, lower latency. + +## DeFi Data (Powered by DefiLlama) + +GET passthrough to DefiLlama — protocols, TVL, yields, token prices. +$0.005/call ($0.001 for price lookups). Methods live on `LLMClient` / +`AsyncLLMClient` / `SolanaLLMClient`: + +```python +client = LLMClient() + +protocols = client.defi_protocols() # all protocols + TVL +aave = client.defi_protocol("aave") # one protocol + historical TVL +chains = client.defi_chains() # TVL by chain +pools = client.defi_yields() # yield pools (APY/TVL) +prices = client.defi_prices(["coingecko:bitcoin", "base:0x833589..."]) + +# Generic escape hatch +data = client.defi("protocol/uniswap-v3") +``` + +## DEX Swaps (Powered by 0x) + +Free passthrough to the 0x Swap + Gasless APIs — **no x402 payment** +(BlockRun takes an on-chain affiliate fee on executed swaps instead). + +```python +# Indicative price, then firm quote (Permit2) +price = client.dex_price(chainId=8453, sellToken="0x...", buyToken="0x...", + sellAmount="1000000") +quote = client.dex_quote(chainId=8453, sellToken="0x...", buyToken="0x...", + sellAmount="1000000", taker="0xYourWallet") + +# Gasless flow: quote -> sign trade.eip712 -> submit -> poll +gq = client.dex_gasless_quote(chainId=8453, sellToken="0x...", + buyToken="0x...", sellAmount="1000000", + taker="0xYourWallet") +res = client.dex_gasless_submit({"trade": {...signed...}}) +status = client.dex_gasless_status(res["tradeHash"]) + +client.dex_chains() # supported swap chains +client.dex_gasless_chains() # supported gasless chains +``` + +## Cloud Compute (Powered by Modal) + +Pay-per-call sandboxed compute — create a sandbox, run commands, tear it +down. $0.01/create (CPU; $0.05 with GPU), $0.001 per exec/status/terminate. + +```python +sb = client.modal_sandbox_create(image="python:3.11") +out = client.modal_sandbox_exec(sb["sandbox_id"], ["python", "-c", "print(40+2)"]) +print(out["stdout"]) # 42 +client.modal_sandbox_terminate(sb["sandbox_id"]) +``` + +## Fund a Wallet with Fiat (Coinbase Onramp) + +Mint a one-time `pay.coinbase.com` link to buy Base USDC with a card or bank +(60+ fiat currencies) — **FREE** (no x402 payment). The signature only +authenticates the wallet, so the funding address **must equal the signing +wallet**. Base / USDC only. The returned URL is single-use and expires in +~5 min, so mint it at click time and never cache it. + +```python +link = client.onramp(client.get_wallet_address()) +print(link["url"]) # https://pay.coinbase.com/... — open to buy USDC on Base +``` + +## Prediction Markets (Powered by Predexon v2) + +Access real-time prediction market data from Polymarket, Kalshi, Limitless, sports, and Binance Futures via [Predexon](https://predexon.com). No API keys needed — pay-per-request via x402. Tier 1 endpoints are $0.001/call, Tier 2 (wallet identity / clustering) are $0.005/call. + +Each method below is available on `LLMClient` (Base), `AsyncLLMClient`, and `SolanaLLMClient`. + +### Typed helpers + +| Method | Endpoint | Tier | +|---|---|---| +| `pm_markets(**filters)` | canonical cross-venue markets | 1 | +| `pm_listings(**filters)` | venue-native executable listings | 1 | +| `pm_outcome(predexon_id)` | resolve a canonical outcome | 1 | +| `pm_polymarket_markets(**filters)` | Polymarket markets (offset pagination) | 1 | +| `pm_polymarket_events(**filters)` | Polymarket events (offset pagination) | 1 | +| `pm_polymarket_markets_keyset(**filters)` | Polymarket markets, cursor pagination | 1 | +| `pm_polymarket_events_keyset(**filters)` | Polymarket events, cursor pagination | 1 | +| `pm_polymarket_positions(**filters)` | per-wallet open positions + PnL | 1 | +| `pm_polymarket_trades(**filters)` | recent trades (token, side, price, tx_hash) | 1 | +| `pm_polymarket_leaderboard(**filters)` | trader leaderboard (window, sort_by) | 1 | +| `pm_kalshi_markets(**filters)` | Kalshi event contracts | 1 | +| `pm_limitless_markets(**filters)` | Limitless binary AMM markets | 1 | +| `pm_sports_categories()` | available sports categories | 1 | +| `pm_sports_markets(**filters)` | sports markets grouped by game | 1 | +| `pm_wallet_identity(wallet)` | identity + profile for one wallet | 2 | +| `pm_wallet_identities(addresses)` | bulk identity for ≤200 wallets (POST) | 2 | +| `pm_wallet_cluster(address)` | on-chain transfer + identity-proof cluster | 2 | + +```python +from blockrun_llm import LLMClient + +client = LLMClient() + +# Canonical cross-venue snapshot +markets = client.pm_markets(status="active", limit=20) +listings = client.pm_listings(venue="polymarket", limit=20) + +# Polymarket +events = client.pm_polymarket_events(limit=10) +positions = client.pm_polymarket_positions(user="0xABC123...") +top = client.pm_polymarket_leaderboard(window="7d", sort_by="pnl", limit=10) + +# Sports + Kalshi + Limitless +games = client.pm_sports_markets(league="NBA", limit=10) +kalshi = client.pm_kalshi_markets(limit=10) +limitless = client.pm_limitless_markets(limit=10) + +# Wallet identity (Tier 2) +profile = client.pm_wallet_identity("0xABC123...") +batch = client.pm_wallet_identities(["0xABC...", "0xDEF..."]) +cluster = client.pm_wallet_cluster("0xABC123...") +``` + +### Generic passthrough + +For endpoints without a typed helper, drop down to `pm()` (GET) or `pm_query()` +(POST). Same pricing tiers, same return shape: + +```python +candles = client.pm("polymarket/candlesticks/0x1234abcd...") # OHLCV +btc = client.pm("binance/candles/BTCUSDT") # crypto candles +pairs = client.pm("matching-markets/pairs") # cross-platform pairs +``` + +## Exa Web Search (Powered by Exa) + +Access [Exa](https://exa.ai)'s neural web search via x402. No API keys needed — pay-per-request in USDC. Available on both `LLMClient` (Base, recommended) and `SolanaLLMClient` (Solana). + +| Endpoint | Method | Price | +|---|---|---| +| `exa_search` | Neural/keyword web search | $0.01/request | +| `exa_find_similar` | Find semantically similar pages | $0.01/request | +| `exa_contents` | Extract full text from URLs | $0.002/URL | +| `exa_answer` | AI answer grounded in web search | $0.01/request | + +```python +from blockrun_llm import LLMClient + +client = LLMClient() # uses BLOCKRUN_WALLET_KEY (Base USDC) + +# Neural web search ($0.01/request) +results = client.exa_search("latest AI safety research", numResults=5) +results = client.exa_search("bitcoin ETF news", category="news", numResults=10) + +# Find similar pages ($0.01/request) +similar = client.exa_find_similar("https://openai.com/research/gpt-4", numResults=5) + +# Extract content from URLs ($0.002/URL) +content = client.exa_contents(["https://arxiv.org/abs/2303.08774"]) +content = client.exa_contents( + ["https://example.com/page1", "https://example.com/page2"], + text=True, + highlights=True, +) + +# AI-generated answer from live web ($0.01/request) +answer = client.exa_answer("What is the current state of AI safety research?") + +# Generic proxy for any Exa endpoint +result = client.exa("search", {"query": "transformer architecture", "numResults": 5}) +``` + +For Solana payments use `from blockrun_llm import SolanaLLMClient` — same method +names, same call shape; the Solana gateway requires the backend to be configured +with `EXA_API_KEY`, so prefer Base unless you need SOL/SPL settlement. + +## Standalone Search + +Search web, X/Twitter, and news without using a chat model: + +```python +from blockrun_llm import LLMClient + +client = LLMClient() + +result = client.search("latest AI agent frameworks 2026") +print(result.summary) +for cite in result.citations or []: + print(f" - {cite}") + +# Filter by source type and date range +result = client.search( + "BlockRun x402", + sources=["web", "x"], + from_date="2026-01-01", + max_results=5, +) +``` + +## Image Editing (img2img) + +Edit existing images with text prompts. The source `image` must be a +`data:image/...;base64,...` data URI (plain URLs are not accepted): + +```python +from blockrun_llm import LLMClient, ImageClient + +# Via LLMClient +client = LLMClient() +result = client.image_edit( + prompt="Make the sky purple and add northern lights", + image="data:image/png;base64,...", # base64 data URI + model="openai/gpt-image-1", +) +print(result.data[0].url) + +# Via ImageClient +img_client = ImageClient() +result = img_client.edit("Add a rainbow", image="data:image/png;base64,...") + +# Multi-image fusion — pass a list of data URIs (e.g. a reference + a logo). +# openai/* accepts up to 4 source images, google/* up to 3. +result = img_client.edit( + "Place the logo on the model's t-shirt", + image=["data:image/png;base64,...", "data:image/png;base64,..."], + model="google/nano-banana", +) +print(result.data[0].url) +``` + +## Usage Examples + +### Simple Chat + +```python +from blockrun_llm import LLMClient + +client = LLMClient() # Uses BLOCKRUN_WALLET_KEY (never sent to server) + +response = client.chat("openai/gpt-5.2", "Explain quantum computing") +print(response) + +# With system prompt +response = client.chat( + "anthropic/claude-sonnet-4.6", + "Write a haiku", + system="You are a creative poet." +) +``` + +### JSON Mode & Stop Sequences + +`response_format` and `stop` are OpenAI-compatible and honored across **all** providers by +the gateway — native for OpenAI/Azure, and emulated for Anthropic/Bedrock (a raw-JSON system +instruction with code-fence stripping for JSON mode, `stop` mapped to `stop_sequences`). + +```python +import json +from blockrun_llm import LLMClient + +client = LLMClient() + +# JSON mode — guaranteed parseable JSON, no markdown fences +response = client.chat( + "openai/gpt-4o", + "List 3 primary colors as a JSON array under key 'colors'.", + response_format={"type": "json_object"}, +) +print(json.loads(response)) # {'colors': ['red', 'green', 'blue']} + +# Stop sequences (str or list, up to 4) +result = client.chat_completion( + "openai/gpt-5.2", + [{"role": "user", "content": "Count: Alpha Beta Gamma"}], + stop=["Beta"], +) +print(result.choices[0].message.content) # "Count: Alpha " +``` + +### Real-time Search (Live Search) + +**Note:** Live Search can take 30-120+ seconds as it searches multiple sources. The SDK automatically uses a 5-minute timeout for search requests. + +```python +from blockrun_llm import LLMClient + +client = LLMClient() + +# Simple: Enable live search with search=True (default 10 sources, ~$0.26) +response = client.chat( + "openai/gpt-5.2", + "What are the latest posts from @blockrunai?", + search=True +) +print(response) + +# Custom: Limit sources to reduce cost (5 sources, ~$0.13) +response = client.chat( + "openai/gpt-5.2", + "What's trending on X?", + search_parameters={"mode": "on", "max_search_results": 5} +) + +# Custom timeout (if 5 min isn't enough) +client = LLMClient(search_timeout=600.0) # 10 minutes +``` + +### Check Spending + +```python +from blockrun_llm import LLMClient + +client = LLMClient() + +response = client.chat("openai/gpt-5.2", "Explain quantum computing") +print(response) + +# Check how much was spent +spending = client.get_spending() +print(f"Spent ${spending['total_usd']:.4f} across {spending['calls']} calls") +``` + +### Full Chat Completion + +```python +from blockrun_llm import LLMClient + +client = LLMClient() # Uses BLOCKRUN_WALLET_KEY (never sent to server) + +messages = [ + {"role": "system", "content": "You are a helpful assistant."}, + {"role": "user", "content": "How do I read a file in Python?"} +] + +result = client.chat_completion("openai/gpt-5.2", messages) +print(result.choices[0].message.content) +``` + +### Async Usage + +```python +import asyncio +from blockrun_llm import AsyncLLMClient + +async def main(): + async with AsyncLLMClient() as client: + # Simple chat + response = await client.chat("openai/gpt-5.2", "Hello!") + print(response) + + # Multiple requests concurrently + tasks = [ + client.chat("openai/gpt-5.2", "What is 2+2?"), + client.chat("anthropic/claude-sonnet-4.6", "What is 3+3?"), + client.chat("google/gemini-2.5-flash", "What is 4+4?"), + ] + responses = await asyncio.gather(*tasks) + for r in responses: + print(r) + +asyncio.run(main()) +``` + +### List Available Models + +```python +from blockrun_llm import LLMClient + +client = LLMClient() +models = client.list_models() + +for model in models: + print(f"{model['id']}: ${model['inputPrice']}/M input, ${model['outputPrice']}/M output") +``` + +## Testnet Usage + +For development and testing without real USDC, use the testnet: + +```python +from blockrun_llm import testnet_client + +# Create testnet client (uses Base Sepolia) +client = testnet_client() # Uses BLOCKRUN_WALLET_KEY + +# Chat with testnet model +response = client.chat("openai/gpt-oss-20b", "Hello!") +print(response) + +# Check testnet USDC balance +balance = client.get_balance() +print(f"Testnet USDC: ${balance:.4f}") +``` + +### Testnet Setup + +1. Get testnet ETH from [Alchemy Base Sepolia Faucet](https://www.alchemy.com/faucets/base-sepolia) +2. Get testnet USDC from [Circle USDC Faucet](https://faucet.circle.com/) +3. Set your wallet key: `export BLOCKRUN_WALLET_KEY=0x...` + +### Available Testnet Models + +- `openai/gpt-oss-20b` - $0.001/request (flat price) +- `openai/gpt-oss-120b` - $0.002/request (flat price) + +### Manual Testnet Configuration + +```python +from blockrun_llm import LLMClient + +# Or configure manually +client = LLMClient(api_url="https://testnet.blockrun.ai/api") +response = client.chat("openai/gpt-oss-20b", "Hello!") +``` + +## Billing & Cost Tracking + +Every paid call appends one line to `~/.blockrun/cost_log.jsonl` capturing +timestamp, endpoint, cost, and (when available) `model`, `wallet`, `network`, +and `client_kind`. The SDK ships a small reader / exporter on top so you can +audit spending without leaving the Python ecosystem. + +### CLI + +```bash +# Aggregated summary, default grouped by endpoint +python -m blockrun_llm.billing summary + +# Group by model / month / wallet / network / client_kind / day +python -m blockrun_llm.billing summary --group-by model +python -m blockrun_llm.billing summary --group-by month --from 2026-04-01 + +# Filter by wallet (when one machine drives multiple keys) +python -m blockrun_llm.billing summary --wallet 0xCC8c... --network base-mainnet + +# Export per-call records +python -m blockrun_llm.billing export csv --from 2026-05-01 --output may.csv +python -m blockrun_llm.billing export json --to 2026-05-09 +``` + +### Python API + +```python +from blockrun_llm import ( + get_cost_log_summary, + export_cost_log_csv, + export_cost_log_json, +) + +summary = get_cost_log_summary(group_by="model", from_date="2026-04-01") +print(summary["total_usd"], summary["calls"]) +for model, slot in summary["groups"].items(): + print(f" {model:40s} {slot['calls']:>5} ${slot['cost_usd']:.4f}") + +# Returns CSV / JSON text; pass output_path to also write to disk +csv_text = export_cost_log_csv("bill.csv", from_date="2026-05-01") +json_text = export_cost_log_json(from_date="2026-05-01") +``` + +### Example output + +Real session — four cheap chat calls across providers, then queried by model: + +``` +$ python -m blockrun_llm.billing summary --from 2026-05-10 --group-by model +================================================================ +BLOCKRUN — LOCAL COST LOG SUMMARY +================================================================ + log file : /Users/me/.blockrun/cost_log.jsonl + from : 2026-05-10 + group_by : model + total : $0.0070 (9 calls) + + KEY CALLS COST + ---------------------------- ------- ---------- + deepseek/deepseek-chat 2 $0.0020 + google/gemini-2.5-flash-lite 1 $0.0010 + anthropic/claude-haiku-4.5 1 $0.0010 + zai/glm-5-turbo 1 $0.0010 + unknown 4 $0.0020 +``` + +The four `unknown` rows are pre-existing entries from before this release — +they had only `{ts, endpoint, cost_usd}` so the model column reads `unknown`. +Calls made after upgrading carry the full metadata (wallet / network / +client_kind / model). CSV export shows it directly: + +``` +$ python -m blockrun_llm.billing export csv --from 2026-05-10 | head -3 +ts_iso,endpoint,model,wallet,network,client_kind,cost_usd +2026-05-10T03:38:28.198937+00:00,/v1/chat/completions,deepseek/deepseek-chat,0xCC8c...5EF8,base-mainnet,LLMClient,0.001 +2026-05-10T03:38:31.192060+00:00,/v1/chat/completions,google/gemini-2.5-flash-lite,0xCC8c...5EF8,base-mainnet,LLMClient,0.001 +``` + +### Scope + +The cost log is per-machine. It records calls made by this Python SDK only — +calls from other clients (TS SDK, MCP, raw curl) are not included. For +organization-wide billing, query the gateway's authoritative ledger. + +## Transaction Log (project-local, on-chain match) + +The cost log above lives in `~/.blockrun/` and is hash-keyed JSON. When you'd +rather have an **eyeballable text log next to your code** that matches the +chain row-for-row, opt into the per-transaction log: + +```python +from blockrun_llm import LLMClient + +# Default: writes ./log/transactions.log +client = LLMClient(transaction_log=True) + +# Or pick a path +client = LLMClient(transaction_log="./var/blockrun.log") + +# Or via env var: BLOCKRUN_TX_LOG=1 (default dir) +# BLOCKRUN_TX_LOG=./var/blockrun.log +``` + +Works the same on `AsyncLLMClient`, `SolanaLLMClient`, and `AsyncSolanaLLMClient`. + +Every paid call appends one row. Example: + +``` +2026-05-21 15:44:46 chat anthropic/claude-sonnet-4.6 in= 3 out=4 $0.034137 0x6513d128… +2026-05-20 04:34:17 chat openai/gpt-5.5 in= 14 out=18 $0.001000 0x421796a3… +``` + +Columns: timestamp · endpoint tag (`chat`/`image`/`video`/`search`/…) · model +(padded to 30) · `in=` prompt tokens · `out=` completion tokens · `$cost` to +6 decimals · first 10 chars of the **on-chain settlement hash**. + +### Why it matches the chain + +The hash comes from the `X-PAYMENT-RESPONSE` header the x402 facilitator +returns after settlement — Base txs use `transaction`, Solana uses +`signature`. Both normalise to the truncated `0x…` / signature shown in +the row, so each line is verifiable in one click: + +- Base mainnet → `https://basescan.org/tx/` +- Solana mainnet → `https://solscan.io/tx/` + +Cached / free responses don't hit the chain, so they show `(no-tx)` instead. + +### Scope and trade-offs + +- **Independent of the cache layer.** Enabling the log does not change + `~/.blockrun/cache/`, `~/.blockrun/data/`, or `~/.blockrun/cost_log.jsonl`. +- **Best-effort writes.** OSErrors are swallowed; a read-only filesystem can't + break a paid call. +- **Plain text only.** If you need a structured ledger as well, query + `~/.blockrun/cost_log.jsonl` via `blockrun_llm.billing`. + +### Programmatic access + +```python +from blockrun_llm import TransactionLogger, format_row + +# Tail the project log +logger = TransactionLogger("./log") +for row in logger.entries()[-5:]: + print(row) + +# Build your own row (e.g. for tests or custom adapters) +print(format_row( + endpoint="/v1/chat/completions", + model="openai/gpt-5.5", + in_tokens=14, + out_tokens=18, + cost_usd=0.001, + tx_hash="0x421796a3deadbeef", +)) +``` + +## Environment Variables + +One credential is required — an API key **or** a wallet key. Everything else is +optional. + +| Variable | Description | Default | +|----------|-------------|---------| +| `BLOCKRUN_API_KEY` | API key from [user.blockrun.ai](https://user.blockrun.ai) (`brk_live_…`). **Takes precedence over every wallet variable.** | — | +| `SOLANA_WALLET_KEY` | bs58 Solana wallet key, for `SolanaLLMClient` | falls back to `~/.*/solana-wallet.json`, then `~/.blockrun/.solana-session` | +| `BLOCKRUN_WALLET_KEY` | Base chain wallet private key | falls back to `~/.blockrun/.session` | +| `BASE_CHAIN_WALLET_KEY` | Alias for `BLOCKRUN_WALLET_KEY` | — | +| `BLOCKRUN_API_KEY_URL` | Override the API-key gateway | `https://api.blockrun.ai` | +| `BLOCKRUN_API_URL` | Override the Base x402 gateway | `https://blockrun.ai/api` | +| `SOLANA_RPC_URL` / `SOLANA_RPC_HEADERS` / `SOLANA_RPC_API_KEY` | RPC for blockhash + mint info while signing | BlockRun's free proxy | +| `BLOCKRUN_CHAT_TIMEOUT` | Chat HTTP timeout, in seconds | `600` | +| `BLOCKRUN_MAX_COST_PER_CALL` / `BLOCKRUN_MAX_SESSION_COST` | Opt-in spend limits (wallet rail) | unlimited | + +`BLOCKRUN_API_KEY_URL` is deliberately not `BLOCKRUN_API_URL`: that one names an +x402 gateway, and an API-key client must never follow it and send your key to a +host you configured for a different rail. + +## Setting Up Your Wallet + +1. Create a wallet on Base network (Coinbase Wallet, MetaMask, etc.) +2. Get some ETH on Base for gas (small amount, ~$1) +3. Get USDC on Base for API payments +4. Export your private key and set it as `BLOCKRUN_WALLET_KEY` + +```bash +# .env file +BLOCKRUN_WALLET_KEY=0x...your_private_key_here +``` + +## Error Handling + +```python +from blockrun_llm import LLMClient, APIError, PaymentError + +client = LLMClient() + +try: + response = client.chat("openai/gpt-5.2", "Hello!") +except PaymentError as e: + print(f"Payment failed: {e}") + # Check your USDC balance +except APIError as e: + print(f"API error ({e.status_code}): {e}") +``` + +## Testing + +### Running Unit Tests + +Unit tests do not require API access or funded wallets: + +```bash +pytest tests/unit # Run unit tests only +pytest tests/unit --cov # Run with coverage report +pytest tests/unit -v # Verbose output +``` + +### Running Integration Tests + +Integration tests call the production API and require: +- A funded Base wallet with USDC ($1+ recommended) +- `BLOCKRUN_WALLET_KEY` environment variable set +- Estimated cost: ~$0.05 per test run + +```bash +export BLOCKRUN_WALLET_KEY=0x... +pytest tests/integration # Run integration tests only +pytest # Run all tests +``` + +Integration tests are automatically skipped if `BLOCKRUN_WALLET_KEY` is not set. + +## Security + +### Private Key Safety + +- **Private key stays local**: Your key is only used for signing on your machine +- **No custody**: BlockRun never holds your funds +- **Verify transactions**: All payments are on-chain and verifiable + +### Best Practices + +**Private Key Management:** +- Use environment variables, never hard-code keys +- Use dedicated wallets for API payments (separate from main holdings) +- Set spending limits by only funding payment wallets with small amounts +- Never commit `.env` files to version control +- Rotate keys periodically + +**Input Validation:** +The SDK validates all inputs before API requests: +- Private keys (format, length, valid hex) +- API URLs (HTTPS required for production, HTTP allowed for localhost) +- Model names and parameters (ranges for max\_tokens, temperature, top\_p) + +**Error Sanitization:** +API errors are automatically sanitized to prevent sensitive information leaks. + +**Monitoring:** +```python +address = client.get_wallet_address() +print(f"View transactions: https://basescan.org/address/{address}") +``` + +**Keep Updated:** +```bash +pip install --upgrade blockrun-llm # Get security patches +``` + +## Agent Wallet Setup + +One-line setup for agent runtimes (Claude Code skills, MCP servers, etc.): + +```python +from blockrun_llm import setup_agent_wallet + +# Auto-creates wallet if none exists, returns ready client +client = setup_agent_wallet() +response = client.chat("openai/gpt-5.4", "Hello!") +``` + +For Solana: + +```python +from blockrun_llm import setup_agent_solana_wallet + +client = setup_agent_solana_wallet() +response = client.chat("anthropic/claude-sonnet-4.6", "Hello!") +``` + +Check wallet status: + +```python +from blockrun_llm import status + +status() +# Wallet: 0xCC8c...5EF8 +# Balance: $5.30 USDC +``` + +## Wallet Discovery and Migration + +The SDK can discover compatible wallets for an explicit, user-confirmed +migration. It never automatically makes a discovered provider wallet active: + +```python +from blockrun_llm.wallet import scan_wallets +from blockrun_llm.solana_wallet import scan_solana_wallets + +# Scans ~/./wallet.json for Base wallets +base_wallets = scan_wallets() + +# Scans ~/./solana-wallet.json +sol_wallets = scan_solana_wallets() +``` + +`get_or_create_wallet()` always uses `~/.blockrun/.session` (or an explicit +wallet environment variable, or the legacy `~/.blockrun/wallet.key`). Review +the discovered addresses and import one explicitly if you intend to switch +wallets. + +### Upgrading from a provider wallet + +Earlier versions adopted the most recently written provider wallet +automatically. If you relied on that, the first run after upgrading creates a +fresh BlockRun wallet and prints the addresses it found, so you can import the +one you actually own: + +``` +NOTICE: BlockRun created a new wallet, but also found existing wallet(s) +belonging to other applications on this system: + + 0x88f9B82462f6C4bf4a0Fb15e5c3971559a316e7f +... +``` + +Adopt one deliberately: + +```python +from blockrun_llm import list_discovered_wallets, import_wallet + +for w in list_discovered_wallets(): + print(w["address"], "from", w["source"]) + +import_wallet("0x88f9B82462f6C4bf4a0Fb15e5c3971559a316e7f") +``` + +`import_wallet()` writes your current wallet to +`~/.blockrun/.session.backup-` before switching, so adopting a wallet +never strands funds in the old one. Solana: `list_discovered_solana_wallets()` +and `import_solana_wallet()`. + +Addresses shown are derived from the discovered key itself, and `import_wallet()` +matches on that derived address — so a wallet file cannot claim an address it +cannot sign for, nor be adopted by one. `list_discovered_wallets()` never returns +private keys. + +For a single run without changing anything, use +`export BLOCKRUN_WALLET_KEY=`. + +## Response Caching + +The SDK caches responses to avoid duplicate payments: + +```python +from blockrun_llm import clear_cache + +# Automatic TTLs by endpoint: +# - Prediction Markets: 30 minutes +# - Search: 15 minutes +# - Models: 24 hours +# - Chat/Image: no cache (every call is unique) + +# Manual cache management +removed = clear_cache() # Remove all cached responses +``` + +Per-session spending is also available on any client (see also +[Billing & Cost Tracking](#billing--cost-tracking) for the full surface): + +```python +from blockrun_llm import LLMClient + +client = LLMClient() +response = client.chat("openai/gpt-5.2", "Hello!") + +spending = client.get_spending() +print(f"Session: ${spending['total_usd']:.4f} across {spending['calls']} calls") +``` + +## Anthropic SDK Compatibility + +Use the official Anthropic Python SDK with BlockRun's API gateway and automatic x402 payments: + +```bash +pip install blockrun-llm[anthropic] +``` + +```python +from blockrun_llm import AnthropicClient + +client = AnthropicClient() # Auto-detects wallet, auto-pays + +response = client.messages.create( + model="claude-sonnet-4-6", + max_tokens=1024, + messages=[{"role": "user", "content": "Hello!"}] +) +print(response.content[0].text) + +# Works with any BlockRun model in Anthropic format +response = client.messages.create( + model="openai/gpt-5.4", + max_tokens=1024, + messages=[{"role": "user", "content": "Hello from GPT!"}] +) +``` + +The `AnthropicClient` wraps `anthropic.Anthropic` with a custom httpx transport that handles x402 payment signing transparently. Your private key never leaves your machine. + +## Links + +- [Website](https://blockrun.ai) +- [Documentation](https://github.com/BlockRunAI/awesome-blockrun/tree/main/docs) +- [GitHub](https://github.com/blockrunai/blockrun-llm) +- [Telegram](https://t.me/+mroQv4-4hGgzOGUx) + +## Frequently Asked Questions + +### What is blockrun-llm? +blockrun-llm is a Python SDK that provides pay-per-request access to 43+ large language models from OpenAI, Anthropic, Google, DeepSeek, NVIDIA, ZAI, and more. It uses the x402 protocol for automatic USDC micropayments — no API keys, no subscriptions, no vendor lock-in. + +### How does payment work? +When you make an API call, the SDK automatically handles x402 payment. It signs a USDC transaction locally using your wallet private key (which never leaves your machine), and includes the payment proof in the request header. Settlement is non-custodial and instant on Base or Solana. + +### What is smart routing / Router Core? +Router Core is BlockRun's built-in routing engine — shared with the TypeScript SDK and the gateway, so the same request routes the same way everywhere. It scores your request across 15 dimensions, drops every model that can't actually handle it (context, output length, tools, vision), then picks the cheapest capable one and keeps the rest as a fallback chain. Routing happens locally in under 1ms and makes no extra model call. It can save up to 84% on LLM costs compared to using premium models for every request. + +### How much does it cost? +Pay only for what you use. Prices start at **FREE** (11 NVIDIA-hosted models). Paid models start at $0.10/M tokens. There are no minimums, subscriptions, or monthly fees. $5 in USDC gets you thousands of requests. + +### Can I use it with Solana? +Yes. Install with `pip install blockrun-llm[solana]` and use `SolanaLLMClient` instead of `LLMClient`. Same API, different payment chain. + +## License + +MIT + +### Changing payment methods safely + +Register at [user.blockrun.ai](https://user.blockrun.ai), add credit in +[Credits](https://user.blockrun.ai/dashboard/credits), and create a key in +[API keys](https://user.blockrun.ai/dashboard/keys). Set `BLOCKRUN_API_KEY` +or pass the key as the client's credential. Check activity and actual charges +in the dashboard; local cost summaries may omit charges without a gateway receipt. + +An explicit wallet credential chooses wallet payments even when `BLOCKRUN_API_KEY` +is set. Choose the Solana wallet client for Solana, or the Base wallet client for +Base. A `BLOCKRUN_API_KEY` set to something that is not a `brk_` key fails instead +of silently selecting a wallet and spending USDC you meant to keep. Blank counts +as unset, so `BLOCKRUN_API_KEY=` in a `.env` file or an unpopulated CI secret +still falls back to the wallet. Create a new client when changing credentials; +an existing client keeps its original account. + +The optional `AnthropicClient` also accepts a BlockRun API key as its credential +or through `BLOCKRUN_API_KEY` (`pip install 'blockrun-llm[anthropic]'`). Automatic +retries are disabled by default on **both** payment rails, because a failed +response may follow a billable request. It matters most on the wallet rail, where +the x402 transport signs a fresh payment for every 402 it sees: at the Anthropic +SDK's own default of 2 retries, one `messages.create()` that 5xxs after the +gateway settled would sign and settle three separate on-chain transfers. Pass +`max_retries=` explicitly to opt back in. + +The optional integration currently supports Anthropic SDK **0.x**. The extra +pins `<1` because Anthropic 1.x moved to a different HTTP transport; upgrading +that dependency independently would break both account and wallet clients. diff --git a/SEEDANCE_CAPABILITIES.md b/SEEDANCE_CAPABILITIES.md new file mode 100644 index 0000000..7a470dc --- /dev/null +++ b/SEEDANCE_CAPABILITIES.md @@ -0,0 +1,76 @@ +# Seedance input and output capabilities + +The three gateways use the same public generation fields. Wallet authentication, +API-key holds, signed poll URLs and settlement timing are unchanged. + +## Supported combinations + +| Model | First + last frame | Reference images | Reference video/audio combinations | +| --- | --- | --- | --- | +| Seedance 1.5-pro | Yes | No | No | +| Seedance 2.0 / Fast / Mini | Yes | 1–9 | Image + video, image + audio, video + audio, or all three; 1–3 clips of each type | +| Seedance 2.5 | Yes | 1–30 | Still held pending render/cost verification | + +`image_url` means a first-frame seed. For a character/style image alongside a +reference video, use `reference_image_urls`, not `image_url`. Frame seeding and +reference mode remain mutually exclusive. Seedance 2.0 audio references require +at least one reference image or video. Upstream media duration, size and content +constraints still apply; accepting a URL does not verify the remote file. + +```json +{ + "model": "bytedance/seedance-2.0-fast", + "prompt": "Use image 1 for the character and video 1 for the motion", + "duration_seconds": 5, + "reference_image_urls": ["https://example.com/character.png"], + "reference_videos": [{"url": "https://example.com/motion.mp4"}], + "input_type": "reference", + "return_last_frame": true +} +``` + +POST to `/v1/videos/generations` or `/api/v1/videos/generations`. Native +`content[]` also works on these endpoints and `/v1/videos`: `reference_image`, +`reference_video`, `reference_audio`, `first_frame`, and `last_frame` roles map +to the corresponding validated flat fields. A role-less single image keeps its +existing first-frame meaning. Alternatively use `frame_images` with `frame_type` +or typed `input_references` with role `reference`. Use one media syntax per +request; conflicting aliases or media fields return 400 before payment. + +## Additional output controls + +| Field | Models | Values | +| --- | --- | --- | +| `bitrate_mode` | Seedance 2.x | `standard`, `high` | +| `output_format` | Seedance 2.5 | `mp4`, `mov` | +| `camera_fixed` | Seedance 1.5-pro | Boolean | +| `safety_identifier` | Seedance family | String | +| `return_last_frame` | Seedance family | Boolean | + +When the upstream returns a last frame, completed `data[0]` includes +`last_frame_url` and `last_frame_backed_up`. The frame uses the same storage +backup/fallback semantics as the video. Solana starts the copy without delaying +settlement, preserving its blockhash timing. An upstream that omits the frame +produces no invented frame URL. Failover is refused when it would drop a +requested control or an asset reference. + +Python uses the snake_case fields above. TypeScript uses `referenceImageUrls`, +`referenceVideos`, `referenceAudios`, `bitrateMode`, `outputFormat`, `cameraFixed`, +`safetyIdentifier`, `returnLastFrame`, and `inputType`. MCP exposes snake_case +fields and reserves the existing reference-media surcharge before payment. + +## Operational limits + +`R2V_ENABLED=false` still refuses NEW reference-video/audio jobs with 503. +This change does not modify deployment configuration or re-enable production. +Jobs already accepted remain pollable. Image-only references are not subject +to that operational switch. + +Automatic duration (`-1`), 2.5 editing/extension task modes, 2.5 reference media, +2.5 1080p, draft/flex service tiers, callbacks, and task-list/cancel APIs remain +outside this change. The first group needs verified cost/output bounds; the +lifecycle features need a separate ownership and settlement design. Known +unsupported request controls are rejected instead of silently discarded. + +New behavior is covered by local request-contract and mocked payment-lifecycle +tests. New paid upstream renders and production rollout are separate checks. diff --git a/VERSION b/VERSION new file mode 100644 index 0000000..092afa1 --- /dev/null +++ b/VERSION @@ -0,0 +1 @@ +1.17.0 diff --git a/blockrun_llm/__init__.py b/blockrun_llm/__init__.py index 8a95900..db25c43 100644 --- a/blockrun_llm/__init__.py +++ b/blockrun_llm/__init__.py @@ -1,31 +1,333 @@ -""" -BlockRun LLM SDK - Pay-per-request AI via x402 on Base - -Usage: - from blockrun_llm import LLMClient - - client = LLMClient() # Uses BLOCKRUN_WALLET_KEY from env - response = client.chat("gpt-4o", "Hello!") - print(response) - -Async usage: - from blockrun_llm import AsyncLLMClient - - async with AsyncLLMClient() as client: - response = await client.chat("gpt-4o", "Hello!") - print(response) -""" - -from .client import LLMClient, AsyncLLMClient -from .types import ChatMessage, ChatResponse, Model, APIError, PaymentError - -__version__ = "0.1.0" -__all__ = [ - "LLMClient", - "AsyncLLMClient", - "ChatMessage", - "ChatResponse", - "Model", - "APIError", - "PaymentError", -] +""" +BlockRun LLM SDK - Pay-per-request AI via x402 on Base (USDC) + +For developers (bring your own wallet): + from blockrun_llm import LLMClient + + client = LLMClient() # Uses BLOCKRUN_WALLET_KEY from env + response = client.chat("openai/gpt-5.2", "Hello!") + print(response) + +For agents (Claude Code skills, auto-creates wallet): + from blockrun_llm import setup_agent_wallet + + client = setup_agent_wallet() # Auto-creates wallet, shows QR + response = client.chat("openai/gpt-5.2", "Hello!") + print(response) + +Async usage: + from blockrun_llm import AsyncLLMClient + + async with AsyncLLMClient() as client: + response = await client.chat("openai/gpt-5.2", "Hello!") + print(response) + +Image generation: + from blockrun_llm import ImageClient + + client = ImageClient() + result = client.generate("A cute cat wearing a space helmet") + print(result.data[0].url) + +Video generation: + from blockrun_llm import VideoClient + + client = VideoClient() + result = client.generate("a red apple slowly spinning on a wooden table") + print(result.data[0].url) # permanent MP4 URL + +Text-to-speech (BlockRun Voice / ElevenLabs): + from blockrun_llm import SpeechClient + + client = SpeechClient() + result = client.generate("Welcome to BlockRun.", voice="sarah") + print(result.data[0].url) # audio URL + +Multi-chain RPC (40+ chains, $0.002/call): + from blockrun_llm import RpcClient + + client = RpcClient() + block = client.call("ethereum", "eth_blockNumber") + print(block.result) + +Other Chains: + - Solana (USDC): Use SolanaLLMClient (pip install blockrun-llm[solana]) +""" + +from __future__ import annotations + +from .anthropic_client import AnthropicClient +from .apikey import ( + DEFAULT_API_KEY_URL, + ENV_API_KEY, + ENV_API_KEY_URL, + PAYMENT_MODE_API_KEY, + PAYMENT_MODE_WALLET, + is_api_key, + resolve_api_key, +) +from .cache import ( + clear_cache, + export_cost_log_csv, + export_cost_log_json, + get_cost_log_summary, +) +from .client import ( + AsyncLLMClient, + LLMClient, + async_testnet_client, + list_image_models, + list_models, + testnet_client, +) +from .image import ImageClient +from .music import MusicClient +from .phone import PhoneClient +from .portrait import PortraitClient +from .price import PriceClient +from .realface import RealFaceClient +from .rpc import NETWORK_ALIASES, SUPPORTED_NETWORKS, RpcClient +from .search import SearchClient +from .solana_client import AsyncSolanaLLMClient, SolanaLLMClient +from .solana_wallet import ( + create_solana_wallet, + format_solana_wallet_migration_notice, + generate_solana_qr_ascii, + get_or_create_solana_wallet, + get_solana_public_key, + get_solana_usdc_balance, + import_solana_wallet, + list_discovered_solana_wallets, + load_solana_wallet, + open_solana_wallet_qr, + scan_solana_wallets, + setup_agent_solana_wallet, +) +from .speech import SpeechClient +from .surf import SurfClient +from .tx_log import TransactionLogger, decode_settlement_header, format_row +from .types import ( + APIError, + AudioModel, + AudioTrack, + # Smart routing types + CandidateScore, + ChatChunkChoice, + ChatChunkDelta, + ChatChunkFunctionCall, + ChatChunkToolCall, + ChatCompletionChunk, + ChatMessage, + ChatResponse, + ImageData, + ImageModel, + ImageResponse, + Model, + # Music / Audio types + MusicResponse, + NewsSearchSource, + PaymentError, + # Virtual Portrait types + PortraitEnrollment, + PortraitList, + PortraitListItem, + PortraitSettlement, + PortraitUsage, + PriceBar, + PriceHistoryResponse, + # Pyth market data types + PricePoint, + RealFaceEnrollment, + # RealFace types + RealFaceInit, + RealFaceList, + RealFaceListItem, + RealFaceStatus, + RetiredEndpointError, + # Smart routing types + RoutingDecision, + RpcError, + # Multi-chain RPC types + RpcResponse, + RssSearchSource, + # Live Search types + SearchParameters, + # Standalone search + SearchResult, + SmartChatCompletionResponse, + SmartChatResponse, + SpeechAudio, + # Speech (TTS / sound effects) types + SpeechResponse, + SpendLimitError, + SymbolListResponse, + VideoClip, + VideoModel, + # Video types + VideoResponse, + WebSearchSource, + XSearchSource, +) +from .video import VideoClient +from .voice import VoiceClient +from .wallet import ( + WALLET_DIR, + WALLET_FILE, + format_error_message, + format_funding_message_compact, + format_needs_funding_message, + format_wallet_created_message, + format_wallet_migration_notice, + generate_wallet_qr_ascii, + get_eip681_uri, + get_or_create_wallet, + get_payment_links, + get_wallet_address, + import_wallet, + list_discovered_wallets, + load_wallet, + open_wallet_qr, + save_wallet_qr, + scan_wallets, + setup_agent_wallet, # Entry point for agents (auto-creates wallet) + status, # One-command verification +) +from .wallet import ( + create_wallet as generate_wallet, # User-friendly alias +) + +__version__ = "1.17.0" +__all__ = [ + "DEFAULT_API_KEY_URL", + "ENV_API_KEY", + "ENV_API_KEY_URL", + "NETWORK_ALIASES", + "PAYMENT_MODE_API_KEY", + "PAYMENT_MODE_WALLET", + "SUPPORTED_NETWORKS", + "WALLET_DIR", + "WALLET_FILE", + "APIError", + "AnthropicClient", + "AsyncLLMClient", + "AsyncSolanaLLMClient", + "AudioModel", + "AudioTrack", + # Smart routing types + "CandidateScore", + "ChatChunkChoice", + "ChatChunkDelta", + "ChatChunkFunctionCall", + "ChatChunkToolCall", + "ChatCompletionChunk", + "ChatMessage", + "ChatResponse", + "ImageClient", + "ImageData", + "ImageModel", + "ImageResponse", + "LLMClient", + "Model", + "MusicClient", + "MusicResponse", + "NewsSearchSource", + "PaymentError", + "PhoneClient", + "PortraitClient", + "PortraitEnrollment", + "PortraitList", + "PortraitListItem", + "PortraitSettlement", + "PortraitUsage", + "PriceBar", + "PriceClient", + "PriceHistoryResponse", + # Pyth market data types + "PricePoint", + "RealFaceClient", + "RealFaceEnrollment", + "RealFaceInit", + "RealFaceList", + "RealFaceListItem", + "RealFaceStatus", + "RetiredEndpointError", + # Smart routing types + "RoutingDecision", + "RpcClient", + "RpcError", + # Multi-chain RPC types + "RpcResponse", + "RssSearchSource", + "SearchClient", + # Live Search types + "SearchParameters", + # Standalone search + "SearchResult", + "SmartChatCompletionResponse", + "SmartChatResponse", + "SolanaLLMClient", + "SpeechAudio", + "SpeechClient", + "SpeechResponse", + "SpendLimitError", + "SurfClient", + "SymbolListResponse", + # Per-transaction log (opt-in, project-local ./log/) + "TransactionLogger", + "VideoClient", + "VideoClip", + "VideoModel", + "VideoResponse", + "VoiceClient", + "WebSearchSource", + "XSearchSource", + "async_testnet_client", + # Cache + billing utilities + "clear_cache", + "create_solana_wallet", + "decode_settlement_header", + "export_cost_log_csv", + "export_cost_log_json", + "format_error_message", + "format_funding_message_compact", + "format_needs_funding_message", + "format_row", + "format_solana_wallet_migration_notice", + "format_wallet_created_message", + "format_wallet_migration_notice", + "generate_solana_qr_ascii", + "generate_wallet", + "generate_wallet_qr_ascii", + "get_cost_log_summary", + "get_eip681_uri", + "get_or_create_solana_wallet", + # Wallet utilities + "get_or_create_wallet", + "get_payment_links", + "get_solana_public_key", + "get_solana_usdc_balance", + "get_wallet_address", + "import_solana_wallet", + "import_wallet", + "is_api_key", + "list_discovered_solana_wallets", + "list_discovered_wallets", + "list_image_models", + # Standalone functions (no wallet required) + "list_models", + "load_solana_wallet", + "load_wallet", + "open_solana_wallet_qr", + "open_wallet_qr", + "resolve_api_key", + "save_wallet_qr", + "scan_solana_wallets", + "scan_wallets", + # Solana wallet utilities + "setup_agent_solana_wallet", + # Entry point for agents (auto-creates wallet) + "setup_agent_wallet", + "status", + # Testnet convenience functions + "testnet_client", +] diff --git a/blockrun_llm/anthropic_client.py b/blockrun_llm/anthropic_client.py new file mode 100644 index 0000000..8242356 --- /dev/null +++ b/blockrun_llm/anthropic_client.py @@ -0,0 +1,233 @@ +""" +AnthropicClient — Use the official Anthropic SDK with BlockRun's API. + +Wraps anthropic.Anthropic with automatic x402 micropayments on Base chain. +Your private key is used ONLY for local EIP-712 signing and NEVER leaves your machine. + +Usage: + from blockrun_llm import AnthropicClient + + client = AnthropicClient() + response = client.messages.create( + model="claude-sonnet-4-6", + max_tokens=1024, + messages=[{"role": "user", "content": "Hello!"}] + ) + print(response.content[0].text) +""" + +from __future__ import annotations + +import os + +import httpx +from dotenv import load_dotenv +from eth_account import Account + +from .apikey import ( + PAYMENT_MODE_API_KEY, + PAYMENT_MODE_WALLET, + api_key_base_url, + resolve_api_key, +) +from .validation import validate_api_url, validate_private_key +from .wallet import load_wallet +from .x402 import create_payment_payload, extract_payment_details, parse_payment_required + +load_dotenv() + +# Default chat HTTP timeout (seconds). Was 120; reasoning models (opus-4.8) think +# 200–300s+, which the old default cut off mid-generation. Override via the +# BLOCKRUN_CHAT_TIMEOUT env var. Mirrors client.py / solana_client.py. +DEFAULT_CHAT_TIMEOUT = float(os.environ.get("BLOCKRUN_CHAT_TIMEOUT", "600")) + + +class _BlockRunX402Transport(httpx.BaseTransport): + """Custom httpx transport that intercepts 402 responses and signs x402 payments.""" + + def __init__( + self, account: Account, api_url: str, base_transport: httpx.BaseTransport | None = None + ): + self._account = account + self._api_url = api_url + self._base = base_transport or httpx.HTTPTransport() + + def handle_request(self, request: httpx.Request) -> httpx.Response: + response = self._base.handle_request(request) + + if response.status_code != 402: + return response + + # Read the 402 body so we can parse payment requirements + response.read() + + payment_header = response.headers.get("payment-required") + if not payment_header: + try: + resp_body = response.json() + if "x402" in resp_body: + payment_header = resp_body + except Exception: + pass + + if not payment_header: + return response + + if isinstance(payment_header, str): + payment_required = parse_payment_required(payment_header) + else: + payment_required = payment_header + + details = extract_payment_details(payment_required) + + resource = details.get("resource") or {} + extensions = payment_required.get("extensions", {}) + payment_payload = create_payment_payload( + account=self._account, + recipient=details["recipient"], + amount=details["amount"], + network=details.get("network", "eip155:8453"), + resource_url=resource.get("url", f"{self._api_url}/v1/messages"), + resource_description=resource.get("description", "BlockRun AI API call"), + max_timeout_seconds=details.get("maxTimeoutSeconds", 300), + extra=details.get("extra"), + extensions=extensions, + asset=details.get("asset"), + ) + + request.headers["PAYMENT-SIGNATURE"] = payment_payload + return self._base.handle_request(request) + + def close(self) -> None: + self._base.close() + + +class AnthropicClient: + """BlockRun-powered Anthropic client with automatic x402 payments. + + Drop-in replacement for anthropic.Anthropic that routes through BlockRun's + multi-model API gateway with automatic USDC micropayments on Base chain. + + Your private key is used ONLY for local EIP-712 signing and NEVER transmitted. + + Usage: + from blockrun_llm import AnthropicClient + + client = AnthropicClient() + response = client.messages.create( + model="claude-sonnet-4-6", + max_tokens=1024, + messages=[{"role": "user", "content": "Hello!"}] + ) + print(response.content[0].text) + + # Works with any BlockRun model in Anthropic format + response = client.messages.create( + model="openai/gpt-5.5", + max_tokens=1024, + messages=[{"role": "user", "content": "Hello from GPT!"}] + ) + """ + + DEFAULT_API_URL = "https://blockrun.ai/api" + + def __init__( + self, + private_key: str | None = None, + api_url: str | None = None, + timeout: float = DEFAULT_CHAT_TIMEOUT, + **kwargs, + ): + """ + Initialize the BlockRun Anthropic client. + + Args: + private_key: BlockRun API key or Base wallet private key. With no argument, + BLOCKRUN_API_KEY takes precedence over wallet configuration. + Wallet keys are used for local signing only. + api_url: BlockRun API endpoint (default: https://blockrun.ai/api). + timeout: Request timeout in seconds (default: 600, override via + BLOCKRUN_CHAT_TIMEOUT env). Reasoning models need 200–300s+. + **kwargs: Additional keyword arguments passed to anthropic.Anthropic. + + Raises: + ImportError: If the `anthropic` package is not installed. + ValueError: If no wallet is configured. + """ + try: + import anthropic + except ImportError: + raise ImportError( + "The 'anthropic' package is required for AnthropicClient.\n" + "Install it with: pip install blockrun-llm[anthropic]" + ) + + # Resolve the payment method before reading or parsing a wallet. + self.api_key = resolve_api_key(private_key) + self.payment_mode = PAYMENT_MODE_API_KEY if self.api_key else PAYMENT_MODE_WALLET + + # An ambiguous failed POST may already be billed, so retrying is an + # explicit caller choice rather than a default inherited from Anthropic. + # This has to hold on BOTH rails, and it matters more on the wallet one: + # the transport below signs a fresh payment for every 402 it sees, so a + # 5xx retried after the gateway already settled signs and settles again. + # At Anthropic's default of 2 that is three on-chain USDC transfers for + # one messages.create(). + kwargs.setdefault("max_retries", 0) + + if self.api_key: + self._api_url = api_key_base_url(api_url) + validate_api_url(self._api_url) + self._client = anthropic.Anthropic( + base_url=self._api_url, + api_key=self.api_key, + http_client=httpx.Client(timeout=timeout, follow_redirects=False), + **kwargs, + ) + return + + key = ( + private_key + or os.environ.get("BLOCKRUN_WALLET_KEY") + or os.environ.get("BASE_CHAIN_WALLET_KEY") + or load_wallet() + ) + if not key: + raise ValueError( + "No wallet configured. Either:\n" + " 1. Set BLOCKRUN_WALLET_KEY environment variable\n" + " 2. Pass private_key to AnthropicClient()\n" + " 3. For agent use: call setup_agent_wallet() first" + ) + + if not key.startswith("0x"): + key = "0x" + key + + validate_private_key(key) + account = Account.from_key(key) + + api_url_resolved = api_url or os.environ.get("BLOCKRUN_API_URL") or self.DEFAULT_API_URL + validate_api_url(api_url_resolved) + self._api_url = api_url_resolved.rstrip("/") + + transport = _BlockRunX402Transport( + account=account, + api_url=self._api_url, + ) + + http_client = httpx.Client(transport=transport, timeout=timeout) + + self._client = anthropic.Anthropic( + base_url=self._api_url, + api_key="blockrun", + http_client=http_client, + **kwargs, + ) + + @property + def messages(self): + """Access the Messages API (client.messages.create(...)).""" + return self._client.messages + + def __getattr__(self, name): + return getattr(self._client, name) diff --git a/blockrun_llm/apikey.py b/blockrun_llm/apikey.py new file mode 100644 index 0000000..2809105 --- /dev/null +++ b/blockrun_llm/apikey.py @@ -0,0 +1,267 @@ +"""The API-key rail. + +BlockRun sells the same catalogue through two front doors. The x402 rail +(``blockrun.ai`` / ``sol.blockrun.ai``) authenticates a caller by wallet +signature and settles USDC on-chain per request. The account rail +(``api.blockrun.ai``) authenticates a caller by API key and draws down prepaid +credit held against the account at `user.blockrun.ai `_. + +The two are the same backend and the same response shapes, which is what makes +one client able to serve both: the only differences are the host, the header +that authenticates, and the fact that a 402 on the account rail means "out of +credit" rather than "sign this". + +A key is never a wallet. In API-key mode ``self.account`` is ``None``, there is +no address, and nothing is signed locally — so the wallet-only helpers report +that plainly instead of returning a zero that looks like an answer. + +Every client class in this package wires itself up through +:func:`configure_credential`, so the rail is one decision made in one place +rather than fourteen copies that can drift. +""" + +from __future__ import annotations + +import os +from typing import Any +from urllib.parse import urlsplit + +from .types import APIError, PaymentError + +#: Prefix every BlockRun API key carries. It is what lets one credential +#: parameter accept either kind: a hex private key can never start with ``brk_``. +API_KEY_PREFIX = "brk_" + +#: The account-rail gateway. Unlike the x402 default it carries no ``/api`` +#: suffix: api.blockrun.ai serves ``/v1/...`` at the root and answers +#: ``/api/v1/...`` with a ``wrong_host`` error. +DEFAULT_API_KEY_URL = "https://api.blockrun.ai" + +#: Holds a BlockRun API key. Setting it puts every client in this process on the +#: account rail, the Solana clients included — the key is the payment method, so +#: the chain stops being a question. +ENV_API_KEY = "BLOCKRUN_API_KEY" + +#: Overrides the account-rail host. Deliberately not ``BLOCKRUN_API_URL``: that +#: one names the x402 gateway, and a developer who has it pointed at a private +#: x402 deployment must not have an API-key client silently follow it there and +#: hand over the key. +ENV_API_KEY_URL = "BLOCKRUN_API_KEY_URL" + +#: Which rail a client pays on, as reported by ``client.payment_mode``. +PAYMENT_MODE_WALLET = "wallet" +PAYMENT_MODE_API_KEY = "apikey" + + +def is_api_key(credential: str | None) -> bool: + """Is this credential a BlockRun API key rather than a wallet private key?""" + return bool(credential) and str(credential).strip().startswith(API_KEY_PREFIX) + + +def resolve_api_key(credential: str | None) -> str | None: + """Decide whether a constructor call is an API-key call. + + Precedence, and the reason for it: an explicit argument beats everything, + because the caller wrote it at the call site. Then ``BLOCKRUN_API_KEY`` + beats the wallet variables, because it is the new variable — a developer + who has not set it keeps the wallet behaviour they already had, and one who + has set it meant to, even if an old ``BLOCKRUN_WALLET_KEY`` is still sitting + in their profile. ``client.payment_mode`` exists so that decision is never + invisible. + """ + if is_api_key(credential): + return str(credential).strip() + # An explicit non-key credential is a deliberate choice of the x402 rail and + # must not be overridden by the environment. + if credential and str(credential).strip(): + return None + # Blank is unset, not invalid. `BLOCKRUN_API_KEY=` in a .env file, a bare + # `docker -e BLOCKRUN_API_KEY`, and an unpopulated `${{ secrets.X }}` all + # land here as the empty string, and every one of them means "I am not on + # the account rail" — raising would break wallet users who never opted in. + # A non-blank value that is not a key is a different thing: someone typed a + # credential and got it wrong, and silently spending USDC instead of credit + # is the wrong way to tell them. + env = os.environ.get(ENV_API_KEY, "").strip() + if not env: + return None + if not is_api_key(env): + raise ValueError( + f"Invalid BLOCKRUN_API_KEY: expected a key starting with {API_KEY_PREFIX!r}. " + "Correct it, clear it, or explicitly pass a wallet key." + ) + return env + + +def api_key_base_url(api_url: str | None = None) -> str: + """Resolve the account-rail host: explicit argument, env override, default.""" + if api_url and api_url.strip(): + return api_url.strip().rstrip("/") + env = os.environ.get(ENV_API_KEY_URL, "").strip() + if env: + return env.rstrip("/") + return DEFAULT_API_KEY_URL + + +def auth_headers(api_key: str | None) -> dict[str, str]: + """The header that authenticates on the account rail, empty on the x402 one. + + ``Authorization: Bearer`` is the OpenAI-SDK shape; the gateway also accepts + ``x-api-key`` for Anthropic-shaped clients. One is sent, not both, so a + proxy that logs headers records the key once. + """ + return {"Authorization": f"Bearer {api_key}"} if api_key else {} + + +def configure_credential( + obj: Any, + private_key: str | None, + api_url: str | None, +) -> bool: + """Put ``obj`` on the account rail if the credential says so. + + Sets ``obj.api_key``, ``obj.api_url`` and ``obj.account`` and returns True + when an API key was found, so a caller's ``__init__`` can skip every + wallet-loading step. Returns False and touches nothing otherwise, leaving + the existing wallet path exactly as it was. + """ + api_key = resolve_api_key(private_key) + if not api_key: + obj.api_key = None + return False + obj.api_key = api_key + obj.account = None + obj.api_url = api_key_base_url(api_url) + return True + + +def payment_mode(obj: Any) -> str: + """Which rail ``obj`` pays on. Worth checking once at startup when both a + key and a wallet are configured: it is the difference between spending + credit and spending USDC.""" + return PAYMENT_MODE_API_KEY if getattr(obj, "api_key", None) else PAYMENT_MODE_WALLET + + +def resolve_poll_url(poll_url: str, api_url: str, api_key: str | None) -> str: + """Resolve a server-supplied relative ``poll_url`` against the API host. + + ``poll_url`` is minted by the x402 gateway and is relative to *its* host, so + it arrives as ``/api/v1/...``. api.blockrun.ai serves the same route at + ``/v1/...`` and answers ``/api/v1/...`` with ``wrong_host``, so on the + account rail the prefix has to come off here — the alternative is every + async job (video, slow images) polling a 404 until its budget runs out. + """ + if api_key: + target = urlsplit(poll_url) + gateway = urlsplit(api_url) + if target.scheme or target.netloc: + if ( + target.scheme.lower() != gateway.scheme.lower() + or target.hostname != gateway.hostname + or (target.port or (443 if target.scheme == "https" else 80)) + != (gateway.port or (443 if gateway.scheme == "https" else 80)) + or target.username is not None + or target.password is not None + ): + raise ValueError("Refusing to send an API key to a different polling origin.") + return poll_url + path = poll_url.removeprefix("/api") if poll_url.startswith("/api/") else poll_url + return f"{api_url.rstrip('/')}/{path.lstrip('/')}" + if poll_url.startswith(("http://", "https://")): + return poll_url + return f"{api_url.removesuffix('/api')}{poll_url}" + + +def api_key_payment_error(body: Any = None) -> PaymentError: + """Explain a 402 that arrived on the account rail. + + On the x402 rail a 402 is the normal opening move of a conversation. On this + one it is a refusal: the account is out of credit, suspended, or past its + limit. Signing is not the answer and there is nothing to sign with, so the + error says what to do instead rather than letting the caller fall into the + wallet path and get a wallet error for a problem that has nothing to do with + wallets. + """ + detail = "" + if body is not None: + detail = str(body).strip() + if len(detail) > 400: + detail = detail[:400] + "…" + message = ( + "402 from api.blockrun.ai: this account has no credit left for that call. " + "Top up at https://user.blockrun.ai/dashboard/credits, or call one of the " + "free models, which need no credit." + ) + if detail: + message = f"{message} Gateway said: {detail}" + return PaymentError(message) + + +def wallet_only(helper: str) -> ValueError: + """The error every wallet-only helper raises on the account rail. + + Naming the helper matters: "no wallet" alone leaves the caller guessing + which of ``get_balance`` / ``onramp`` / ``get_wallet_address`` they should + not have called. + """ + return ValueError( + f"{helper}() is wallet-only and this client authenticates with a BlockRun " + f"API key. Credit balance, usage and top-ups live at " + f"https://user.blockrun.ai/dashboard. Construct the client with a wallet " + f"private key (or unset {ENV_API_KEY}) to use {helper}()." + ) + + +def missing_credential_error(*, extra: str = "") -> ValueError: + """The 'nothing configured' error, now that a key is one of the options. + + Every client raised its own wording listing only wallet routes, which + stopped being the whole truth the moment a key became a credential. One + message, listing both. + """ + lines = [ + "No credential configured. Either:", + f" 1. Set {ENV_API_KEY} to an API key from https://user.blockrun.ai", + " 2. Set BLOCKRUN_WALLET_KEY to a wallet private key", + " 3. Pass either one as the first constructor argument", + ] + if extra: + lines.append(f" 4. {extra}") + lines.append("NOTE: a wallet key never leaves your machine — only signatures are sent.") + return ValueError("\n".join(lines)) + + +def raise_for_api_key_402(response: Any, api_key: str | None) -> None: + """Turn a 402 on the account rail into a credit refusal, before any signing. + + A no-op on the x402 rail, so every request site can call it unconditionally. + """ + if not api_key or response.status_code != 402: + return + try: + body = response.json() + except Exception: + body = response.text + raise api_key_payment_error(body) + + +__all__ = [ + "API_KEY_PREFIX", + "DEFAULT_API_KEY_URL", + "ENV_API_KEY", + "ENV_API_KEY_URL", + "PAYMENT_MODE_API_KEY", + "PAYMENT_MODE_WALLET", + "APIError", + "api_key_base_url", + "api_key_payment_error", + "auth_headers", + "configure_credential", + "is_api_key", + "missing_credential_error", + "payment_mode", + "raise_for_api_key_402", + "resolve_api_key", + "resolve_poll_url", + "wallet_only", +] diff --git a/blockrun_llm/billing.py b/blockrun_llm/billing.py new file mode 100644 index 0000000..d5dfb56 --- /dev/null +++ b/blockrun_llm/billing.py @@ -0,0 +1,187 @@ +"""Local billing / cost-tracking CLI for the BlockRun Python SDK. + +Reads ``~/.blockrun/cost_log.jsonl`` and prints / exports filtered, grouped +summaries. No HTTP requests, no payment, no wallet signing — operates purely +on the local append-only ledger written by ``save_to_cache`` after every paid +API call. + +Usage: + python -m blockrun_llm.billing summary [--from] [--to] + [--wallet] [--network] + [--group-by FIELD] + python -m blockrun_llm.billing export {csv|json} + [--from] [--to] + [--wallet] [--network] + [--output PATH] + +Examples: + python -m blockrun_llm.billing summary + python -m blockrun_llm.billing summary --group-by model + python -m blockrun_llm.billing summary --from 2026-04-01 --group-by month + python -m blockrun_llm.billing export csv --from 2026-05-01 --output may.csv + python -m blockrun_llm.billing export json --wallet 0x... +""" + +from __future__ import annotations + +import argparse +import sys + +from .cache import ( + COST_LOG_PATH, + export_cost_log_csv, + export_cost_log_json, + get_cost_log_summary, +) + + +def _fmt_usd(value: float) -> str: + return f"${value:.4f}" + + +def _print_summary(args: argparse.Namespace) -> int: + summary = get_cost_log_summary( + from_date=args.from_date, + to_date=args.to_date, + wallet=args.wallet, + network=args.network, + group_by=args.group_by, + ) + + print("=" * 64) + print("BLOCKRUN — LOCAL COST LOG SUMMARY") + print("=" * 64) + print(f" log file : {COST_LOG_PATH}") + if args.from_date: + print(f" from : {args.from_date}") + if args.to_date: + print(f" to : {args.to_date}") + if args.wallet: + print(f" wallet : {args.wallet}") + if args.network: + print(f" network : {args.network}") + print(f" group_by : {args.group_by}") + print(f" total : {_fmt_usd(summary['total_usd'])} ({summary['calls']} calls)") + print() + + groups = summary.get("groups") or { + # legacy no-arg shape — adapt + k: {"calls": 0, "cost_usd": v} + for k, v in (summary.get("by_endpoint") or {}).items() + } + if not groups: + print(" (no entries match the filter)") + return 0 + + rows = sorted(groups.items(), key=lambda kv: kv[1]["cost_usd"], reverse=True) + width = max((len(str(k)) for k, _ in rows), default=10) + width = min(max(width, 12), 60) + print(f" {'KEY':<{width}} {'CALLS':>7} {'COST':>10}") + print(f" {'-' * width} {'-' * 7} {'-' * 10}") + for key, slot in rows: + calls = slot.get("calls") or 0 + cost = slot.get("cost_usd") or 0.0 + display = (str(key) or "(none)")[:width] + print(f" {display:<{width}} {calls:>7} {_fmt_usd(cost):>10}") + print() + return 0 + + +def _print_export(args: argparse.Namespace) -> int: + fmt = args.format + if fmt == "csv": + text = export_cost_log_csv( + args.output, + from_date=args.from_date, + to_date=args.to_date, + wallet=args.wallet, + network=args.network, + ) + elif fmt == "json": + text = export_cost_log_json( + args.output, + from_date=args.from_date, + to_date=args.to_date, + wallet=args.wallet, + network=args.network, + ) + else: + sys.stderr.write(f"unknown format: {fmt}\n") + return 2 + + if args.output: + print(f"wrote {args.output}", file=sys.stderr) + else: + sys.stdout.write(text) + if not text.endswith("\n"): + sys.stdout.write("\n") + return 0 + + +def _add_filter_args(p: argparse.ArgumentParser) -> None: + p.add_argument( + "--from", + dest="from_date", + type=str, + default=None, + help="lower bound (inclusive) — YYYY-MM-DD or ISO datetime", + ) + p.add_argument( + "--to", + dest="to_date", + type=str, + default=None, + help="upper bound (inclusive) — YYYY-MM-DD or ISO datetime", + ) + p.add_argument("--wallet", type=str, default=None, help="filter to one wallet") + p.add_argument( + "--network", + type=str, + default=None, + help="filter to one network (base-mainnet / base-sepolia / solana-mainnet)", + ) + + +def build_parser() -> argparse.ArgumentParser: + parser = argparse.ArgumentParser( + prog="python -m blockrun_llm.billing", + description="Local BlockRun cost-log reader / exporter.", + ) + sub = parser.add_subparsers(dest="command", required=True) + + s_summary = sub.add_parser("summary", help="Print an aggregated summary.") + _add_filter_args(s_summary) + s_summary.add_argument( + "--group-by", + choices=("endpoint", "model", "wallet", "network", "client_kind", "day", "month"), + default="endpoint", + help="bucket dimension (default: endpoint)", + ) + s_summary.set_defaults(handler=_print_summary) + + s_export = sub.add_parser("export", help="Export raw entries.") + s_export.add_argument("format", choices=("csv", "json"), help="output format") + _add_filter_args(s_export) + s_export.add_argument( + "--output", + type=str, + default=None, + help="write to this path (default: stdout)", + ) + s_export.set_defaults(handler=_print_export) + + return parser + + +def main(argv: list[str] | None = None) -> int: + parser = build_parser() + args = parser.parse_args(argv) + try: + return int(args.handler(args) or 0) + except ValueError as exc: + sys.stderr.write(f"error: {exc}\n") + return 2 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/blockrun_llm/cache.py b/blockrun_llm/cache.py new file mode 100644 index 0000000..2f9f079 --- /dev/null +++ b/blockrun_llm/cache.py @@ -0,0 +1,510 @@ +""" +Local response cache, archive, and cost log for paid BlockRun API calls. + +Three storage layers: +1. **Cache** (~/.blockrun/cache/) — hash-keyed, TTL-based dedup to avoid paying twice +2. **Data** (~/.blockrun/data/) — human-readable JSON files for every paid call +3. **Cost log** (~/.blockrun/cost_log.jsonl) — append-only ledger for billing / audit + +Cost log entries (one per line) include endpoint, cost, plus model / wallet / +network / client_kind metadata when the caller provides it. Older entries with +only `{ts, endpoint, cost_usd}` are still readable — missing fields surface as +``None`` in the summary / export views. +""" + +from __future__ import annotations + +import csv +import hashlib +import io +import json +import re +import time +from collections.abc import Iterator +from datetime import datetime, timezone +from pathlib import Path +from typing import Any + +# Default TTL in seconds per endpoint pattern +DEFAULT_TTL: dict[str, int] = { + # X/Twitter data — cache 1 hour (followers/tweets don't change every minute) + "/v1/partner/": 3600, + # Prediction markets — cache 30 minutes + "/v1/pm/": 1800, + # Chat completions — no cache (each call is unique) + "/v1/chat/": 0, + # Search — cache 15 minutes + "/v1/search": 900, + # Image — no cache + "/v1/image": 0, + # Video generation — no cache (each clip is unique / expensive) + "/v1/videos": 0, + # Audio (music / speech / sound-effects) — no cache + "/v1/audio": 0, + # Media enrollment (portrait / realface) — no cache + "/v1/portrait/": 0, + "/v1/realface/": 0, + # Live JSON-RPC — no cache (chain state is realtime) + "/v1/rpc/": 0, + # Pyth market data — no cache (realtime quotes) + "/v1/crypto": 0, + "/v1/fx": 0, + "/v1/commodity": 0, + "/v1/stocks": 0, + "/v1/usstock": 0, +} + +CACHE_DIR = Path.home() / ".blockrun" / "cache" +DATA_DIR = Path.home() / ".blockrun" / "data" +COST_LOG_PATH = Path.home() / ".blockrun" / "cost_log.jsonl" + + +def _get_ttl(endpoint: str) -> int: + """Get TTL for an endpoint based on pattern matching.""" + for pattern, ttl in DEFAULT_TTL.items(): + if pattern in endpoint: + return ttl + # Default: cache 1 hour for unknown endpoints + return 3600 + + +def _cache_key(endpoint: str, body: dict[str, Any]) -> str: + """Generate a deterministic cache key from endpoint + request body.""" + key_data = json.dumps({"endpoint": endpoint, "body": body}, sort_keys=True) + return hashlib.sha256(key_data.encode()).hexdigest()[:16] + + +def _cache_path(key: str) -> Path: + """Get the file path for a cache entry.""" + return CACHE_DIR / f"{key}.json" + + +def get_cached(endpoint: str, body: dict[str, Any]) -> dict[str, Any] | None: + """ + Check if a cached response exists and is still fresh. + + Returns the cached response dict if hit, None if miss or expired. + """ + ttl = _get_ttl(endpoint) + if ttl <= 0: + return None + + key = _cache_key(endpoint, body) + path = _cache_path(key) + + if not path.exists(): + return None + + try: + entry = json.loads(path.read_text()) + cached_at = entry.get("cached_at", 0) + + if time.time() - cached_at > ttl: + # Expired + path.unlink(missing_ok=True) + return None + + return entry.get("response") + except (json.JSONDecodeError, OSError): + return None + + +def _readable_filename(endpoint: str, body: dict[str, Any]) -> str: + """ + Generate a human-readable filename from endpoint + request body. + """ + ts = datetime.now().strftime("%Y-%m-%d_%H%M%S") + + ep = endpoint.rstrip("/").rsplit("/", 1)[-1] + if "/v1/chat/" in endpoint: + ep = "chat" + elif "/v1/search" in endpoint: + ep = "search" + elif "/v1/image" in endpoint: + ep = "image" + + if not isinstance(body, dict): + # JSON-RPC batch requests send a list body — no labelable fields. + body = {} + label = ( + body.get("query") + or body.get("username") + or body.get("handle") + or body.get("model") + or body.get("prompt", "")[:40] + or "" + ) + label = re.sub(r"[^a-zA-Z0-9_\-]", "_", str(label))[:40].strip("_") + + return f"{ep}_{ts}_{label}.json" if label else f"{ep}_{ts}.json" + + +def save_to_cache( + endpoint: str, + body: dict[str, Any], + response: dict[str, Any], + cost_usd: float = 0.0, + *, + model: str | None = None, + wallet: str | None = None, + network: str | None = None, + client_kind: str | None = None, +) -> None: + """ + Save a paid API response locally. + + 1. Hash-keyed cache file (for TTL-based dedup) + 2. Human-readable data file (browsable archive of every paid call) + 3. Cost log entry (with billing metadata when supplied) + """ + CACHE_DIR.mkdir(parents=True, exist_ok=True) + + key = _cache_key(endpoint, body) + entry = { + "cached_at": time.time(), + "endpoint": endpoint, + "body": body, + "response": response, + "cost_usd": cost_usd, + } + + try: + _cache_path(key).write_text(json.dumps(entry, default=str)) + except OSError: + pass + + # Save human-readable copy to ~/.blockrun/data/ + _save_readable(endpoint, body, response, cost_usd) + + # Append to the cost log (never overwritten). Pull model from the body + # if the caller didn't pass one explicitly. + _append_cost_log( + endpoint, + cost_usd, + model=model or (body.get("model") if isinstance(body, dict) else None), + wallet=wallet, + network=network, + client_kind=client_kind, + ) + + +def _save_readable( + endpoint: str, + body: dict[str, Any], + response: dict[str, Any], + cost_usd: float, +) -> None: + """Save a human-readable JSON file to ~/.blockrun/data/.""" + DATA_DIR.mkdir(parents=True, exist_ok=True) + filename = _readable_filename(endpoint, body) + entry = { + "saved_at": datetime.now().isoformat(), + "endpoint": endpoint, + "cost_usd": cost_usd, + "request": body, + "response": response, + } + try: + (DATA_DIR / filename).write_text(json.dumps(entry, indent=2, default=str)) + except OSError: + pass + + +def _append_cost_log( + endpoint: str, + cost_usd: float, + *, + model: str | None = None, + wallet: str | None = None, + network: str | None = None, + client_kind: str | None = None, +) -> None: + """Append one JSONL row to ``~/.blockrun/cost_log.jsonl``. + + The full schema is:: + + {ts, endpoint, cost_usd, model, wallet, network, client_kind} + + Older rows that only carry ``{ts, endpoint, cost_usd}`` are still readable + — missing fields surface as ``None`` in summary / export views. + """ + if cost_usd <= 0: + return + + try: + COST_LOG_PATH.parent.mkdir(parents=True, exist_ok=True) + with open(COST_LOG_PATH, "a") as f: + entry: dict[str, Any] = { + "ts": time.time(), + "endpoint": endpoint, + "cost_usd": cost_usd, + } + if model is not None: + entry["model"] = model + if wallet is not None: + entry["wallet"] = wallet + if network is not None: + entry["network"] = network + if client_kind is not None: + entry["client_kind"] = client_kind + f.write(json.dumps(entry) + "\n") + except OSError: + pass + + +def clear_cache() -> int: + """Clear all cached responses. Returns number of files removed.""" + if not CACHE_DIR.exists(): + return 0 + count = 0 + for f in CACHE_DIR.glob("*.json"): + f.unlink(missing_ok=True) + count += 1 + return count + + +# --------------------------------------------------------------------------- +# Cost-log readers +# --------------------------------------------------------------------------- + + +def _parse_date(value: str | None) -> float | None: + """Parse a YYYY-MM-DD or ISO 8601 date into a unix timestamp. + + Bare dates (YYYY-MM-DD) anchor to UTC midnight. Returns ``None`` when + ``value`` is ``None``; raises ``ValueError`` on malformed input. + """ + if value is None: + return None + s = value.strip() + if not s: + return None + if len(s) == 10 and s[4] == "-" and s[7] == "-": + s = s + "T00:00:00+00:00" + try: + dt = datetime.fromisoformat(s.replace("Z", "+00:00")) + except ValueError as exc: + raise ValueError(f"could not parse date {value!r}: {exc}") from exc + if dt.tzinfo is None: + dt = dt.replace(tzinfo=timezone.utc) + return dt.timestamp() + + +def _iter_cost_log( + *, + from_ts: float | None = None, + to_ts: float | None = None, + wallet: str | None = None, + network: str | None = None, +) -> Iterator[dict[str, Any]]: + """Yield cost-log entries that match the optional filters.""" + if not COST_LOG_PATH.exists(): + return + try: + text = COST_LOG_PATH.read_text() + except OSError: + return + for line in text.splitlines(): + if not line.strip(): + continue + try: + entry = json.loads(line) + except json.JSONDecodeError: + continue + ts = entry.get("ts") + if not isinstance(ts, (int, float)): + continue + if from_ts is not None and ts < from_ts: + continue + if to_ts is not None and ts > to_ts: + continue + if wallet is not None and entry.get("wallet") != wallet: + continue + if network is not None and entry.get("network") != network: + continue + yield entry + + +def _group_key(entry: dict[str, Any], group_by: str) -> str: + """Compute the bucket key for a given grouping field.""" + if group_by == "day": + ts = entry.get("ts") + if not isinstance(ts, (int, float)): + return "unknown" + return datetime.fromtimestamp(ts, tz=timezone.utc).strftime("%Y-%m-%d") + if group_by == "month": + ts = entry.get("ts") + if not isinstance(ts, (int, float)): + return "unknown" + return datetime.fromtimestamp(ts, tz=timezone.utc).strftime("%Y-%m") + return str(entry.get(group_by) or "unknown") + + +_VALID_GROUP_BY = {"endpoint", "model", "wallet", "network", "client_kind", "day", "month"} + + +def get_cost_log_summary( + *, + from_date: str | None = None, + to_date: str | None = None, + wallet: str | None = None, + network: str | None = None, + group_by: str = "endpoint", +) -> dict[str, Any]: + """Read the cost log and return an aggregated summary. + + Args: + from_date: ISO date / datetime — entries strictly older are skipped. + to_date: ISO date / datetime — entries strictly newer are skipped. + wallet: Filter to a single wallet address. + network: Filter to a single network (``base-mainnet`` etc.). + group_by: One of ``endpoint`` (default), ``model``, ``wallet``, + ``network``, ``client_kind``, ``day``, ``month``. + + Returns: + ``{"from_date", "to_date", "total_usd", "calls", "group_by", + "groups"}`` where ``groups`` maps each bucket key to + ``{"calls": int, "cost_usd": float}``. When ``group_by == "endpoint"`` + the response also includes a ``by_endpoint`` alias mapping endpoint + path to total cost (a float) for backwards compatibility with the + original 3-key shape. + """ + if group_by not in _VALID_GROUP_BY: + raise ValueError(f"group_by must be one of {sorted(_VALID_GROUP_BY)}; got {group_by!r}") + + from_ts = _parse_date(from_date) + to_ts = _parse_date(to_date) + + total = 0.0 + calls = 0 + groups: dict[str, dict[str, Any]] = {} + + for entry in _iter_cost_log(from_ts=from_ts, to_ts=to_ts, wallet=wallet, network=network): + cost = float(entry.get("cost_usd") or 0.0) + bucket = _group_key(entry, group_by) + total += cost + calls += 1 + slot = groups.setdefault(bucket, {"calls": 0, "cost_usd": 0.0}) + slot["calls"] += 1 + slot["cost_usd"] += cost + + result: dict[str, Any] = { + "from_date": from_date, + "to_date": to_date, + "total_usd": total, + "calls": calls, + "group_by": group_by, + "groups": groups, + } + # Backwards-compat: callers that depended on the historical + # ``by_endpoint`` mapping (cost only, no calls) keep it when the user + # is grouping by endpoint. + if group_by == "endpoint": + result["by_endpoint"] = {k: v["cost_usd"] for k, v in groups.items()} + return result + + +# --------------------------------------------------------------------------- +# Cost-log exporters +# --------------------------------------------------------------------------- + + +_EXPORT_COLUMNS = ( + "ts_iso", + "endpoint", + "model", + "wallet", + "network", + "client_kind", + "cost_usd", +) + + +def _entry_to_record(entry: dict[str, Any]) -> dict[str, Any]: + """Normalize one raw cost-log entry into the export record shape.""" + ts = entry.get("ts") + ts_iso = ( + datetime.fromtimestamp(ts, tz=timezone.utc).isoformat() + if isinstance(ts, (int, float)) + else None + ) + return { + "ts_iso": ts_iso, + "endpoint": entry.get("endpoint"), + "model": entry.get("model"), + "wallet": entry.get("wallet"), + "network": entry.get("network"), + "client_kind": entry.get("client_kind"), + "cost_usd": float(entry.get("cost_usd") or 0.0), + } + + +def _filtered_records( + *, + from_date: str | None, + to_date: str | None, + wallet: str | None, + network: str | None, +) -> list[dict[str, Any]]: + from_ts = _parse_date(from_date) + to_ts = _parse_date(to_date) + return [ + _entry_to_record(e) + for e in _iter_cost_log(from_ts=from_ts, to_ts=to_ts, wallet=wallet, network=network) + ] + + +def export_cost_log_csv( + output_path: str | Path | None = None, + *, + from_date: str | None = None, + to_date: str | None = None, + wallet: str | None = None, + network: str | None = None, +) -> str: + """Render filtered cost-log entries as CSV. + + Columns: ``ts_iso, endpoint, model, wallet, network, client_kind, cost_usd``. + + When ``output_path`` is supplied the CSV is also written to that file (the + parent directory is created if missing). The CSV text is always returned + so callers can pipe it into other tools. + """ + records = _filtered_records( + from_date=from_date, to_date=to_date, wallet=wallet, network=network + ) + buffer = io.StringIO() + writer = csv.DictWriter(buffer, fieldnames=list(_EXPORT_COLUMNS)) + writer.writeheader() + for record in records: + writer.writerow({k: ("" if record.get(k) is None else record[k]) for k in _EXPORT_COLUMNS}) + text = buffer.getvalue() + if output_path is not None: + path = Path(output_path) + path.parent.mkdir(parents=True, exist_ok=True) + path.write_text(text) + return text + + +def export_cost_log_json( + output_path: str | Path | None = None, + *, + from_date: str | None = None, + to_date: str | None = None, + wallet: str | None = None, + network: str | None = None, +) -> str: + """Render filtered cost-log entries as a JSON array of records. + + Same fields as ``export_cost_log_csv``. Pretty-printed with 2-space + indentation. Returns the JSON text; writes to ``output_path`` when given. + """ + records = _filtered_records( + from_date=from_date, to_date=to_date, wallet=wallet, network=network + ) + text = json.dumps(records, indent=2, default=str) + if output_path is not None: + path = Path(output_path) + path.parent.mkdir(parents=True, exist_ok=True) + path.write_text(text) + return text diff --git a/blockrun_llm/client.py b/blockrun_llm/client.py index 5c878cf..e48a8e0 100644 --- a/blockrun_llm/client.py +++ b/blockrun_llm/client.py @@ -1,552 +1,4287 @@ -""" -BlockRun LLM Client - Main SDK entry point. - -Usage: - from blockrun_llm import LLMClient - - # Initialize with private key from env (BLOCKRUN_WALLET_KEY) - client = LLMClient() - - # Or pass private key directly - client = LLMClient(private_key="0x...") - - # Simple 1-line chat - response = client.chat("gpt-4o", "What is 2+2?") - print(response) - - # Full chat with messages - messages = [ - {"role": "system", "content": "You are a helpful assistant."}, - {"role": "user", "content": "Hello!"} - ] - result = client.chat_completion("gpt-4o", messages) - print(result.choices[0].message.content) -""" - -import os -from typing import List, Dict, Any, Optional, Union -import httpx -from eth_account import Account -from dotenv import load_dotenv - -from .types import ChatMessage, ChatResponse, APIError, PaymentError -from .x402 import create_payment_payload, parse_payment_required, extract_payment_details -from .validation import ( - validate_private_key, - validate_api_url, - validate_model, - validate_max_tokens, - validate_temperature, - validate_top_p, - sanitize_error_response, - validate_resource_url, -) - - -# Load environment variables -load_dotenv() - - -class LLMClient: - """ - BlockRun LLM Gateway Client. - - Provides access to multiple LLM providers (OpenAI, Anthropic, Google, etc.) - with automatic x402 micropayments on Base chain. - """ - - DEFAULT_API_URL = "https://blockrun.ai/api" - DEFAULT_MAX_TOKENS = 1024 - - def __init__( - self, - private_key: Optional[str] = None, - api_url: Optional[str] = None, - timeout: float = 60.0, - ): - """ - Initialize the BlockRun LLM client. - - Args: - private_key: EVM wallet private key (or set BLOCKRUN_WALLET_KEY env var) - api_url: API endpoint URL (default: https://blockrun.ai/api) - timeout: Request timeout in seconds (default: 60) - - Raises: - ValueError: If no private key is provided or found in env - """ - # Get private key from param or environment - key = private_key or os.environ.get("BLOCKRUN_WALLET_KEY") - if not key: - raise ValueError( - "Private key required. Either pass private_key parameter or set " - "BLOCKRUN_WALLET_KEY environment variable." - ) - - # Validate private key format - validate_private_key(key) - - # Initialize wallet account (key stays local, never transmitted) - self.account = Account.from_key(key) - - # Validate and set API URL - api_url_raw = api_url or os.environ.get("BLOCKRUN_API_URL") or self.DEFAULT_API_URL - validate_api_url(api_url_raw) - self.api_url = api_url_raw.rstrip("/") - - self.timeout = timeout - - # HTTP client - self._client = httpx.Client(timeout=timeout) - - def chat( - self, - model: str, - prompt: str, - *, - system: Optional[str] = None, - max_tokens: Optional[int] = None, - temperature: Optional[float] = None, - ) -> str: - """ - Simple 1-line chat interface. - - Args: - model: Model ID (e.g., "gpt-4o", "claude-3-5-sonnet", "gemini-2.5-pro") - prompt: User message - system: Optional system prompt - max_tokens: Max tokens to generate (default: 1024) - temperature: Sampling temperature - - Returns: - Assistant's response text - - Example: - response = client.chat("gpt-4o", "What is the capital of France?") - print(response) # "The capital of France is Paris." - """ - messages: List[Dict[str, str]] = [] - - if system: - messages.append({"role": "system", "content": system}) - - messages.append({"role": "user", "content": prompt}) - - result = self.chat_completion( - model=model, - messages=messages, - max_tokens=max_tokens, - temperature=temperature, - ) - - return result.choices[0].message.content - - def chat_completion( - self, - model: str, - messages: List[Dict[str, str]], - *, - max_tokens: Optional[int] = None, - temperature: Optional[float] = None, - top_p: Optional[float] = None, - ) -> ChatResponse: - """ - Full chat completion interface (OpenAI-compatible). - - Args: - model: Model ID - messages: List of message dicts with 'role' and 'content' - max_tokens: Max tokens to generate - temperature: Sampling temperature - top_p: Nucleus sampling parameter - - Returns: - ChatResponse object with choices and usage - - Example: - messages = [ - {"role": "system", "content": "You are helpful."}, - {"role": "user", "content": "Hello!"} - ] - result = client.chat_completion("gpt-4o", messages) - """ - # Validate inputs - validate_model(model) - validate_max_tokens(max_tokens) - validate_temperature(temperature) - validate_top_p(top_p) - - # Build request body - body: Dict[str, Any] = { - "model": model, - "messages": messages, - "max_tokens": max_tokens or self.DEFAULT_MAX_TOKENS, - } - - if temperature is not None: - body["temperature"] = temperature - if top_p is not None: - body["top_p"] = top_p - - # Make request (with automatic payment handling) - return self._request_with_payment("/v1/chat/completions", body) - - def _request_with_payment(self, endpoint: str, body: Dict[str, Any]) -> ChatResponse: - """ - Make a request with automatic x402 payment handling. - - 1. Send initial request - 2. If 402, parse payment requirements - 3. Sign payment locally - 4. Retry with X-Payment header - """ - url = f"{self.api_url}{endpoint}" - - # First attempt (will likely return 402) - response = self._client.post( - url, - json=body, - headers={"Content-Type": "application/json"}, - ) - - # Handle 402 Payment Required - if response.status_code == 402: - return self._handle_payment_and_retry(url, body, response) - - # Handle other errors - if response.status_code != 200: - try: - error_body = response.json() - except Exception: - error_body = {"error": "Request failed"} - raise APIError( - f"API error: {response.status_code}", - response.status_code, - sanitize_error_response(error_body), - ) - - # Parse successful response - return ChatResponse(**response.json()) - - def _handle_payment_and_retry( - self, - url: str, - body: Dict[str, Any], - response: httpx.Response, - ) -> ChatResponse: - """Handle 402 response: parse requirements, sign payment, retry.""" - # Get payment required header - payment_header = response.headers.get("X-Payment-Required") - if not payment_header: - # Try to get from response body - try: - resp_body = response.json() - if "x402" in resp_body: - payment_header = resp_body - except Exception: - pass - - if not payment_header: - raise PaymentError("402 response but no payment requirements found") - - # Parse payment requirements - if isinstance(payment_header, str): - payment_required = parse_payment_required(payment_header) - else: - payment_required = payment_header - - # Extract payment details - details = extract_payment_details(payment_required) - - # Create signed payment payload (v2 format) - resource = details.get("resource") or {} - # Pass through extensions from server (for Bazaar discovery) - extensions = payment_required.get("extensions", {}) - payment_payload = create_payment_payload( - account=self.account, - recipient=details["recipient"], - amount=details["amount"], - network=details.get("network", "eip155:8453"), - resource_url=validate_resource_url( - resource.get("url", f"{self.api_url}/v1/chat/completions"), - self.api_url - ), - resource_description=resource.get("description", "BlockRun AI API call"), - max_timeout_seconds=details.get("maxTimeoutSeconds", 300), - extra=details.get("extra"), - extensions=extensions, - ) - - # Retry with payment - retry_response = self._client.post( - url, - json=body, - headers={ - "Content-Type": "application/json", - "X-Payment": payment_payload, - }, - ) - - # Check for errors - if retry_response.status_code == 402: - raise PaymentError("Payment was rejected. Check your wallet balance.") - - if retry_response.status_code != 200: - try: - error_body = retry_response.json() - except Exception: - error_body = {"error": "Request failed"} - raise APIError( - f"API error after payment: {retry_response.status_code}", - retry_response.status_code, - sanitize_error_response(error_body), - ) - - return ChatResponse(**retry_response.json()) - - def list_models(self) -> List[Dict[str, Any]]: - """ - List available models with pricing. - - Returns: - List of model information dicts - """ - response = self._client.get(f"{self.api_url}/v1/models") - - if response.status_code != 200: - raise APIError( - f"Failed to list models: {response.status_code}", - response.status_code, - ) - - return response.json().get("data", []) - - def get_wallet_address(self) -> str: - """Get the wallet address being used for payments.""" - return self.account.address - - def close(self): - """Close the HTTP client.""" - self._client.close() - - def __enter__(self): - return self - - def __exit__(self, exc_type, exc_val, exc_tb): - self.close() - - -# Async client for async/await usage -class AsyncLLMClient: - """ - Async version of BlockRun LLM Client. - - Usage: - async with AsyncLLMClient() as client: - response = await client.chat("gpt-4o", "Hello!") - """ - - DEFAULT_API_URL = "https://blockrun.ai/api" - DEFAULT_MAX_TOKENS = 1024 - - def __init__( - self, - private_key: Optional[str] = None, - api_url: Optional[str] = None, - timeout: float = 60.0, - ): - key = private_key or os.environ.get("BLOCKRUN_WALLET_KEY") - if not key: - raise ValueError( - "Private key required. Set BLOCKRUN_WALLET_KEY env or pass private_key." - ) - - # Validate private key format - validate_private_key(key) - - self.account = Account.from_key(key) - - # Validate and set API URL - api_url_raw = api_url or os.environ.get("BLOCKRUN_API_URL") or self.DEFAULT_API_URL - validate_api_url(api_url_raw) - self.api_url = api_url_raw.rstrip("/") - - self.timeout = timeout - self._client = httpx.AsyncClient(timeout=timeout) - - async def chat( - self, - model: str, - prompt: str, - *, - system: Optional[str] = None, - max_tokens: Optional[int] = None, - temperature: Optional[float] = None, - ) -> str: - """Async 1-line chat interface.""" - messages: List[Dict[str, str]] = [] - - if system: - messages.append({"role": "system", "content": system}) - - messages.append({"role": "user", "content": prompt}) - - result = await self.chat_completion( - model=model, - messages=messages, - max_tokens=max_tokens, - temperature=temperature, - ) - - return result.choices[0].message.content - - async def chat_completion( - self, - model: str, - messages: List[Dict[str, str]], - *, - max_tokens: Optional[int] = None, - temperature: Optional[float] = None, - top_p: Optional[float] = None, - ) -> ChatResponse: - """Async full chat completion interface.""" - # Validate inputs - validate_model(model) - validate_max_tokens(max_tokens) - validate_temperature(temperature) - validate_top_p(top_p) - - body: Dict[str, Any] = { - "model": model, - "messages": messages, - "max_tokens": max_tokens or self.DEFAULT_MAX_TOKENS, - } - - if temperature is not None: - body["temperature"] = temperature - if top_p is not None: - body["top_p"] = top_p - - return await self._request_with_payment("/v1/chat/completions", body) - - async def _request_with_payment(self, endpoint: str, body: Dict[str, Any]) -> ChatResponse: - """Make async request with automatic payment handling.""" - url = f"{self.api_url}{endpoint}" - - response = await self._client.post( - url, - json=body, - headers={"Content-Type": "application/json"}, - ) - - if response.status_code == 402: - return await self._handle_payment_and_retry(url, body, response) - - if response.status_code != 200: - try: - error_body = response.json() - except Exception: - error_body = {"error": "Request failed"} - raise APIError( - f"API error: {response.status_code}", - response.status_code, - sanitize_error_response(error_body), - ) - - return ChatResponse(**response.json()) - - async def _handle_payment_and_retry( - self, - url: str, - body: Dict[str, Any], - response: httpx.Response, - ) -> ChatResponse: - """Handle 402 response asynchronously.""" - payment_header = response.headers.get("X-Payment-Required") - if not payment_header: - try: - resp_body = response.json() - if "x402" in resp_body: - payment_header = resp_body - except Exception: - pass - - if not payment_header: - raise PaymentError("402 response but no payment requirements found") - - if isinstance(payment_header, str): - payment_required = parse_payment_required(payment_header) - else: - payment_required = payment_header - - details = extract_payment_details(payment_required) - - # Create signed payment payload (v2 format) - resource = details.get("resource") or {} - # Pass through extensions from server (for Bazaar discovery) - extensions = payment_required.get("extensions", {}) - payment_payload = create_payment_payload( - account=self.account, - recipient=details["recipient"], - amount=details["amount"], - network=details.get("network", "eip155:8453"), - resource_url=validate_resource_url( - resource.get("url", f"{self.api_url}/v1/chat/completions"), - self.api_url - ), - resource_description=resource.get("description", "BlockRun AI API call"), - max_timeout_seconds=details.get("maxTimeoutSeconds", 300), - extra=details.get("extra"), - extensions=extensions, - ) - - retry_response = await self._client.post( - url, - json=body, - headers={ - "Content-Type": "application/json", - "X-Payment": payment_payload, - }, - ) - - if retry_response.status_code == 402: - raise PaymentError("Payment was rejected. Check your wallet balance.") - - if retry_response.status_code != 200: - try: - error_body = retry_response.json() - except Exception: - error_body = {"error": "Request failed"} - raise APIError( - f"API error after payment: {retry_response.status_code}", - retry_response.status_code, - sanitize_error_response(error_body), - ) - - return ChatResponse(**retry_response.json()) - - async def list_models(self) -> List[Dict[str, Any]]: - """List available models asynchronously.""" - response = await self._client.get(f"{self.api_url}/v1/models") - - if response.status_code != 200: - raise APIError( - f"Failed to list models: {response.status_code}", - response.status_code, - ) - - return response.json().get("data", []) - - def get_wallet_address(self) -> str: - """Get the wallet address.""" - return self.account.address - - async def close(self): - """Close the async HTTP client.""" - await self._client.aclose() - - async def __aenter__(self): - return self - - async def __aexit__(self, exc_type, exc_val, exc_tb): - await self.close() +""" +BlockRun LLM Client - Main SDK entry point. + +SECURITY NOTE - Private Key Handling: +===================================== +Your private key NEVER leaves your machine. Here's what happens: + +1. Key stays local - only used to sign an EIP-712 typed data message +2. Only the SIGNATURE is sent in the PAYMENT-SIGNATURE header +3. BlockRun verifies the signature on-chain via Coinbase CDP facilitator +4. Your actual private key is NEVER transmitted to any server + +This is the same security model as: +- Signing a MetaMask transaction +- Any on-chain swap or trade +- Standard EIP-3009 TransferWithAuthorization + +Usage: + from blockrun_llm import LLMClient + + # Initialize with private key from env (BLOCKRUN_WALLET_KEY) + client = LLMClient() + + # Or pass private key directly + client = LLMClient(private_key="0x...") + + # Simple 1-line chat + response = client.chat("gpt-5.2", "What is 2+2?") + print(response) + + # Full chat with messages + messages = [ + {"role": "system", "content": "You are a helpful assistant."}, + {"role": "user", "content": "Hello!"} + ] + result = client.chat_completion("gpt-5.2", messages) + print(result.choices[0].message.content) +""" + +from __future__ import annotations + +import json as _json +import os +import re +import sys +from collections.abc import AsyncIterator, Iterator +from typing import Any + +import httpx +from dotenv import load_dotenv +from eth_account import Account + +from .apikey import ( + api_key_base_url, + auth_headers, + missing_credential_error, + payment_mode, + raise_for_api_key_402, + resolve_api_key, + wallet_only, +) +from .router_adapter import ( + BASE_MINIMUM_PAYMENT_USD, + build_model_pricing, + route_with_catalog, + routing_profile_for_model, + routing_text, +) +from .tx_log import ( + TransactionLogger, + _resolve_log_dir, + decode_settlement_header, + paid_request_error_prefix, + read_settlement_header, +) +from .types import ( + APIError, + ChatCompletionChunk, + ChatResponse, + ImageResponse, + PaymentError, + RetiredEndpointError, + RoutingDecision, + RoutingProfile, + SearchResult, + SmartChatCompletionResponse, + SmartChatResponse, + chunk_meta, + chunk_usage_dict, + retry_after_of, + stream_choice_content, + stream_choice_finish_reason, +) +from .validation import ( + check_spend_limits, + resolve_spend_limit, + sanitize_error_response, + validate_api_url, + validate_eth_address, + validate_max_tokens, + validate_model, + validate_private_key, + validate_resource_url, + validate_temperature, + validate_top_p, +) +from .x402 import create_payment_payload, extract_payment_details, parse_payment_required + +# Load environment variables +load_dotenv() + +# Default chat HTTP timeout (seconds). Was 120; reasoning models (opus-4.8, +# deepseek-v4-pro) routinely take 200–300s, so 120 timed out non-streaming +# calls. Override via the BLOCKRUN_CHAT_TIMEOUT env var. +DEFAULT_CHAT_TIMEOUT = float(os.environ.get("BLOCKRUN_CHAT_TIMEOUT", "600")) + + +# User-Agent for client identification in server logs +# Version read lazily to avoid circular import with __init__.py +def _get_user_agent() -> str: + from . import __version__ + + return f"blockrun-python/{__version__}" + + +# ============================================================================= +# Standalone Functions (no wallet required) +# ============================================================================= + + +def list_models(api_url: str | None = None) -> list[dict[str, Any]]: + """ + List available LLM models with pricing (no wallet required). + + This is a standalone function that queries the public API endpoint. + No wallet or authentication needed. + + Args: + api_url: API endpoint (default: https://blockrun.ai/api) + + Returns: + List of model dicts with id, name, provider, pricing, context window, etc. + + Example: + from blockrun_llm import list_models + models = list_models() + for m in models: + print(f"{m['id']}: ${m.get('inputPrice', 'N/A')}/M input") + """ + # No credential parameter, so the environment decides. On the account rail + # the catalogue lives on another host and needs the key, so honouring one + # without the other would 404 or 401. + api_key = resolve_api_key(None) + api_url = api_url or (api_key_base_url(None) if api_key else "https://blockrun.ai/api") + with httpx.Client(timeout=30, headers=auth_headers(api_key)) as client: + # The account rail publishes the catalogue at /v1/models only; /pricing + # is the x402 gateway's own sheet. + path = "/v1/models" if api_key else "/pricing" + response = client.get(f"{api_url.rstrip('/')}{path}") + if response.status_code != 200: + raise APIError( + f"Failed to list models: {response.status_code}", + response.status_code, + {}, + retry_after=retry_after_of(response), + ) + data = response.json() + return data.get("data", []) if api_key else data.get("models", []) + + +def list_image_models(api_url: str | None = None) -> list[dict[str, Any]]: + """ + List available image generation models without requiring a wallet. + + Filters the unified ``/v1/models`` catalog by ``categories: ["image"]``. + The dedicated ``/v1/images/models`` endpoint was deprecated server-side; + image models now live alongside chat models under one catalog. + """ + api_key = resolve_api_key(None) + api_url = api_url or (api_key_base_url(None) if api_key else "https://blockrun.ai/api") + with httpx.Client(timeout=30, headers=auth_headers(api_key)) as client: + response = client.get(f"{api_url.rstrip('/')}/v1/models") + if response.status_code != 200: + raise APIError( + f"Failed to list models: {response.status_code}", + response.status_code, + {}, + retry_after=retry_after_of(response), + ) + models = response.json().get("data", []) + return [m for m in models if "image" in (m.get("categories") or [])] + + +# ============================================================================= +# Shared helpers +# ============================================================================= + + +_SETTLED_ATTR = "blockrun_payment_settled" + + +def _mark_settled(exc: BaseException) -> BaseException: + """Tag an exception raised after the x402 payment for this call was signed. + + Signing is settlement. Once the PAYMENT-SIGNATURE has gone out, a retry on + another model is not a free retry: it triggers a fresh 402, a fresh + signature, and a fresh settlement. A six-model fallback chain can therefore + settle six times and return nothing, which the CHANGELOG already records as + a live outcome class ("CHARGED BUT REQUEST FAILED"). The tag is an + attribute rather than a new exception type so callers catching + ``httpx.TimeoutException`` keep working unchanged. + + Applied to every exception escaping the paid leg, not just timeouts. The + dominant post-settlement failure is a paid 5xx, which surfaces as + ``APIError(status_code=503)`` — precisely a status :func:`_should_fallback` + treats as retriable, so tagging only timeouts left the six-settlement path + fully open. Over-tagging is the safe direction here: the handlers begin at an + already-read 402 response and ``create_payment_payload`` is local signing + with no network I/O, so the only exceptions that can be tagged without a + settlement are ones :func:`_should_fallback` already refuses. + """ + setattr(exc, _SETTLED_ATTR, True) + return exc + + +def _should_fallback(exc: Exception) -> bool: + """Whether ``exc`` is the kind of transient failure that warrants trying + the next model in a fallback chain. + + True for: timeouts, network/connection errors, and APIError with 5xx + status codes typically associated with upstream availability problems. + + False for: 4xx client errors, PaymentError (wallet/balance issues), + anything that already cost the caller a settled payment (see + :func:`_mark_settled`), and everything else — those are not "swap upstream + and retry" situations. + """ + if getattr(exc, _SETTLED_ATTR, False): + return False + if isinstance(exc, httpx.TimeoutException): + return True + if isinstance(exc, httpx.NetworkError): + return True + # 429 is retriable here for the same reason the TypeScript adapter treats it + # as transient: it means THIS upstream is saturated, and the next model in + # the chain is a different upstream. Observed live on the free tier — a + # rate-limited free model returned 429 and the three remaining free models + # in the ranked chain were never tried. Permanent payment failures and + # settled calls are refused above, before this line. + return bool(isinstance(exc, APIError) and exc.status_code in (429, 502, 503, 504, 522, 524)) + + +# The gateway states the output-token ceiling it actually quoted in the 402's +# ``resource.description``, e.g. "claude-opus-4.8 ... 128000 max output tokens". +# +# The alternation is bounded on both branches and the leading lookbehind stops a +# match from starting mid-number, so there is no super-linear backtracking. The +# earlier `(\d[\d,]*)` was quadratic on a digit run — measured on CPython 3.13, +# a string of N '9's: 4k 0.13s, 8k 0.49s, 16k 1.95s, and it keeps squaring. This +# runs on a server-controlled string inside the payment path, so a long +# description would have stalled every paid call. Same input, bounded pattern: +# 0.0002s. `test_pattern_itself_is_not_backtracking` pins it. +_QUOTED_MAX_TOKENS_RE = re.compile( + r"(? None: + """Warn when the gateway quoted fewer output tokens than the caller asked for. + + An over-ceiling ``max_tokens`` is not rejected. The gateway silently clamps + to the model's ceiling and prices the clamped value, so the caller pays for + a ceiling they never asked for and never hears about it. The 402's + ``resource.description`` is the only disclosure, and it would otherwise be + passed straight into the signature and discarded. Surfacing it here is the + caller's one chance to learn their value was dropped before they pay. + + Best-effort by construction: if the description doesn't carry exactly one + recognizable ceiling, stay silent rather than guess. A missed warning costs + the caller nothing beyond today's behavior; a wrong one would erode trust in + all of them. The whole body is guarded because this runs on server-controlled + text immediately before signing, and a diagnostic must never be the reason a + paid request fails. + """ + try: + requested = body.get("max_tokens") + # bool is an int subclass; a stray True is not a token count. + if not isinstance(requested, int) or isinstance(requested, bool): + return + # The gateway sends a string here, but the field is server-controlled and + # JSON allows anything; a non-string must not reach re.search. + if not isinstance(resource_description, str) or not resource_description: + return + + matches = _QUOTED_MAX_TOKENS_RE.findall(resource_description[:_DESCRIPTION_SCAN_LIMIT]) + # Two candidates means the format is not what we think it is (a rate like + # "per 1000 max output tokens" would otherwise read as the ceiling). + if len(matches) != 1: + return + quoted = int(matches[0].replace(",", "")) + + if quoted < requested: + sys.stderr.write( + f"[blockrun_llm] max_tokens clamped by the gateway: you asked for " + f"{requested}, {body.get('model', 'this model')} tops out at {quoted}. " + f"You are being quoted for {quoted} output tokens, not {requested}.\n" + ) + except Exception: + # A warning that breaks the request it is warning about is worse than no + # warning. Includes a closed/broken stderr. + return + + +def _enforce_spend_limits(client: Any, cost_usd: float, model: str | None = None) -> None: + """Refuse a quote that breaches a limit the caller configured, before the + paid request is sent. + + A free function rather than a method because the four client classes (sync + and async, Base and Solana) do not share a base class, and a spend limit + that applies to three of them is not a spend limit. + + No-op unless the caller opted in. See + :func:`blockrun_llm.validation.check_spend_limits`. + """ + check_spend_limits( + cost_usd, + max_cost_per_call=client._max_cost_per_call, + max_session_cost=client._max_session_cost, + session_spent_usd=client._session_total_usd, + model=model, + ) + + +def _detect_network(api_url: str) -> str: + """Map an API URL to the canonical network label used in billing + records. Returns ``base-mainnet`` / ``base-sepolia`` / ``solana-mainnet`` + / ``unknown``. + """ + if not api_url: + return "unknown" + if "sol.blockrun" in api_url: + return "solana-mainnet" + if "testnet" in api_url: + return "base-sepolia" + if "blockrun.ai" in api_url: + return "base-mainnet" + return "unknown" + + +# ============================================================================= +# LLM Client Class (requires wallet) +# ============================================================================= + + +class LLMClient: + """ + BlockRun LLM Gateway Client. + + Provides access to multiple LLM providers (OpenAI, Anthropic, Google, etc.) + with automatic x402 micropayments on Base chain. + + Security: Your private key is used ONLY for local EIP-712 signing. + The key NEVER leaves your machine - only signatures are transmitted. + + Networks: + - Mainnet: https://blockrun.ai/api (Base, Chain ID 8453) + - Testnet: https://testnet.blockrun.ai/api (Base Sepolia, Chain ID 84532) + + Testnet Usage: + For development and testing without real USDC: + + client = LLMClient(api_url="https://testnet.blockrun.ai/api") + + # Or use the testnet convenience method + from blockrun_llm import testnet_client + client = testnet_client() + + Note: Testnet has limited models (openai/gpt-oss-20b, openai/gpt-oss-120b) + """ + + DEFAULT_API_URL = "https://blockrun.ai/api" + TESTNET_API_URL = "https://testnet.blockrun.ai/api" + DEFAULT_MAX_TOKENS = 1024 + + def __init__( + self, + private_key: str | None = None, + api_url: str | None = None, + timeout: float = DEFAULT_CHAT_TIMEOUT, + search_timeout: float = 300.0, + transaction_log: bool | str | os.PathLike[str] | None = None, + max_cost_per_call: float | None = None, + max_session_cost: float | None = None, + ): + """ + Initialize the BlockRun LLM client. + + Args: + private_key: Base chain wallet private key (or set BLOCKRUN_WALLET_KEY env var) + NOTE: Key is used for LOCAL signing only - never transmitted + api_url: API endpoint URL (default: https://blockrun.ai/api) + timeout: Request timeout in seconds (default: 600, override via BLOCKRUN_CHAT_TIMEOUT env). Used for regular chat requests. + search_timeout: Timeout for xAI Live Search requests (default: 300 = 5 minutes). + Live Search can be slow as it searches X, web, and news sources. + Auto-detected when search_parameters or search=True is passed. + transaction_log: Opt-in per-call log written to a project folder. + ``True`` → ``./log/``; pass a string/Path for a custom dir; + ``None`` (default) honors the ``BLOCKRUN_TX_LOG`` env var + (set to ``1`` or a path). Each paid call appends one row to + ``transactions.jsonl`` (model, input, output, cost_usd, + tx_hash, on-chain amount, payer, payee, network) and + writes a pretty-printed JSON file next to it. + + Raises: + ValueError: If no wallet is configured. For agent use, call setup_agent_wallet() first. + + Security: + Your private key NEVER leaves your machine. It is only used to sign + EIP-712 typed data locally. Only the signature is sent to the server. + """ + # Get private key from param, environment, or ~/.blockrun/.session file + # SECURITY: Key is stored in memory only, used for LOCAL signing + from .wallet import load_wallet + + # Account rail first, and before any wallet variable is read: an API + # key is a complete credential on its own, so demanding a private key + # alongside it would make every API-key user invent a wallet they + # never use. See apikey.py for the precedence rule. + api_key = resolve_api_key(private_key) + key = ( + None + if api_key + else ( + private_key + or os.environ.get("BLOCKRUN_WALLET_KEY") + or os.environ.get("BASE_CHAIN_WALLET_KEY") + or load_wallet() # Loads from ~/.blockrun/.session + ) + ) + if not api_key and not key: + raise missing_credential_error() + + # Normalize private key format (add 0x prefix if missing) + if key and not key.startswith("0x"): + key = "0x" + key + + # Validate private key format + if key: + validate_private_key(key) + + # Initialize wallet account + # SECURITY: Key stays local, only used to sign EIP-712 messages + # The key is NEVER transmitted - only signatures are sent + self.api_key = api_key + # No wallet on the account rail: nothing is signed locally. + self.account = Account.from_key(key) if key else None + + # Validate and set API URL + # BLOCKRUN_API_URL names an x402 gateway; an API-key client must not + # follow it and hand the key to a host set up for another rail. + api_url_raw = ( + api_key_base_url(api_url) + if api_key + else (api_url or os.environ.get("BLOCKRUN_API_URL") or self.DEFAULT_API_URL) + ) + validate_api_url(api_url_raw) + self.api_url = api_url_raw.rstrip("/") + + self.timeout = timeout + self.search_timeout = search_timeout + + self._client = httpx.Client( + headers=auth_headers(api_key), + timeout=timeout, + limits=httpx.Limits(max_connections=200, max_keepalive_connections=50), + ) + + # Session spending tracking + self._session_total_usd: float = 0.0 + # Opt-in spend limits. None (the default) means unlimited, which is the + # behavior every release before 1.9.0 had: every 402 quote was signed + # automatically with nothing compared against anything. + self._max_cost_per_call = resolve_spend_limit( + max_cost_per_call, "BLOCKRUN_MAX_COST_PER_CALL" + ) + self._max_session_cost = resolve_spend_limit(max_session_cost, "BLOCKRUN_MAX_SESSION_COST") + self._session_calls: int = 0 + self._last_call_cost: float = 0.0 + + # Model pricing cache for smart routing + self._model_pricing_cache: dict[str, dict[str, float]] | None = None + + # Opt-in transaction log + last on-chain settlement payload. The + # settlement is populated from PAYMENT-RESPONSE on every paid retry + # and cleared right before save_to_cache fires so it can't bleed + # across calls when logging is disabled. + log_dir = _resolve_log_dir(transaction_log) + self._tx_logger: TransactionLogger | None = ( + TransactionLogger(log_dir) if log_dir is not None else None + ) + self._last_settlement: dict[str, Any] | None = None + + def _capture_settlement(self, response: httpx.Response) -> dict[str, Any] | None: + """Decode the x402 settlement header on a successful paid response. + + Returns the decoded settlement dict (also stashed on + ``self._last_settlement``) so callers can pass it straight into + ``save_to_cache``. ``None`` when the facilitator didn't include a + settlement header — older facilitators / cached free responses. + """ + header = read_settlement_header(response.headers) + settlement = decode_settlement_header(header) + self._last_settlement = settlement + return settlement + + def _get_model_pricing(self) -> dict[str, dict[str, float]]: + """ + Get model pricing for smart routing (cached for the client's lifetime). + + Returns: + Dict mapping model_id -> {"input_price": x, "output_price": y, + "flat_price": z}. ``flat_price`` is 0 for per-token billing and + non-zero (USD per call) for flat-billed models. + """ + if self._model_pricing_cache is not None: + return self._model_pricing_cache + + pricing = build_model_pricing(self.list_models()) + self._model_pricing_cache = pricing + return pricing + + def route( + self, + prompt: str, + *, + system: str | None = None, + max_tokens: int | None = None, + routing_profile: RoutingProfile = "auto", + requires_structured_output: bool = False, + ) -> RoutingDecision: + """ + Inspect a routing decision without making or paying for a model call. + + The first invocation may fetch the public model catalog for current + prices; routing itself is local and costs nothing. + + Example: + decision = client.route("Prove the Riemann hypothesis") + print(decision.model) # 'deepseek/deepseek-v4-pro' + print(decision.task_type) # 'reasoning' + print(decision.candidates) # ordered fallback chain + """ + decision = route_with_catalog( + prompt, + system, + max_tokens or self.DEFAULT_MAX_TOKENS, + self._get_model_pricing(), + routing_profile=routing_profile, + requires_structured_output=requires_structured_output, + minimum_payment_usd=BASE_MINIMUM_PAYMENT_USD, + ) + return RoutingDecision(**decision) + + def smart_chat( + self, + prompt: str, + *, + system: str | None = None, + max_tokens: int | None = None, + temperature: float | None = None, + routing_profile: RoutingProfile = "auto", + ) -> SmartChatResponse: + """ + Smart chat with automatic model routing. + + Uses BlockRun's product-neutral Router Core portfolio strategy — the + same engine the TypeScript SDK and the gateway run. It classifies the + task shape locally (<1ms, no extra model call), enforces capability + constraints as hard filters, and ranks an ordered candidate portfolio: + the cheapest model that can handle the request wins, and the rest become + the transient-error fallback chain. + + Args: + prompt: User message + system: Optional system prompt + max_tokens: Max tokens to generate (default: 1024) + temperature: Sampling temperature + routing_profile: "free" | "eco" | "auto" | "premium" + - free: NVIDIA's $0 models only — no wallet needed + - eco: Cheapest capable model per tier + - auto: Best balance of cost/quality (default) + - premium: Top-tier models (Anthropic, OpenAI, Moonshot) + + Returns: + SmartChatResponse with response, model, and routing decision + + Example: + result = client.smart_chat("What is 2+2?") + print(result.response) # '4' + print(result.model) # 'google/gemini-3.5-flash' + print(result.routing.method) # 'portfolio' + print(f"Saved {result.routing.savings * 100:.0f}%") + + # With routing profile + result = client.smart_chat( + "Prove the Riemann hypothesis", + routing_profile="premium" # Use top-tier models for complex tasks + ) + """ + decision = route_with_catalog( + prompt, + system, + max_tokens or self.DEFAULT_MAX_TOKENS, + self._get_model_pricing(), + routing_profile=routing_profile, + minimum_payment_usd=BASE_MINIMUM_PAYMENT_USD, + ) + + # Make the chat request with selected model. Pass the remaining ranked + # candidates as fallbacks so a hung upstream (e.g. NVIDIA NIM) doesn't + # hard-fail when smart_chat could just walk to the next capable model. + response = self.chat( + model=decision["model"], + prompt=prompt, + system=system, + max_tokens=max_tokens, + temperature=temperature, + fallback_models=decision.get("fallbacks") or None, + ) + + return SmartChatResponse( + response=response, + model=decision["model"], + routing=RoutingDecision(**decision), + ) + + def smart_chat_completion( + self, + messages: list[dict[str, Any]], + *, + max_tokens: int | None = None, + temperature: float | None = None, + top_p: float | None = None, + search: bool | None = None, + search_parameters: dict[str, Any] | None = None, + tools: list[dict[str, Any]] | None = None, + tool_choice: Any | None = None, + response_format: dict[str, Any] | None = None, + stop: str | list[str] | None = None, + fallback_models: list[str] | None = None, + routing_profile: RoutingProfile = "auto", + **extra: Any, + ) -> SmartChatCompletionResponse: + """ + Smart routing for a full message list (OpenAI-compatible). + + The routing counterpart of ``chat_completion``: tools, tool_choice and + response_format are part of the routing decision, not just the request. + A turn that must call a tool routes to a tool-capable model, a JSON + schema forces a structured-output-capable tier, and image parts route to + a vision model. + + Capacity is checked against the WHOLE transcript, not the last message — + an agent conversation can be 100x its final turn, and a context overflow + is a non-transient error the fallback chain cannot rescue. + + Example: + result = client.smart_chat_completion( + [{"role": "user", "content": "Cancel order B-42"}], + tools=[{"type": "function", "function": {"name": "cancel_order", ...}}], + tool_choice="required", + ) + print(result.model) # a tool-capable model + print(result.routing.task_type) # 'tool_agent' + """ + view = routing_text(messages) + decision = route_with_catalog( + view["prompt"], + view["system_prompt"], + max_tokens or self.DEFAULT_MAX_TOKENS, + self._get_model_pricing(), + routing_profile=routing_profile, + requires_structured_output=response_format is not None, + tools=tools, + tool_choice=tool_choice, + conversation_chars=view["conversation_chars"], + has_vision=view["has_vision"], + minimum_payment_usd=BASE_MINIMUM_PAYMENT_USD, + ) + response = self.chat_completion( + decision["model"], + messages, + max_tokens=max_tokens, + temperature=temperature, + top_p=top_p, + search=search, + search_parameters=search_parameters, + tools=tools, + tool_choice=tool_choice, + response_format=response_format, + stop=stop, + # An explicit caller-supplied chain wins over the routed one. + fallback_models=fallback_models or decision.get("fallbacks") or None, + **extra, + ) + return SmartChatCompletionResponse( + response=response, + model=decision["model"], + routing=RoutingDecision(**decision), + ) + + def get_spending(self) -> dict[str, Any]: + """ + Get current session spending. + + Returns: + Dict with total_usd and calls count + + Example: + spending = client.get_spending() + print(f"Spent ${spending['total_usd']:.4f} across {spending['calls']} calls") + """ + return { + "total_usd": self._session_total_usd, + "calls": self._session_calls, + } + + def chat( + self, + model: str, + prompt: str, + *, + system: str | None = None, + max_tokens: int | None = None, + temperature: float | None = None, + search: bool | None = None, + search_parameters: dict[str, Any] | None = None, + response_format: dict[str, Any] | None = None, + stop: str | list[str] | None = None, + fallback_models: list[str] | None = None, + **extra: Any, + ) -> str: + """ + Simple 1-line chat interface. + + Args: + model: Model ID (e.g., "openai/gpt-5.2", "anthropic/claude-sonnet-4.6", "openai/gpt-5.2") + prompt: User message + system: Optional system prompt + max_tokens: Max tokens to generate (default: 1024) + temperature: Sampling temperature + search: Enable xAI Live Search (shortcut for search_parameters={"mode": "on"}) + search_parameters: Full xAI Live Search configuration (for search-enabled models) + See: https://docs.x.ai/docs/guides/live-search + + Returns: + Assistant's response text + + Example: + response = client.chat("openai/gpt-5.2", "What is the capital of France?") + + # Check spending after calls + spending = client.get_spending() + print(f"Spent ${spending['total_usd']:.4f}") + + # With xAI Live Search (for real-time X/Twitter data) + response = client.chat( + "openai/gpt-5.2", + "What are the latest posts from @blockrunai?", + search=True # Enable live search + ) + """ + messages: list[dict[str, str]] = [] + + if system: + messages.append({"role": "system", "content": system}) + + messages.append({"role": "user", "content": prompt}) + + result = self.chat_completion( + model=model, + messages=messages, + max_tokens=max_tokens, + temperature=temperature, + search=search, + search_parameters=search_parameters, + response_format=response_format, + stop=stop, + fallback_models=fallback_models, + **extra, + ) + + return result.choices[0].message.content + + def chat_completion( + self, + model: str, + messages: list[dict[str, Any]], + *, + max_tokens: int | None = None, + temperature: float | None = None, + top_p: float | None = None, + search: bool | None = None, + search_parameters: dict[str, Any] | None = None, + tools: list[dict[str, Any]] | None = None, + tool_choice: Any | None = None, + response_format: dict[str, Any] | None = None, + stop: str | list[str] | None = None, + fallback_models: list[str] | None = None, + **extra: Any, + ) -> ChatResponse: + """ + Full chat completion interface (OpenAI-compatible). + + Args: + model: Model ID + messages: List of message dicts with 'role' and 'content' + max_tokens: Max tokens to generate + temperature: Sampling temperature + top_p: Nucleus sampling parameter + search: Enable xAI Live Search (shortcut for search_parameters={"mode": "on"}) + search_parameters: Full xAI Live Search configuration (for search-enabled models) + tools: List of tool definitions for function calling + tool_choice: Tool selection strategy ("none", "auto", "required", or specific tool) + response_format: OpenAI response format, e.g. {"type": "json_object"} for JSON mode. + Works across all providers — the gateway natively forwards it to OpenAI/Azure + and injects a raw-JSON system instruction (stripping any code fence) for + Anthropic/Bedrock models. + stop: Up to 4 stop sequences (str or list of str). The gateway forwards these + natively to OpenAI and maps them to stop_sequences for Anthropic/Bedrock. + + Returns: + ChatResponse object with choices, usage, and citations (if search enabled) + + Raises: + PaymentError: If the gateway rejects the signed payment (most often + an insufficient USDC balance). + SpendLimitError: If the quote exceeds ``max_cost_per_call`` or would + push the client past ``max_session_cost``. Both are opt-in and + unset by default; when unset, every 402 quote is signed + automatically. Raised before the request is sent, so a refused + quote costs nothing. ``SpendLimitError`` subclasses + ``PaymentError``. + + Example: + messages = [ + {"role": "system", "content": "You are helpful."}, + {"role": "user", "content": "Hello!"} + ] + result = client.chat_completion("gpt-5.2", messages) + + # With xAI Live Search + result = client.chat_completion( + "openai/gpt-5.2", + [{"role": "user", "content": "Latest news about AI?"}], + search=True + ) + print(result.citations) # URLs of sources used + + # With tool calling + tools = [{ + "type": "function", + "function": { + "name": "get_weather", + "description": "Get the current weather", + "parameters": { + "type": "object", + "properties": { + "location": {"type": "string"} + }, + "required": ["location"] + } + } + }] + result = client.chat_completion("gpt-5.2", messages, tools=tools) + if result.choices[0].message.tool_calls: + for tc in result.choices[0].message.tool_calls: + print(f"Call: {tc.function.name}({tc.function.arguments})") + + # Virtual routing ids pick the model for you + result = client.chat_completion("blockrun/auto", messages) + """ + # `blockrun/auto` | `blockrun/eco` | `blockrun/premium` are not models — + # they select a routing profile. Hand the turn to the routed path, which + # also supplies the ranked fallback chain. + virtual_profile = routing_profile_for_model(model) + if virtual_profile is not None: + return self.smart_chat_completion( + messages, + max_tokens=max_tokens, + temperature=temperature, + top_p=top_p, + search=search, + search_parameters=search_parameters, + tools=tools, + tool_choice=tool_choice, + response_format=response_format, + stop=stop, + fallback_models=fallback_models, + routing_profile=virtual_profile, # type: ignore[arg-type] + **extra, + ).response + + # Validate inputs + validate_model(model) + validate_max_tokens(max_tokens) + validate_temperature(temperature) + validate_top_p(top_p) + + # Build request body + body: dict[str, Any] = { + "model": model, + "messages": messages, + "max_tokens": max_tokens or self.DEFAULT_MAX_TOKENS, + } + + if temperature is not None: + body["temperature"] = temperature + if top_p is not None: + body["top_p"] = top_p + + # Handle xAI Live Search parameters + if search_parameters is not None: + body["search_parameters"] = search_parameters + elif search is True: + # Simple shortcut: search=True enables live search with defaults + body["search_parameters"] = {"mode": "on"} + + # Handle tool calling + if tools is not None: + body["tools"] = tools + if tool_choice is not None: + body["tool_choice"] = tool_choice + + # OpenAI-compatible response shaping (honored by the gateway across providers) + if response_format is not None: + body["response_format"] = response_format + if stop is not None: + body["stop"] = stop + + # Passthrough: forward any other caller-supplied params verbatim. Named + # params above take precedence; `extra` only fills keys not already set. + for k, v in extra.items(): + if v is not None: + body.setdefault(k, v) + + # Walk [model, *fallback_models] on retriable errors (timeouts, 5xx, + # network errors). Default behavior — single attempt — is preserved + # when fallback_models is None or empty. + attempts = [model, *(fallback_models or [])] + last_exc: Exception | None = None + for i, attempt_model in enumerate(attempts): + body["model"] = attempt_model + try: + return self._request_with_payment("/v1/chat/completions", body) + except Exception as exc: + if not _should_fallback(exc): + raise + last_exc = exc + if i + 1 < len(attempts): + next_model = attempts[i + 1] + sys.stderr.write( + f"[blockrun_llm] {attempt_model} -> {next_model} " + f"({type(exc).__name__}: {str(exc)[:80]})\n" + ) + # Exhausted all attempts — re-raise the last retriable error. + assert last_exc is not None # at least one attempt always runs + raise last_exc + + # ------------------------------------------------------------------ + # Streaming (SSE) chat completions + # ------------------------------------------------------------------ + + def chat_completion_stream( + self, + model: str, + messages: list[dict[str, Any]], + *, + max_tokens: int | None = None, + temperature: float | None = None, + top_p: float | None = None, + tools: list[dict[str, Any]] | None = None, + tool_choice: Any | None = None, + search: bool | None = None, + search_parameters: dict[str, Any] | None = None, + response_format: dict[str, Any] | None = None, + stop: str | list[str] | None = None, + fallback_models: list[str] | None = None, + **extra: Any, + ) -> Iterator[ChatCompletionChunk]: + """ + Stream a chat completion via Server-Sent Events. + + Yields one :class:`ChatCompletionChunk` per SSE ``data:`` line until + the upstream emits ``data: [DONE]``. The first chunk's ``delta`` is + typically ``{"role": "assistant"}``; subsequent chunks carry + ``content`` deltas; the final chunk carries ``finish_reason``. + + Payment flow is the same as :meth:`chat_completion`: the first + request returns 402, the SDK signs an EIP-712 payment locally, then + re-issues the request with ``stream=true`` and the + ``PAYMENT-SIGNATURE`` header. Free models (e.g. + ``nvidia/deepseek-v4-flash``) skip the 402 and stream directly. + + Fallback semantics + ------------------ + ``fallback_models=[...]`` walks the list when the primary upstream + produces a retriable error (timeouts, network errors, 5xx). Unlike + the non-streaming :meth:`chat_completion` path, fallback is only + possible **before the first chunk is yielded** — once any byte has + reached the caller, switching models would concatenate two distinct + responses. After-first-chunk failures propagate to the caller. + + Example:: + + for chunk in client.chat_completion_stream( + "nvidia/deepseek-v4-flash", + [{"role": "user", "content": "Hello"}], + fallback_models=["nvidia/llama-4-maverick"], + ): + delta = chunk.choices[0].delta + if delta.content: + print(delta.content, end="", flush=True) + + Note: ``search`` / ``search_parameters`` are not supported in stream + mode by the BlockRun backend — the server will reject with 400. + Codex / GPT-5.4 Pro also do not support streaming. + """ + validate_model(model) + validate_max_tokens(max_tokens) + validate_temperature(temperature) + validate_top_p(top_p) + + body: dict[str, Any] = { + "model": model, + "messages": messages, + "stream": True, + "max_tokens": max_tokens or self.DEFAULT_MAX_TOKENS, + } + if temperature is not None: + body["temperature"] = temperature + if top_p is not None: + body["top_p"] = top_p + if tools is not None: + body["tools"] = tools + if tool_choice is not None: + body["tool_choice"] = tool_choice + if search_parameters is not None: + body["search_parameters"] = search_parameters + elif search is True: + body["search_parameters"] = {"mode": "on"} + if response_format is not None: + body["response_format"] = response_format + if stop is not None: + body["stop"] = stop + + # Passthrough: forward any other caller-supplied params verbatim. + for k, v in extra.items(): + if v is not None: + body.setdefault(k, v) + + attempts = [model, *(fallback_models or [])] + last_exc: Exception | None = None + + for i, attempt_model in enumerate(attempts): + body["model"] = attempt_model + inner = self._stream_with_payment("/v1/chat/completions", body) + chunks_yielded = 0 + try: + for chunk in inner: + chunks_yielded += 1 + yield chunk + return # finished cleanly + except Exception as exc: + if chunks_yielded > 0: + # Already streamed partial output; can't swap models now. + raise + if not _should_fallback(exc): + raise + last_exc = exc + if i + 1 < len(attempts): + next_model = attempts[i + 1] + sys.stderr.write( + f"[blockrun_llm] stream {attempt_model} -> {next_model} " + f"({type(exc).__name__}: {str(exc)[:80]})\n" + ) + # Exhausted all attempts — re-raise the last retriable error. + assert last_exc is not None # at least one attempt always runs + raise last_exc + + # Streaming retry policy. Both the probe (unauthenticated) and the + # paid-retry (with PAYMENT-SIGNATURE) honor this — total tries per + # phase is ``1 + len(_STREAM_5XX_BACKOFFS)`` (== 4 here). Exponential + # backoff so we don't hammer a struggling upstream. + _STREAM_5XX_STATUSES = (500, 502, 503, 504) + _STREAM_5XX_BACKOFFS = (1.0, 2.0, 4.0) + + def _stream_with_payment( + self, + endpoint: str, + body: dict[str, Any], + ) -> Iterator[ChatCompletionChunk]: + """ + Run the 402 → sign → retry dance, then yield SSE chunks. + + Free models return 200 + SSE on the first request; paid models + return JSON 402 first, after which we sign locally and re-stream. + Transient 5xx responses (NVIDIA NIM hiccups, etc.) are retried + in-band with exponential backoff before raising. + """ + url = f"{self.api_url}{endpoint}" + req_headers = {"Content-Type": "application/json", "User-Agent": _get_user_agent()} + + is_search = "search_parameters" in body or body.get("search") is True + timeout = self.search_timeout if is_search else self.timeout + + # ----- Phase 1: probe (no payment header) ----- + payment_headers: dict[str, str] | None = None + cost_usd = 0.0 + + backoffs = self._STREAM_5XX_BACKOFFS + for attempt in range(len(backoffs) + 1): + with self._client.stream( + "POST", url, json=body, headers=req_headers, timeout=timeout + ) as resp1: + if resp1.status_code == 200: + # Free model (or already-authed session) — stream directly. + yield from self._iter_sse_chunks(resp1) + return + resp1.read() + if resp1.status_code == 402: + # Account rail: a 402 is the account being out of credit, not a + # challenge to sign. Nothing here can sign, so say so plainly. + raise_for_api_key_402(resp1, self.api_key) + payment_headers, cost_usd = self._sign_payment_from_response(body, resp1) + break # advance to phase 2 + if resp1.status_code in self._STREAM_5XX_STATUSES and attempt < len(backoffs): + import time + + time.sleep(backoffs[attempt]) + continue + # Out of retries on 5xx, or non-retriable 4xx. + self._raise_stream_error(resp1, after_payment=False) + else: + # Loop exhausted without 402 or 200 — shouldn't reach here because + # the final iteration above raises, but defensive. + raise APIError("stream probe exhausted retries", 0, None) + + # ----- Phase 2: stream with PAYMENT-SIGNATURE ----- + # Signing above was settlement. A timeout here has already been paid + # for, so tag it: the stream fallback chain must not settle again on + # the next model just because zero chunks arrived. + assert payment_headers is not None # break implies signing succeeded + try: + yield from self._stream_paid_phase(url, body, payment_headers, cost_usd, timeout) + except (httpx.HTTPError, APIError) as exc: + _mark_settled(exc) + raise + + def _stream_paid_phase( + self, + url: str, + body: dict[str, Any], + payment_headers: dict[str, str], + cost_usd: float, + timeout: float | None, + ) -> Iterator[ChatCompletionChunk]: + """Phase 2 of :meth:`_stream_with_payment`: the paid, already-settled leg.""" + backoffs = self._STREAM_5XX_BACKOFFS + for attempt in range(len(backoffs) + 1): + with self._client.stream( + "POST", url, json=body, headers=payment_headers, timeout=timeout + ) as resp2: + if resp2.status_code == 200: + if cost_usd > 0: + self._session_calls += 1 + self._session_total_usd += cost_usd + self._last_call_cost = cost_usd + self._capture_settlement(resp2) + yield from self._iter_and_archive(resp2, body, cost_usd, streaming=True) + return + resp2.read() + if resp2.status_code == 402: + # Account rail: a 402 is the account being out of credit, not a + # challenge to sign. Nothing here can sign, so say so plainly. + raise_for_api_key_402(resp2, self.api_key) + raise PaymentError("Payment was rejected. Check your wallet balance.") + if resp2.status_code in self._STREAM_5XX_STATUSES and attempt < len(backoffs): + import time + + time.sleep(backoffs[attempt]) + continue + self._raise_stream_error(resp2, after_payment=True) + + def _iter_and_archive( + self, + response: httpx.Response, + body: dict[str, Any], + cost_usd: float, + *, + streaming: bool = True, + ) -> Iterator[ChatCompletionChunk]: + """Yield each SSE chunk, accumulate content for the local archive, + then once ``data: [DONE]`` arrives ``save_to_cache`` the assembled + ``chat.completion`` response so paid streaming calls show up in + ``~/.blockrun/cost_log.jsonl`` and ``~/.blockrun/data/`` the same + way non-stream paid calls do.""" + assembled_id: str | None = None + assembled_model: str | None = None + assembled_created: int = 0 + content_parts: list[str] = [] + finish_reason: str | None = None + usage_dict: dict[str, Any] | None = None + + for chunk in self._iter_sse_chunks(response): + if chunk.choices: + choice = chunk.choices[0] + content = stream_choice_content(choice) + if content: + content_parts.append(content) + fr = stream_choice_finish_reason(choice) + if fr: + finish_reason = fr + if assembled_id is None: + _id, _model, _created = chunk_meta(chunk) + if _id: + assembled_id = _id + assembled_model = _model + assembled_created = _created + _usage = chunk_usage_dict(chunk) + if _usage is not None: + usage_dict = _usage + # Attach the real per-call x402 charge to every chunk. This is the + # streaming analogue of ChatResponse.cost_usd: it rides on the + # per-call chunk object (race-free), unlike self._last_call_cost + # which goes stale under shared-client concurrency. Consumers + # (e.g. the blockrun-litellm adapter) read it off the chunk to + # report the real wallet deduction instead of a list-price estimate. + chunk.cost_usd = cost_usd + yield chunk + + # Stream complete (saw [DONE]). Free models have cost_usd == 0; only + # archive paid calls to mirror the non-stream save_to_cache path. + if cost_usd > 0: + from .cache import save_to_cache + + response_data: dict[str, Any] = { + "id": assembled_id or "stream", + "object": "chat.completion", + "created": assembled_created or int(__import__("time").time()), + "model": assembled_model or body.get("model"), + "choices": [ + { + "index": 0, + "message": { + "role": "assistant", + "content": "".join(content_parts), + }, + "finish_reason": finish_reason, + } + ], + "stream": streaming, + } + if usage_dict: + response_data["usage"] = usage_dict + try: + save_to_cache( + "/v1/chat/completions", + body, + response_data, + cost_usd=cost_usd, + **self._billing_meta(), + ) + except Exception: + # Logging never breaks the call. + pass + self._log_transaction("/v1/chat/completions", body, response_data, cost_usd) + + @staticmethod + def _iter_sse_chunks(response: httpx.Response) -> Iterator[ChatCompletionChunk]: + """Parse a ``text/event-stream`` response into chunk objects. + + OpenAI format: each event is ``data: {json}\\n\\n``; the terminator is + ``data: [DONE]\\n\\n``. Non-``data:`` lines (comments, heartbeats) + are ignored, and malformed chunks are skipped rather than abort the + stream — partial output is still useful. + """ + for raw_line in response.iter_lines(): + if not raw_line or not raw_line.startswith("data: "): + continue + payload = raw_line[6:].strip() + if payload == "[DONE]": + return + try: + chunk_dict = _json.loads(payload) + except Exception: + continue + try: + yield ChatCompletionChunk(**chunk_dict) + except Exception: + # Schema drift — surface the raw dict shape via a permissive + # model construction to avoid silently dropping output. + yield ChatCompletionChunk.model_construct(**chunk_dict) + + def _sign_payment_from_response( + self, + body: dict[str, Any], + response: httpx.Response, + ) -> tuple[dict[str, str], float]: + """ + Extract a 402's payment requirements, sign locally, and return + ``(headers_with_PAYMENT_SIGNATURE, cost_usd)``. + + Mirrors the inline signing logic in :meth:`_handle_payment_and_retry` + but returns the signed headers instead of doing the retry POST — + which lets the streaming path open an SSE connection for the retry. + """ + payment_header = response.headers.get("payment-required") + price_info: dict[str, Any] = {} + if not payment_header: + try: + resp_body = response.json() + if "x402" in resp_body: + payment_header = resp_body + price_info = resp_body.get("price", {}) + except Exception: + pass + + if not payment_header: + raise PaymentError("402 response but no payment requirements found") + + if isinstance(payment_header, str): + payment_required = parse_payment_required(payment_header) + else: + payment_required = payment_header + + details = extract_payment_details(payment_required) + + cost_usd = ( + float(price_info.get("amount", 0)) + if price_info + else float(details.get("amount", 0)) / 1e6 + ) + # Before signing: a refused quote is never sent, so nothing settles. + _enforce_spend_limits(self, cost_usd, body.get("model") if isinstance(body, dict) else None) + + resource = details.get("resource") or {} + _warn_if_clamped(body, resource.get("description")) + extensions = payment_required.get("extensions", {}) + payment_payload = create_payment_payload( + account=self.account, + recipient=details["recipient"], + amount=details["amount"], + network=details.get("network", "eip155:84532" if self.is_testnet() else "eip155:8453"), + resource_url=validate_resource_url( + resource.get("url", f"{self.api_url}/v1/chat/completions"), self.api_url + ), + resource_description=resource.get("description", "BlockRun AI API call"), + max_timeout_seconds=details.get("maxTimeoutSeconds", 300), + extra=details.get("extra"), + extensions=extensions, + asset=details.get("asset"), + ) + + return ( + { + "Content-Type": "application/json", + "User-Agent": _get_user_agent(), + "PAYMENT-SIGNATURE": payment_payload, + }, + cost_usd, + ) + + @staticmethod + def _raise_stream_error(response: httpx.Response, *, after_payment: bool) -> None: + """Common error path for unexpected HTTP statuses during streaming.""" + try: + error_body = response.json() + except Exception: + error_body = {"error": "Stream request failed"} + prefix = paid_request_error_prefix(response.headers) if after_payment else "API error" + raise APIError( + f"{prefix}: {response.status_code}", + response.status_code, + sanitize_error_response(error_body), + retry_after=retry_after_of(response), + ) + + def _request_with_payment(self, endpoint: str, body: dict[str, Any]) -> ChatResponse: + """ + Make a request with automatic x402 payment handling. + + 1. Send initial request + 2. If 402, parse payment requirements + 3. Sign payment locally + 4. Retry with X-Payment header + """ + url = f"{self.api_url}{endpoint}" + req_headers = {"Content-Type": "application/json", "User-Agent": _get_user_agent()} + + # First attempt (will likely return 402) + response = self._client.post(url, json=body, headers=req_headers) + + # Auto-retry on transient server errors + if response.status_code in (502, 503): + import time + + time.sleep(1) + response = self._client.post(url, json=body, headers=req_headers) + + # Handle 402 Payment Required + if response.status_code == 402: + # Account rail: a 402 is the account being out of credit, not a + # challenge to sign. Nothing here can sign, so say so plainly. + raise_for_api_key_402(response, self.api_key) + # Everything inside signs first, then makes the paid request, so a + # timeout or network error escaping it already cost a settlement. + # Tag it so the fallback chain doesn't settle again on the next model. + try: + return self._handle_payment_and_retry(url, body, response) + except (httpx.HTTPError, APIError) as exc: + _mark_settled(exc) + raise + + # Handle other errors + if response.status_code != 200: + try: + error_body = response.json() + except Exception: + error_body = {"error": "Request failed"} + raise APIError( + f"API error: {response.status_code}", + response.status_code, + sanitize_error_response(error_body), + retry_after=retry_after_of(response), + ) + + # Parse successful response. A 200 on the first attempt means no payment + # was required (free model / cached upstream), so the real charge is $0. + chat_response = ChatResponse(**response.json()) + chat_response.cost_usd = 0.0 + return chat_response + + def _handle_payment_and_retry( + self, + url: str, + body: dict[str, Any], + response: httpx.Response, + ) -> ChatResponse: + """ + Handle 402 response: parse requirements, sign payment locally, retry. + + SECURITY: Payment signing happens entirely on your machine. + Only the signature is sent - your private key never leaves. + """ + # Get payment required header (x402 library uses lowercase) + payment_header = response.headers.get("payment-required") + price_info = {} + if not payment_header: + # Try to get from response body + try: + resp_body = response.json() + if "x402" in resp_body: + payment_header = resp_body + # Extract price info for spending report + price_info = resp_body.get("price", {}) + except Exception: + pass + + if not payment_header: + raise PaymentError("402 response but no payment requirements found") + + # Parse payment requirements + if isinstance(payment_header, str): + payment_required = parse_payment_required(payment_header) + else: + payment_required = payment_header + + # Extract payment details + details = extract_payment_details(payment_required) + + # Get the cost being paid + cost_usd = ( + float(price_info.get("amount", 0)) + if price_info + else float(details.get("amount", 0)) / 1e6 + ) + # Before signing: a refused quote is never sent, so nothing settles. + _enforce_spend_limits(self, cost_usd, body.get("model") if isinstance(body, dict) else None) + + # Create signed payment payload (v2 format) + # SECURITY: Signing happens locally - only the signature is sent to server + resource = details.get("resource") or {} + _warn_if_clamped(body, resource.get("description")) + # Pass through extensions from server (for Bazaar discovery) + extensions = payment_required.get("extensions", {}) + payment_payload = create_payment_payload( + account=self.account, + recipient=details["recipient"], + amount=details["amount"], + network=details.get("network", "eip155:84532" if self.is_testnet() else "eip155:8453"), + resource_url=validate_resource_url( + resource.get("url", f"{self.api_url}/v1/chat/completions"), self.api_url + ), + resource_description=resource.get("description", "BlockRun AI API call"), + max_timeout_seconds=details.get("maxTimeoutSeconds", 300), + extra=details.get("extra"), + extensions=extensions, + asset=details.get("asset"), + ) + + # Retry with payment (x402 library expects PAYMENT-SIGNATURE header) + # Use longer timeout for Live Search requests + is_search_request = "search_parameters" in body or body.get("search") is True + request_timeout = self.search_timeout if is_search_request else self.timeout + + payment_headers = { + "Content-Type": "application/json", + "User-Agent": _get_user_agent(), + "PAYMENT-SIGNATURE": payment_payload, + } + + # Retry with payment, with one automatic retry on 502/503 + retry_response = self._client.post( + url, json=body, headers=payment_headers, timeout=request_timeout + ) + if retry_response.status_code in (502, 503): + import time + + time.sleep(1) + retry_response = self._client.post( + url, json=body, headers=payment_headers, timeout=request_timeout + ) + + # Check for errors + if retry_response.status_code == 402: + # Account rail: a 402 is the account being out of credit, not a + # challenge to sign. Nothing here can sign, so say so plainly. + raise_for_api_key_402(retry_response, self.api_key) + raise PaymentError("Payment was rejected. Check your wallet balance.") + + if retry_response.status_code != 200: + try: + error_body = retry_response.json() + except Exception: + error_body = {"error": "Request failed"} + raise APIError( + f"{paid_request_error_prefix(retry_response.headers)}: {retry_response.status_code}", + retry_response.status_code, + sanitize_error_response(error_body), + retry_after=retry_after_of(retry_response), + ) + + # Parse response + response_data = retry_response.json() + chat_response = ChatResponse(**response_data) + + # Update session spending + self._session_calls += 1 + self._session_total_usd += cost_usd + self._last_call_cost = cost_usd + settlement = self._capture_settlement(retry_response) + + # Attach the real x402 charge (and on-chain settlement) to THIS response + # object so callers get a per-call, race-free cost. Use the value + # _capture_settlement returns rather than re-reading self._last_settlement + # (shared state a concurrent call on the same client could overwrite), + # and a local cost_usd rather than self._last_call_cost which goes stale. + chat_response.cost_usd = cost_usd + if settlement: + chat_response.settlement = dict(settlement) + + # Save full response locally (cost log + response archive) + from .cache import save_to_cache + + save_to_cache( + "/v1/chat/completions", + body, + response_data, + cost_usd=cost_usd, + **self._billing_meta(), + ) + self._log_transaction("/v1/chat/completions", body, response_data, cost_usd) + + return chat_response + + def _request_with_payment_raw(self, endpoint: str, body: dict[str, Any]) -> dict[str, Any]: + """ + Make a request with automatic x402 payment handling, returning raw JSON. + + Same flow as _request_with_payment() but returns Dict instead of ChatResponse. + Used for endpoints that don't return the chat completion shape. + Checks local cache first to avoid paying twice for the same data. + """ + from .cache import get_cached, save_to_cache + + # Check cache first — don't pay twice for same data + cached = get_cached(endpoint, body) + if cached is not None: + return cached + + url = f"{self.api_url}{endpoint}" + req_headers = {"Content-Type": "application/json", "User-Agent": _get_user_agent()} + + response = self._client.post(url, json=body, headers=req_headers) + + # Auto-retry on transient server errors + if response.status_code in (502, 503): + import time + + time.sleep(1) + response = self._client.post(url, json=body, headers=req_headers) + + if response.status_code == 402: + # Account rail: a 402 is the account being out of credit, not a + # challenge to sign. Nothing here can sign, so say so plainly. + raise_for_api_key_402(response, self.api_key) + try: + result = self._handle_payment_and_retry_raw(url, body, response) + except (httpx.HTTPError, APIError) as exc: + _mark_settled(exc) + raise + # Save paid response to cache + save_to_cache( + endpoint, + body, + result, + cost_usd=self._last_call_cost, + **self._billing_meta(), + ) + self._log_transaction(endpoint, body, result, self._last_call_cost) + return result + + if response.status_code != 200: + try: + error_body = response.json() + except Exception: + error_body = {"error": "Request failed"} + raise APIError( + f"API error: {response.status_code}", + response.status_code, + sanitize_error_response(error_body), + retry_after=retry_after_of(response), + ) + + return response.json() + + def _handle_payment_and_retry_raw( + self, + url: str, + body: dict[str, Any], + response: httpx.Response, + ) -> dict[str, Any]: + """Handle 402 response for raw endpoints: parse requirements, sign payment, retry.""" + payment_header = response.headers.get("payment-required") + price_info = {} + if not payment_header: + try: + resp_body = response.json() + if "x402" in resp_body: + payment_header = resp_body + price_info = resp_body.get("price", {}) + except Exception: + pass + + if not payment_header: + raise PaymentError("402 response but no payment requirements found") + + if isinstance(payment_header, str): + payment_required = parse_payment_required(payment_header) + else: + payment_required = payment_header + + details = extract_payment_details(payment_required) + + cost_usd = ( + float(price_info.get("amount", 0)) + if price_info + else float(details.get("amount", 0)) / 1e6 + ) + # Before signing: a refused quote is never sent, so nothing settles. + _enforce_spend_limits(self, cost_usd, body.get("model") if isinstance(body, dict) else None) + + resource = details.get("resource") or {} + _warn_if_clamped(body, resource.get("description")) + extensions = payment_required.get("extensions", {}) + payment_payload = create_payment_payload( + account=self.account, + recipient=details["recipient"], + amount=details["amount"], + network=details.get("network", "eip155:84532" if self.is_testnet() else "eip155:8453"), + resource_url=validate_resource_url(resource.get("url", url), self.api_url), + resource_description=resource.get("description", "BlockRun AI API call"), + max_timeout_seconds=details.get("maxTimeoutSeconds", 300), + extra=details.get("extra"), + extensions=extensions, + asset=details.get("asset"), + ) + + payment_headers = { + "Content-Type": "application/json", + "User-Agent": _get_user_agent(), + "PAYMENT-SIGNATURE": payment_payload, + } + + # Retry with payment, with one automatic retry on 502/503 + retry_response = self._client.post( + url, json=body, headers=payment_headers, timeout=self.timeout + ) + if retry_response.status_code in (502, 503): + import time + + time.sleep(1) + retry_response = self._client.post( + url, json=body, headers=payment_headers, timeout=self.timeout + ) + + if retry_response.status_code == 402: + # Account rail: a 402 is the account being out of credit, not a + # challenge to sign. Nothing here can sign, so say so plainly. + raise_for_api_key_402(retry_response, self.api_key) + raise PaymentError("Payment was rejected. Check your wallet balance.") + + if retry_response.status_code != 200: + try: + error_body = retry_response.json() + except Exception: + error_body = {"error": "Request failed"} + raise APIError( + f"{paid_request_error_prefix(retry_response.headers)}: {retry_response.status_code}", + retry_response.status_code, + sanitize_error_response(error_body), + retry_after=retry_after_of(retry_response), + ) + + self._session_calls += 1 + self._session_total_usd += cost_usd + self._last_call_cost = cost_usd + self._capture_settlement(retry_response) + + return retry_response.json() + + def _get_with_payment_raw( + self, endpoint: str, params: dict[str, Any] | None = None + ) -> dict[str, Any]: + """ + GET with automatic x402 payment handling, returning raw JSON. + + Same flow as _request_with_payment_raw() but uses GET with query params + instead of POST with JSON body. Used for Predexon prediction market endpoints. + """ + from .cache import get_cached, save_to_cache + + cache_key_body = params or {} + cached = get_cached(endpoint, cache_key_body) + if cached is not None: + return cached + + url = f"{self.api_url}{endpoint}" + req_headers = {"User-Agent": _get_user_agent()} + + response = self._client.get(url, params=params, headers=req_headers) + + if response.status_code in (502, 503): + import time + + time.sleep(1) + response = self._client.get(url, params=params, headers=req_headers) + + if response.status_code == 402: + # Account rail: a 402 is the account being out of credit, not a + # challenge to sign. Nothing here can sign, so say so plainly. + raise_for_api_key_402(response, self.api_key) + result = self._handle_get_payment_and_retry(url, params, response) + save_to_cache( + endpoint, + cache_key_body, + result, + cost_usd=self._last_call_cost, + **self._billing_meta(), + ) + self._log_transaction(endpoint, cache_key_body, result, self._last_call_cost) + return result + + if response.status_code != 200: + try: + error_body = response.json() + except Exception: + error_body = {"error": "Request failed"} + raise APIError( + f"API error: {response.status_code}", + response.status_code, + sanitize_error_response(error_body), + retry_after=retry_after_of(response), + ) + + return response.json() + + def _handle_get_payment_and_retry( + self, + url: str, + params: dict[str, Any] | None, + response: httpx.Response, + ) -> dict[str, Any]: + """Handle 402 response for GET endpoints: parse requirements, sign payment, retry with GET.""" + payment_header = response.headers.get("payment-required") + price_info = {} + if not payment_header: + try: + resp_body = response.json() + if "x402" in resp_body: + payment_header = resp_body + price_info = resp_body.get("price", {}) + except Exception: + pass + + if not payment_header: + raise PaymentError("402 response but no payment requirements found") + + if isinstance(payment_header, str): + payment_required = parse_payment_required(payment_header) + else: + payment_required = payment_header + + details = extract_payment_details(payment_required) + + cost_usd = ( + float(price_info.get("amount", 0)) + if price_info + else float(details.get("amount", 0)) / 1e6 + ) + # Before signing: a refused quote is never sent, so nothing settles. + _enforce_spend_limits(self, cost_usd) + + resource = details.get("resource") or {} + extensions = payment_required.get("extensions", {}) + payment_payload = create_payment_payload( + account=self.account, + recipient=details["recipient"], + amount=details["amount"], + network=details.get("network", "eip155:84532" if self.is_testnet() else "eip155:8453"), + resource_url=validate_resource_url(resource.get("url", url), self.api_url), + resource_description=resource.get("description", "BlockRun AI API call"), + max_timeout_seconds=details.get("maxTimeoutSeconds", 300), + extra=details.get("extra"), + extensions=extensions, + asset=details.get("asset"), + ) + + payment_headers = { + "User-Agent": _get_user_agent(), + "PAYMENT-SIGNATURE": payment_payload, + } + + retry_response = self._client.get( + url, params=params, headers=payment_headers, timeout=self.timeout + ) + if retry_response.status_code in (502, 503): + import time + + time.sleep(1) + retry_response = self._client.get( + url, params=params, headers=payment_headers, timeout=self.timeout + ) + + if retry_response.status_code == 402: + # Account rail: a 402 is the account being out of credit, not a + # challenge to sign. Nothing here can sign, so say so plainly. + raise_for_api_key_402(retry_response, self.api_key) + raise PaymentError("Payment was rejected. Check your wallet balance.") + + if retry_response.status_code != 200: + try: + error_body = retry_response.json() + except Exception: + error_body = {"error": "Request failed"} + raise APIError( + f"{paid_request_error_prefix(retry_response.headers)}: {retry_response.status_code}", + retry_response.status_code, + sanitize_error_response(error_body), + retry_after=retry_after_of(retry_response), + ) + + self._session_calls += 1 + self._session_total_usd += cost_usd + self._last_call_cost = cost_usd + self._capture_settlement(retry_response) + + return retry_response.json() + + def image_edit( + self, + prompt: str, + image: str | list[str], + *, + model: str = "openai/gpt-image-2", + mask: str | None = None, + size: str = "1024x1024", + n: int = 1, + ) -> ImageResponse: + """ + Edit an image using img2img, or fuse multiple source images. + + Args: + prompt: Text description of the desired edit + image: A single base64 "data:image/...;base64,..." data URI, or a + list of 1-4 such data URIs to fuse multiple sources. Plain + URLs are not accepted — the source must be a data URI. + model: Model ID (default: "openai/gpt-image-2") + Edit-supported: "openai/gpt-image-1", "openai/gpt-image-2", + "google/nano-banana", "google/nano-banana-pro". + Multi-image caps: openai/* up to 4, google/* up to 3. + mask: Optional base64-encoded mask image (OpenAI gpt-image-* only; + cannot be combined with multiple source images). + size: Output image size (default: "1024x1024") + n: Number of images to generate (default: 1) + + Returns: + ImageResponse with edited image URLs + """ + body: dict[str, Any] = { + "model": model, + "prompt": prompt, + "image": image, + "size": size, + "n": n, + } + if mask is not None: + body["mask"] = mask + + data = self._request_with_payment_raw("/v1/images/image2image", body) + return ImageResponse(**data) + + def search( + self, + query: str, + *, + sources: list[str] | None = None, + max_results: int = 10, + from_date: str | None = None, + to_date: str | None = None, + ) -> SearchResult: + """ + Standalone search (web, X/Twitter, news). + + Args: + query: Search query + sources: Source types to search (e.g. ["web", "x", "news"]) + max_results: Maximum number of results (default: 10) + from_date: Start date filter (YYYY-MM-DD) + to_date: End date filter (YYYY-MM-DD) + + Returns: + SearchResult with summary and citations + """ + body: dict[str, Any] = { + "query": query, + "max_results": max_results, + } + if sources is not None: + body["sources"] = sources + if from_date is not None: + body["from_date"] = from_date + if to_date is not None: + body["to_date"] = to_date + + data = self._request_with_payment_raw("/v1/search", body) + return SearchResult(**data) + + # ── Exa Web Search (Powered by Exa) ───────────────────────────────────── + + def exa(self, path: str, body: dict[str, Any]) -> dict[str, Any]: + """Generic Exa endpoint proxy via x402 USDC on Base. + + Args: + path: Exa endpoint — one of: "search", "find-similar", "contents", "answer" + body: Request body (see https://docs.exa.ai) + + Example:: + + result = client.exa("search", {"query": "latest AI research", "numResults": 5}) + """ + return self._request_with_payment_raw(f"/v1/exa/{path}", body) + + def exa_search(self, query: str, **kwargs: Any) -> dict[str, Any]: + """Neural and keyword web search via Exa ($0.01/request, Base USDC). + + Args: + query: Search query string + **kwargs: Additional Exa parameters (numResults, category, useAutoprompt, etc.) + + Example:: + + results = client.exa_search("latest AI papers", numResults=5) + """ + return self._request_with_payment_raw("/v1/exa/search", {"query": query, **kwargs}) + + def exa_find_similar(self, url: str, **kwargs: Any) -> dict[str, Any]: + """Find pages semantically similar to a given URL via Exa + ($0.01/request, Base USDC). + + Args: + url: URL to find similar pages for + **kwargs: Additional Exa parameters (numResults, etc.) + + Example:: + + similar = client.exa_find_similar("https://openai.com/research/gpt-4", numResults=5) + """ + return self._request_with_payment_raw("/v1/exa/find-similar", {"url": url, **kwargs}) + + def exa_contents(self, urls: list[str], **kwargs: Any) -> dict[str, Any]: + """Extract full text content from URLs via Exa ($0.002/URL, Base USDC). + + Args: + urls: List of URLs to extract content from + **kwargs: Additional Exa parameters (text, highlights, summary, etc.) + + Example:: + + data = client.exa_contents(["https://arxiv.org/abs/2303.08774"]) + """ + return self._request_with_payment_raw("/v1/exa/contents", {"urls": urls, **kwargs}) + + def exa_answer(self, query: str, **kwargs: Any) -> dict[str, Any]: + """AI-generated answer grounded in live web search via Exa + ($0.01/request, Base USDC). + + Args: + query: Question to answer + **kwargs: Additional Exa parameters + + Example:: + + answer = client.exa_answer("What is the current state of AI safety research?") + """ + return self._request_with_payment_raw("/v1/exa/answer", {"query": query, **kwargs}) + + # ── Prediction Markets (Powered by Predexon) ──────────────────────────── + + def pm(self, path: str, **params: Any) -> dict[str, Any]: + """ + Query Predexon prediction market data (GET endpoints). + + Access real-time data across Polymarket, Kalshi, Limitless, Opinion, + Predict.Fun, dFlow, sports, and Binance Futures. Powered by Predexon v2. + Tier 1 = $0.001/call, Tier 2 = $0.005/call. + + Args: + path: Endpoint path, e.g. "polymarket/events", "kalshi/markets/12345" + **params: Query parameters passed to the endpoint + + Returns: + Raw response dict from Predexon API + + Example: + events = client.pm("polymarket/events") + market = client.pm("kalshi/markets/KXBTC-25MAR14") + results = client.pm("polymarket/search", q="bitcoin") + # v2 canonical cross-venue + markets = client.pm("markets", venue="polymarket", status="active") + # v2 sports + games = client.pm("sports/markets", league="NBA") + # v2 wallet identity + ident = client.pm("polymarket/wallet/identity/0xabc...") + """ + return self._get_with_payment_raw(f"/v1/pm/{path}", params or None) + + def pm_query(self, path: str, query: dict[str, Any]) -> dict[str, Any]: + """ + Structured query for Predexon prediction market data (POST endpoints). + + For endpoints that require a JSON body, e.g. bulk wallet identity lookup. + Tier 1 = $0.001/call, Tier 2 = $0.005/call. + + Args: + path: Endpoint path, e.g. "polymarket/wallet/identities" + query: JSON body for the structured query + + Returns: + Raw response dict from Predexon API + + Example: + # v2 bulk wallet identity (up to 200 addresses) + batch = client.pm_query("polymarket/wallet/identities", { + "addresses": ["0xabc...", "0xdef..."], + }) + """ + return self._request_with_payment_raw(f"/v1/pm/{path}", query) + + # ── PM convenience helpers (Predexon v2) ──────────────────────────────── + # Thin wrappers over pm() / pm_query() for the most common v2 endpoints. + # All accept arbitrary keyword filters that are forwarded as query params. + + def pm_markets(self, **params: Any) -> dict[str, Any]: + """RETIRED — ``/v1/pm/markets`` no longer exists. + + Predexon sunset market matching on 2026-07-20 and the whole + canonical layer went with it, so this path returns 410 upstream. + Use ``pm("markets/search", q=...)`` for cross-venue lookups. + + Kept as a raising stub rather than deleted so upgrading does not + break imports or attribute access; calling it fails immediately + instead of after a paid round trip. + + :raises RetiredEndpointError: always. + """ + raise RetiredEndpointError( + "/v1/pm/markets was sunset by Predexon on 2026-07-20 (upstream 410). Use pm('markets/search', q=...) for cross-venue lookups." + ) + + def pm_listings(self, **params: Any) -> dict[str, Any]: + """RETIRED — ``/v1/pm/markets/listings`` no longer exists. + + Predexon sunset market matching on 2026-07-20 and the whole + canonical layer went with it, so this path returns 410 upstream. + Use ``pm("markets/search", q=...)`` for cross-venue lookups. + + Kept as a raising stub rather than deleted so upgrading does not + break imports or attribute access; calling it fails immediately + instead of after a paid round trip. + + :raises RetiredEndpointError: always. + """ + raise RetiredEndpointError( + "/v1/pm/markets/listings was sunset by Predexon on 2026-07-20 (upstream 410). Use pm('markets/search', q=...) for cross-venue lookups." + ) + + def pm_outcome(self, predexon_id: str) -> dict[str, Any]: + """RETIRED — ``/v1/pm/outcomes/{predexon_id}`` no longer exists. + + Predexon sunset market matching on 2026-07-20 and the whole + canonical layer went with it, so this path returns 410 upstream. + Use ``pm("markets/search", q=...)`` for cross-venue lookups. + + Kept as a raising stub rather than deleted so upgrading does not + break imports or attribute access; calling it fails immediately + instead of after a paid round trip. + + :raises RetiredEndpointError: always. + """ + raise RetiredEndpointError( + "/v1/pm/outcomes/{predexon_id} was sunset by Predexon on 2026-07-20 (upstream 410). Use pm('markets/search', q=...) for cross-venue lookups." + ) + + def pm_polymarket_markets(self, **params: Any) -> dict[str, Any]: + """List Polymarket markets (Predexon v2). Tier 1 ($0.001/call). + + For high-volume traversal use ``pm_polymarket_markets_keyset()``. + """ + return self.pm("polymarket/markets", **params) + + def pm_polymarket_events(self, **params: Any) -> dict[str, Any]: + """List Polymarket events (Predexon v2). Tier 1 ($0.001/call). + + For high-volume traversal use ``pm_polymarket_events_keyset()``. + """ + return self.pm("polymarket/events", **params) + + def pm_polymarket_markets_keyset(self, **params: Any) -> dict[str, Any]: + """Polymarket markets with cursor-based keyset pagination + (use pagination_key=). Tier 1 ($0.001/call).""" + return self.pm("polymarket/markets/keyset", **params) + + def pm_polymarket_events_keyset(self, **params: Any) -> dict[str, Any]: + """Polymarket events with cursor-based keyset pagination + (use pagination_key=). Tier 1 ($0.001/call).""" + return self.pm("polymarket/events/keyset", **params) + + def pm_polymarket_positions(self, **params: Any) -> dict[str, Any]: + """Polymarket open positions (per-wallet, market-level PnL). + Tier 1 ($0.001/call).""" + return self.pm("polymarket/positions", **params) + + def pm_polymarket_trades(self, **params: Any) -> dict[str, Any]: + """Recent Polymarket trades (token, side, shares, price, tx_hash). + Tier 1 ($0.001/call).""" + return self.pm("polymarket/trades", **params) + + def pm_polymarket_leaderboard(self, **params: Any) -> dict[str, Any]: + """Polymarket trader leaderboard (rank by window, sort_by). + Tier 1 ($0.001/call).""" + return self.pm("polymarket/leaderboard", **params) + + def pm_kalshi_markets(self, **params: Any) -> dict[str, Any]: + """List Kalshi markets (CFTC-regulated event contracts). + Tier 1 ($0.001/call).""" + return self.pm("kalshi/markets", **params) + + def pm_limitless_markets(self, **params: Any) -> dict[str, Any]: + """List Limitless markets (binary AMM-style outcomes). + Tier 1 ($0.001/call).""" + return self.pm("limitless/markets", **params) + + def pm_sports_categories(self) -> dict[str, Any]: + """List available sports categories. Tier 1 ($0.001/call). + + .. warning:: + Upstream is returning 500 for every ``sports/*`` path as of + 2026-08-04. The route still resolves, so this keeps working the + moment Predexon restores it, but do not build on it yet. + """ + return self.pm("sports/categories") + + def pm_sports_markets(self, **params: Any) -> dict[str, Any]: + """List sports markets grouped by game. Filter with league=, + sport_type=, status=, venue=. Tier 1 ($0.001/call). + + .. warning:: + Upstream is returning 500 for every ``sports/*`` path as of + 2026-08-04. The route still resolves, so this keeps working the + moment Predexon restores it, but do not build on it yet. + """ + return self.pm("sports/markets", **params) + + def pm_wallet_identity(self, wallet: str) -> dict[str, Any]: + """Fetch identity + profile metadata for one wallet (ENS, Twitter, + portfolio, etc.). Tier 2 ($0.005/call).""" + return self.pm(f"polymarket/wallet/identity/{wallet}") + + def pm_wallet_identities(self, addresses: list[str]) -> dict[str, Any]: + """Bulk identity lookup for up to 200 wallet addresses (POST). + Tier 2 ($0.005/call).""" + return self.pm_query("polymarket/wallet/identities", {"addresses": addresses}) + + def pm_wallet_cluster(self, address: str) -> dict[str, Any]: + """Discover wallets connected to a seed address via on-chain transfers + and identity proofs. Tier 2 ($0.005/call).""" + return self.pm(f"polymarket/wallet/{address}/cluster") + + # ── DefiLlama (DeFi protocols / TVL / yields / prices) ────────────────── + + def defi(self, path: str, **params: Any) -> dict[str, Any]: + """ + Query DefiLlama DeFi data (GET passthrough). Powered by DefiLlama. + + $0.005/call for protocols / protocol/{slug} / chains / yields; + $0.001/call for prices/{coins}. + + Args: + path: Endpoint path — "protocols", "protocol/{slug}", "chains", + "yields", or "prices/{coins}" (coins comma-separated, e.g. + "coingecko:bitcoin,base:0x..."). + **params: Query parameters passed through to DefiLlama. + + Example:: + + protocols = client.defi("protocols") + aave = client.defi("protocol/aave") + """ + return self._get_with_payment_raw(f"/v1/defillama/{path}", params or None) + + def defi_protocols(self) -> dict[str, Any]: + """All DeFi protocols with TVL ($0.005/call).""" + return self.defi("protocols") + + def defi_protocol(self, slug: str) -> dict[str, Any]: + """Single protocol details + historical TVL ($0.005/call).""" + return self.defi(f"protocol/{slug}") + + def defi_chains(self) -> dict[str, Any]: + """Current TVL of every chain ($0.005/call).""" + return self.defi("chains") + + def defi_yields(self, **params: Any) -> dict[str, Any]: + """Yield pools with APY/TVL ($0.005/call).""" + return self.defi("yields", **params) + + def defi_prices(self, coins: list[str] | str) -> dict[str, Any]: + """Token price lookup ($0.001/call). + + Args: + coins: Coin ids like "coingecko:bitcoin" or "{chain}:{address}" — + a list or a pre-joined comma-separated string. + """ + joined = ",".join(coins) if isinstance(coins, list) else coins + return self.defi(f"prices/{joined}") + + # ── 0x DEX (swap quotes + gasless) — free passthrough ─────────────────── + + def dex( + self, + path: str, + *, + method: str = "GET", + body: dict[str, Any] | None = None, + **params: Any, + ) -> dict[str, Any]: + """ + Query the 0x Swap / Gasless APIs (free — no x402 payment; BlockRun + takes an on-chain affiliate fee on executed swaps instead). + + Args: + path: Endpoint path — "price", "quote", "gasless/price", + "gasless/quote", "gasless/submit" (POST), "gasless/status/{hash}", + "gasless/approval-tokens", "gasless/chains", "swap/chains". + method: "GET" (default) or "POST" (gasless/submit only). + body: JSON body for POST endpoints. + **params: Query parameters (chainId, sellToken, buyToken, + sellAmount, taker, ...). + + Example:: + + quote = client.dex("quote", chainId=8453, + sellToken="0x...", buyToken="0x...", + sellAmount="1000000", taker="0x...") + """ + endpoint = f"/v1/zerox/{path}" + if method.upper() == "POST": + return self._request_with_payment_raw(endpoint, body or {}) + return self._get_with_payment_raw(endpoint, params or None) + + def dex_price(self, **params: Any) -> dict[str, Any]: + """Indicative Permit2 swap price — no commitment (free).""" + return self.dex("price", **params) + + def dex_quote(self, **params: Any) -> dict[str, Any]: + """Firm Permit2 swap quote with permit2.eip712 + tx data (free).""" + return self.dex("quote", **params) + + def dex_gasless_price(self, **params: Any) -> dict[str, Any]: + """Gasless indicative price quote (free).""" + return self.dex("gasless/price", **params) + + def dex_gasless_quote(self, **params: Any) -> dict[str, Any]: + """Gasless firm quote — returns trade.eip712 to sign (free).""" + return self.dex("gasless/quote", **params) + + def dex_gasless_submit(self, body: dict[str, Any]) -> dict[str, Any]: + """Submit a signed gasless trade; the 0x relayer pays gas (free).""" + return self.dex("gasless/submit", method="POST", body=body) + + def dex_gasless_status(self, trade_hash: str) -> dict[str, Any]: + """Poll a gasless trade's status by tradeHash (free).""" + return self.dex(f"gasless/status/{trade_hash}") + + def dex_chains(self) -> dict[str, Any]: + """Chains where the Swap API is supported (free).""" + return self.dex("swap/chains") + + def dex_gasless_chains(self) -> dict[str, Any]: + """Chains where the Gasless API is supported (free).""" + return self.dex("gasless/chains") + + # ── Modal Sandbox (pay-per-call cloud compute) ─────────────────────────── + + def modal(self, path: str, body: dict[str, Any] | None = None) -> dict[str, Any]: + """ + Call the Modal sandbox compute API (POST passthrough). + + Args: + path: "sandbox/create" ($0.01 CPU / $0.05 GPU), "sandbox/exec" + ($0.001), "sandbox/status" ($0.001), "sandbox/terminate" ($0.001). + body: JSON body for the endpoint. + """ + return self._request_with_payment_raw(f"/v1/modal/{path}", body or {}) + + def modal_sandbox_create(self, **body: Any) -> dict[str, Any]: + """Create a sandboxed compute environment ($0.01 CPU / $0.05 GPU). + + Common fields: image ("python:3.11"), gpu (optional GPU type), + timeout. Returns a sandbox_id for exec/status/terminate. + """ + return self.modal("sandbox/create", body) + + def modal_sandbox_exec( + self, sandbox_id: str, command: list[str], **body: Any + ) -> dict[str, Any]: + """Execute a command in a sandbox; returns stdout/stderr ($0.001).""" + return self.modal("sandbox/exec", {"sandbox_id": sandbox_id, "command": command, **body}) + + def modal_sandbox_status(self, sandbox_id: str) -> dict[str, Any]: + """Check a sandbox's status ($0.001).""" + return self.modal("sandbox/status", {"sandbox_id": sandbox_id}) + + def modal_sandbox_terminate(self, sandbox_id: str) -> dict[str, Any]: + """Terminate a sandbox ($0.001).""" + return self.modal("sandbox/terminate", {"sandbox_id": sandbox_id}) + + # ── Coinbase Onramp ────────────────────────────────────────────────────── + + def onramp(self, address: str) -> dict[str, Any]: + # Onramp funds a wallet, and an API-key account has none: credit is + # bought with a card at user.blockrun.ai, not minted into an address. + if self.api_key: + raise wallet_only("onramp") + """Mint a one-time Coinbase Onramp link to fund a wallet with fiat (FREE). + + Opens the door to buying Base USDC with a card or bank (60+ fiat + currencies) via pay.coinbase.com. FREE — the x402 signature only + authenticates the wallet, so the funding ``address`` MUST equal the + signing wallet (use ``client.get_wallet_address()``). Base / USDC only. + + The returned URL is single-use and expires in ~5 minutes, so mint it at + click time and never cache it. + + Args: + address: Destination wallet (0x-prefixed Base address). Must match + the signing wallet, since the link funds that exact address. + + Returns: + Dict with a ``url`` pointing at ``https://pay.coinbase.com/``. + + Example:: + + link = client.onramp(client.get_wallet_address()) + print(link["url"]) # open in a browser to buy USDC on Base + """ + validate_eth_address(address) + data = self._request_with_payment_raw( + "/v1/onramp/token", + {"address": address, "network": "base", "asset": "USDC"}, + ) + url = data.get("url") if isinstance(data, dict) else None + if not isinstance(url, str) or not url.startswith("https://pay.coinbase.com/"): + raise APIError("gateway returned no onramp url", 0, None) + return data + + def list_models(self) -> list[dict[str, Any]]: + """ + List available LLM models with pricing. + + Returns: + List of model information dicts + """ + response = self._client.get(f"{self.api_url}/v1/models") + + if response.status_code != 200: + try: + error_body = response.json() + except Exception: + error_body = {"error": "Request failed"} + raise APIError( + f"Failed to list models: {response.status_code}", + response.status_code, + sanitize_error_response(error_body), + retry_after=retry_after_of(response), + ) + + return response.json().get("data", []) + + def list_image_models(self) -> list[dict[str, Any]]: + """ + List available image generation models with pricing. + + Returns: + List of image model information dicts (id, name, pricing, etc.) + + Notes: + The dedicated ``/v1/images/models`` endpoint was deprecated + server-side; the catalog now lives in ``/v1/models`` with + ``categories: ["image", ...]``. This method filters the unified + catalog so existing callers keep working. + """ + return [m for m in self.list_models() if "image" in (m.get("categories") or [])] + + def list_all_models(self) -> list[dict[str, Any]]: + """ + List all available models (chat, image, music, etc.) with pricing. + + Returns: + List of all model information dicts with a ``type`` field set to + the first category (``llm`` for chat, ``image`` / ``music`` / + ``audio`` etc. for media). Backwards-compat: chat models always + report ``type: "llm"``. + """ + all_models = self.list_models() + for m in all_models: + cats = m.get("categories") or [] + if "chat" in cats: + m["type"] = "llm" + elif "image" in cats: + m["type"] = "image" + elif "music" in cats or "audio" in cats: + m["type"] = "music" + else: + m["type"] = cats[0] if cats else "llm" + return all_models + + @property + def payment_mode(self) -> str: + """Which rail this client pays on: ``"apikey"`` or ``"wallet"``. + + Worth checking once at startup when both a key and a wallet are + configured in the environment: it is the difference between + spending credit and spending USDC.""" + return payment_mode(self) + + def get_wallet_address(self) -> str: + # No address on the account rail: payment comes from prepaid + # credit, so there is nothing to return but the empty string. + if self.api_key: + return "" + """Get the wallet address being used for payments.""" + return self.account.address + + def is_testnet(self) -> bool: + """Check if client is configured for testnet.""" + return "testnet.blockrun.ai" in self.api_url + + def _billing_meta(self) -> dict[str, str | None]: + """Return billing metadata (wallet / network / client_kind) for the + cost log. Used by ``save_to_cache`` call sites.""" + return { + "wallet": self.account.address, + "network": _detect_network(self.api_url), + "client_kind": type(self).__name__, + } + + def _log_transaction( + self, + endpoint: str, + body: dict[str, Any], + response: Any, + cost_usd: float, + ) -> None: + """Append one row to the project-local transaction log, if enabled. + + Pulls the on-chain settlement out of ``self._last_settlement`` + (captured from ``PAYMENT-RESPONSE`` on the paid retry) and + consumes it — so a subsequent free / cached call right after a + paid one cannot reuse stale tx fields. No-op when the logger is + disabled; never raises (best-effort logging by design).""" + logger = self._tx_logger + if logger is None: + return + settlement = self._last_settlement + self._last_settlement = None + try: + logger.log( + endpoint=endpoint, + request=body, + response=response, + cost_usd=cost_usd, + model=(body.get("model") if isinstance(body, dict) else None), + wallet=self.account.address, + network=_detect_network(self.api_url), + client_kind=type(self).__name__, + settlement=settlement, + ) + except Exception: + pass + + def get_balance(self) -> float: + # Returning 0 would be the worst available answer: it is + # indistinguishable from an empty wallet, and an agent gating on it + # would stop calling a well-funded account. + if self.api_key: + raise wallet_only("get_balance") + """ + Get USDC balance on Base network. + + Automatically detects mainnet vs testnet based on API URL: + - Mainnet: Base (Chain ID 8453) + - Testnet: Base Sepolia (Chain ID 84532) + + Returns: + float: USDC balance (6 decimal places normalized) + + Example: + balance = client.get_balance() + print(f"Balance: ${balance:.2f} USDC") + """ + # USDC contracts + # Mainnet: Base + # Testnet: Base Sepolia + if self.is_testnet(): + usdc_contract = "0x036CbD53842c5426634e7929541eC2318f3dCF7e" + rpcs = [ + "https://sepolia.base.org", + "https://base-sepolia-rpc.publicnode.com", + ] + else: + usdc_contract = "0x833589fCD6eDb6E08f4c7C32D4f71b54bdA02913" + rpcs = [ + "https://base.publicnode.com", + "https://mainnet.base.org", + "https://base.meowrpc.com", + ] + + # balanceOf(address) function selector + selector = "0x70a08231" + # Pad wallet address to 32 bytes + padded_address = self.account.address[2:].lower().zfill(64) + data = selector + padded_address + + payload = { + "jsonrpc": "2.0", + "method": "eth_call", + "params": [{"to": usdc_contract, "data": data}, "latest"], + "id": 1, + } + + last_error = None + for rpc in rpcs: + try: + response = httpx.post(rpc, json=payload, timeout=10) + result = response.json().get("result", "0x0") + # Convert from hex and normalize (USDC has 6 decimals) + balance_raw = int(result, 16) + return balance_raw / 1_000_000 + except Exception as e: + last_error = e + continue + + # If all RPCs failed, raise the last error + raise last_error or Exception("All RPCs failed") + + def close(self): + """Close the HTTP client.""" + self._client.close() + + def __enter__(self): + return self + + def __exit__(self, exc_type, exc_val, exc_tb): + self.close() + + +# Async client for async/await usage +class AsyncLLMClient: + """ + Async version of BlockRun LLM Client. + + Usage: + async with AsyncLLMClient() as client: + response = await client.chat("gpt-5.2", "Hello!") + + # For testnet: + async with AsyncLLMClient(api_url="https://testnet.blockrun.ai/api") as client: + response = await client.chat("openai/gpt-oss-20b", "Hello!") + """ + + DEFAULT_API_URL = "https://blockrun.ai/api" + TESTNET_API_URL = "https://testnet.blockrun.ai/api" + DEFAULT_MAX_TOKENS = 1024 + + def __init__( + self, + private_key: str | None = None, + api_url: str | None = None, + timeout: float = DEFAULT_CHAT_TIMEOUT, + search_timeout: float = 300.0, + transaction_log: bool | str | os.PathLike[str] | None = None, + max_cost_per_call: float | None = None, + max_session_cost: float | None = None, + ): + """ + Initialize the async BlockRun LLM client. + + Args: + private_key: Base chain wallet private key (or set BLOCKRUN_WALLET_KEY env var) + api_url: API endpoint URL (default: https://blockrun.ai/api) + timeout: Request timeout in seconds (default: 600, override via BLOCKRUN_CHAT_TIMEOUT env). Used for regular chat requests. + search_timeout: Timeout for xAI Live Search requests (default: 300 = 5 minutes). + Auto-detected when search_parameters or search=True is passed. + transaction_log: Same opt-in per-call log as ``LLMClient``. ``True`` → + ``./log/``; pass a string/Path for a custom dir; ``None`` + honors the ``BLOCKRUN_TX_LOG`` env var. See ``LLMClient`` + for the full record schema. + + Raises: + ValueError: If no wallet is configured + """ + from .wallet import load_wallet + + # Account rail first, and before any wallet variable is read: an API + # key is a complete credential on its own, so demanding a private key + # alongside it would make every API-key user invent a wallet they + # never use. See apikey.py for the precedence rule. + api_key = resolve_api_key(private_key) + key = ( + None + if api_key + else ( + private_key + or os.environ.get("BLOCKRUN_WALLET_KEY") + or os.environ.get("BASE_CHAIN_WALLET_KEY") + or load_wallet() # Loads from ~/.blockrun/.session + ) + ) + if not api_key and not key: + raise missing_credential_error() + + # Normalize private key format (add 0x prefix if missing) + if key and not key.startswith("0x"): + key = "0x" + key + + # Validate private key format + if key: + validate_private_key(key) + + self.api_key = api_key + # No wallet on the account rail: nothing is signed locally. + self.account = Account.from_key(key) if key else None + + # Validate and set API URL + # BLOCKRUN_API_URL names an x402 gateway; an API-key client must not + # follow it and hand the key to a host set up for another rail. + api_url_raw = ( + api_key_base_url(api_url) + if api_key + else (api_url or os.environ.get("BLOCKRUN_API_URL") or self.DEFAULT_API_URL) + ) + validate_api_url(api_url_raw) + self.api_url = api_url_raw.rstrip("/") + + self.timeout = timeout + self.search_timeout = search_timeout + # Default httpx pool (max_connections=100) is exhausted by ~50 concurrent + # paid requests because each request uses two HTTP connections: Phase 1 + # (402 probe) + Phase 2 (authenticated SSE stream). Raise the limit so + # high-concurrency deployments don't hit pool exhaustion before hitting + # any upstream rate limit. + self._client = httpx.AsyncClient( + headers=auth_headers(api_key), + timeout=timeout, + limits=httpx.Limits(max_connections=200, max_keepalive_connections=50), + ) + self._last_call_cost: float = 0.0 + # This client tracks no session total (see chat_completion), so the + # session limit has nothing to accumulate against; the per-call limit + # still applies. Kept as an attribute so the shared check is uniform. + self._session_total_usd: float = 0.0 + # Opt-in spend limits. None (the default) means unlimited, which is the + # behavior every release before 1.9.0 had: every 402 quote was signed + # automatically with nothing compared against anything. + self._max_cost_per_call = resolve_spend_limit( + max_cost_per_call, "BLOCKRUN_MAX_COST_PER_CALL" + ) + self._max_session_cost = resolve_spend_limit(max_session_cost, "BLOCKRUN_MAX_SESSION_COST") + + log_dir = _resolve_log_dir(transaction_log) + self._tx_logger: TransactionLogger | None = ( + TransactionLogger(log_dir) if log_dir is not None else None + ) + # Model pricing cache for smart routing + self._model_pricing_cache: dict[str, dict[str, float]] | None = None + self._last_settlement: dict[str, Any] | None = None + + def _capture_settlement(self, response: httpx.Response) -> dict[str, Any] | None: + """Async-client twin of :meth:`LLMClient._capture_settlement`.""" + header = read_settlement_header(response.headers) + settlement = decode_settlement_header(header) + self._last_settlement = settlement + return settlement + + async def _get_model_pricing(self) -> dict[str, dict[str, float]]: + """Model pricing for smart routing (cached for the client's lifetime).""" + if self._model_pricing_cache is not None: + return self._model_pricing_cache + pricing = build_model_pricing(await self.list_models()) + self._model_pricing_cache = pricing + return pricing + + async def route( + self, + prompt: str, + *, + system: str | None = None, + max_tokens: int | None = None, + routing_profile: RoutingProfile = "auto", + requires_structured_output: bool = False, + ) -> RoutingDecision: + """Inspect a routing decision without making or paying for a call.""" + decision = route_with_catalog( + prompt, + system, + max_tokens or self.DEFAULT_MAX_TOKENS, + await self._get_model_pricing(), + routing_profile=routing_profile, + requires_structured_output=requires_structured_output, + minimum_payment_usd=BASE_MINIMUM_PAYMENT_USD, + ) + return RoutingDecision(**decision) + + async def smart_chat( + self, + prompt: str, + *, + system: str | None = None, + max_tokens: int | None = None, + temperature: float | None = None, + routing_profile: RoutingProfile = "auto", + ) -> SmartChatResponse: + """Async smart chat with automatic model routing. + + Same Router Core portfolio strategy as the sync client — see + :meth:`LLMClient.smart_chat`. + """ + decision = route_with_catalog( + prompt, + system, + max_tokens or self.DEFAULT_MAX_TOKENS, + await self._get_model_pricing(), + routing_profile=routing_profile, + minimum_payment_usd=BASE_MINIMUM_PAYMENT_USD, + ) + response = await self.chat( + decision["model"], + prompt, + system=system, + max_tokens=max_tokens, + temperature=temperature, + fallback_models=decision.get("fallbacks") or None, + ) + return SmartChatResponse( + response=response, + model=decision["model"], + routing=RoutingDecision(**decision), + ) + + async def smart_chat_completion( + self, + messages: list[dict[str, Any]], + *, + max_tokens: int | None = None, + temperature: float | None = None, + top_p: float | None = None, + search: bool | None = None, + search_parameters: dict[str, Any] | None = None, + tools: list[dict[str, Any]] | None = None, + tool_choice: Any | None = None, + response_format: dict[str, Any] | None = None, + stop: str | list[str] | None = None, + fallback_models: list[str] | None = None, + routing_profile: RoutingProfile = "auto", + **extra: Any, + ) -> SmartChatCompletionResponse: + """Async smart routing for a full message list — see + :meth:`LLMClient.smart_chat_completion`.""" + view = routing_text(messages) + decision = route_with_catalog( + view["prompt"], + view["system_prompt"], + max_tokens or self.DEFAULT_MAX_TOKENS, + await self._get_model_pricing(), + routing_profile=routing_profile, + requires_structured_output=response_format is not None, + tools=tools, + tool_choice=tool_choice, + conversation_chars=view["conversation_chars"], + has_vision=view["has_vision"], + minimum_payment_usd=BASE_MINIMUM_PAYMENT_USD, + ) + response = await self.chat_completion( + decision["model"], + messages, + max_tokens=max_tokens, + temperature=temperature, + top_p=top_p, + search=search, + search_parameters=search_parameters, + tools=tools, + tool_choice=tool_choice, + response_format=response_format, + stop=stop, + fallback_models=fallback_models or decision.get("fallbacks") or None, + **extra, + ) + return SmartChatCompletionResponse( + response=response, + model=decision["model"], + routing=RoutingDecision(**decision), + ) + + async def chat( + self, + model: str, + prompt: str, + *, + system: str | None = None, + max_tokens: int | None = None, + temperature: float | None = None, + search: bool | None = None, + search_parameters: dict[str, Any] | None = None, + response_format: dict[str, Any] | None = None, + stop: str | list[str] | None = None, + fallback_models: list[str] | None = None, + **extra: Any, + ) -> str: + """Async 1-line chat interface with optional xAI Live Search.""" + messages: list[dict[str, str]] = [] + + if system: + messages.append({"role": "system", "content": system}) + + messages.append({"role": "user", "content": prompt}) + + result = await self.chat_completion( + model=model, + messages=messages, + max_tokens=max_tokens, + temperature=temperature, + search=search, + search_parameters=search_parameters, + response_format=response_format, + stop=stop, + fallback_models=fallback_models, + **extra, + ) + + return result.choices[0].message.content + + async def chat_completion( + self, + model: str, + messages: list[dict[str, Any]], + *, + max_tokens: int | None = None, + temperature: float | None = None, + top_p: float | None = None, + search: bool | None = None, + search_parameters: dict[str, Any] | None = None, + tools: list[dict[str, Any]] | None = None, + tool_choice: Any | None = None, + response_format: dict[str, Any] | None = None, + stop: str | list[str] | None = None, + fallback_models: list[str] | None = None, + **extra: Any, + ) -> ChatResponse: + """Async full chat completion interface with optional xAI Live Search and tool calling. + + ``blockrun/auto`` | ``blockrun/eco`` | ``blockrun/premium`` are routing + profiles rather than models: passing one routes the turn and returns the + routed response, ranked fallback chain included. + """ + virtual_profile = routing_profile_for_model(model) + if virtual_profile is not None: + return ( + await self.smart_chat_completion( + messages, + max_tokens=max_tokens, + temperature=temperature, + top_p=top_p, + search=search, + search_parameters=search_parameters, + tools=tools, + tool_choice=tool_choice, + response_format=response_format, + stop=stop, + fallback_models=fallback_models, + routing_profile=virtual_profile, # type: ignore[arg-type] + **extra, + ) + ).response + + # Validate inputs + validate_model(model) + validate_max_tokens(max_tokens) + validate_temperature(temperature) + validate_top_p(top_p) + + body: dict[str, Any] = { + "model": model, + "messages": messages, + "max_tokens": max_tokens or self.DEFAULT_MAX_TOKENS, + } + + if temperature is not None: + body["temperature"] = temperature + if top_p is not None: + body["top_p"] = top_p + + # Handle xAI Live Search parameters + if search_parameters is not None: + body["search_parameters"] = search_parameters + elif search is True: + # Simple shortcut: search=True enables live search with defaults + body["search_parameters"] = {"mode": "on"} + + # Handle tool calling + if tools is not None: + body["tools"] = tools + if tool_choice is not None: + body["tool_choice"] = tool_choice + + # OpenAI-compatible response shaping (honored by the gateway across providers) + if response_format is not None: + body["response_format"] = response_format + if stop is not None: + body["stop"] = stop + + # Passthrough: forward any other caller-supplied params verbatim. + for k, v in extra.items(): + if v is not None: + body.setdefault(k, v) + + # Walk [model, *fallback_models] on retriable errors. See sync + # chat_completion() above for the rationale. + attempts = [model, *(fallback_models or [])] + last_exc: Exception | None = None + for i, attempt_model in enumerate(attempts): + body["model"] = attempt_model + try: + return await self._request_with_payment("/v1/chat/completions", body) + except Exception as exc: + if not _should_fallback(exc): + raise + last_exc = exc + if i + 1 < len(attempts): + next_model = attempts[i + 1] + sys.stderr.write( + f"[blockrun_llm] {attempt_model} -> {next_model} " + f"({type(exc).__name__}: {str(exc)[:80]})\n" + ) + assert last_exc is not None + raise last_exc + + # ------------------------------------------------------------------ + # Streaming (SSE) chat completions — async mirror of LLMClient + # ------------------------------------------------------------------ + + async def chat_completion_stream( + self, + model: str, + messages: list[dict[str, Any]], + *, + max_tokens: int | None = None, + temperature: float | None = None, + top_p: float | None = None, + tools: list[dict[str, Any]] | None = None, + tool_choice: Any | None = None, + search: bool | None = None, + search_parameters: dict[str, Any] | None = None, + response_format: dict[str, Any] | None = None, + stop: str | list[str] | None = None, + fallback_models: list[str] | None = None, + **extra: Any, + ) -> AsyncIterator[ChatCompletionChunk]: + """ + Async streaming chat completion. See :meth:`LLMClient.chat_completion_stream` + for protocol details and the ``fallback_models`` semantics — + identical here, only the iteration protocol differs (``async for``). + """ + validate_model(model) + validate_max_tokens(max_tokens) + validate_temperature(temperature) + validate_top_p(top_p) + + body: dict[str, Any] = { + "model": model, + "messages": messages, + "stream": True, + "max_tokens": max_tokens or self.DEFAULT_MAX_TOKENS, + } + if temperature is not None: + body["temperature"] = temperature + if top_p is not None: + body["top_p"] = top_p + if tools is not None: + body["tools"] = tools + if tool_choice is not None: + body["tool_choice"] = tool_choice + if search_parameters is not None: + body["search_parameters"] = search_parameters + elif search is True: + body["search_parameters"] = {"mode": "on"} + if response_format is not None: + body["response_format"] = response_format + if stop is not None: + body["stop"] = stop + + # Passthrough: forward any other caller-supplied params verbatim. + for k, v in extra.items(): + if v is not None: + body.setdefault(k, v) + + attempts = [model, *(fallback_models or [])] + last_exc: Exception | None = None + + for i, attempt_model in enumerate(attempts): + body["model"] = attempt_model + inner = self._stream_with_payment("/v1/chat/completions", body) + chunks_yielded = 0 + try: + async for chunk in inner: + chunks_yielded += 1 + yield chunk + return + except Exception as exc: + if chunks_yielded > 0: + raise + if not _should_fallback(exc): + raise + last_exc = exc + if i + 1 < len(attempts): + next_model = attempts[i + 1] + sys.stderr.write( + f"[blockrun_llm] stream {attempt_model} -> {next_model} " + f"({type(exc).__name__}: {str(exc)[:80]})\n" + ) + finally: + # `async for` alone does not close `inner` when this generator + # is closed or an exception leaves the loop, so an abandoned + # stream would strand the paid `async with stream(...)` and its + # connection until GC. The sync path gets this from `yield + # from`; async has to ask. + await inner.aclose() + assert last_exc is not None + raise last_exc + + async def _stream_with_payment( + self, + endpoint: str, + body: dict[str, Any], + ) -> AsyncIterator[ChatCompletionChunk]: + """Async version of LLMClient._stream_with_payment. + + Honors :data:`LLMClient._STREAM_5XX_STATUSES` and + :data:`LLMClient._STREAM_5XX_BACKOFFS` for retries (in-band exponential + backoff on transient upstream errors before raising). + """ + url = f"{self.api_url}{endpoint}" + req_headers = {"Content-Type": "application/json", "User-Agent": _get_user_agent()} + + is_search = "search_parameters" in body or body.get("search") is True + timeout = self.search_timeout if is_search else self.timeout + + backoffs = LLMClient._STREAM_5XX_BACKOFFS + statuses_5xx = LLMClient._STREAM_5XX_STATUSES + + # ----- Phase 1: probe (no payment header) ----- + payment_headers: dict[str, str] | None = None + cost_usd = 0.0 + + for attempt in range(len(backoffs) + 1): + async with self._client.stream( + "POST", url, json=body, headers=req_headers, timeout=timeout + ) as resp1: + if resp1.status_code == 200: + async for chunk in self._aiter_sse_chunks(resp1): + yield chunk + return + await resp1.aread() + if resp1.status_code == 402: + # Account rail: a 402 is the account being out of credit, not a + # challenge to sign. Nothing here can sign, so say so plainly. + raise_for_api_key_402(resp1, self.api_key) + payment_headers, cost_usd = self._sign_payment_from_response(body, resp1) + break + if resp1.status_code in statuses_5xx and attempt < len(backoffs): + import asyncio + + await asyncio.sleep(backoffs[attempt]) + continue + self._raise_stream_error(resp1, after_payment=False) + else: + raise APIError("stream probe exhausted retries", 0, None) + + # ----- Phase 2: stream with PAYMENT-SIGNATURE ----- + # Settled from here on; see the sync path. + assert payment_headers is not None + # `async for` does NOT close the inner async generator when this one is + # closed or an exception leaves the loop, so the paid `async with + # self._client.stream(...)` inside it would stay suspended and hold the + # connection until GC finalization. The sync path gets this for free: + # `yield from` propagates close() into the subgenerator. Close it here. + paid = self._astream_paid_phase(url, body, payment_headers, cost_usd, timeout) + try: + async for chunk in paid: + yield chunk + except (httpx.HTTPError, APIError) as exc: + _mark_settled(exc) + raise + finally: + await paid.aclose() + + async def _astream_paid_phase( + self, + url: str, + body: dict[str, Any], + payment_headers: dict[str, str], + cost_usd: float, + timeout: float | None, + ) -> AsyncIterator[ChatCompletionChunk]: + """Phase 2 of the async stream: the paid, already-settled leg.""" + backoffs = LLMClient._STREAM_5XX_BACKOFFS + statuses_5xx = LLMClient._STREAM_5XX_STATUSES + for attempt in range(len(backoffs) + 1): + async with self._client.stream( + "POST", url, json=body, headers=payment_headers, timeout=timeout + ) as resp2: + if resp2.status_code == 200: + # AsyncLLMClient only tracks ``_last_call_cost`` (no session + # totals in the async path — matches the existing async + # chat_completion convention). + if cost_usd > 0: + self._last_call_cost = cost_usd + self._capture_settlement(resp2) + async for chunk in self._aiter_and_archive( + resp2, body, cost_usd, streaming=True + ): + yield chunk + return + await resp2.aread() + if resp2.status_code == 402: + # Account rail: a 402 is the account being out of credit, not a + # challenge to sign. Nothing here can sign, so say so plainly. + raise_for_api_key_402(resp2, self.api_key) + raise PaymentError("Payment was rejected. Check your wallet balance.") + if resp2.status_code in statuses_5xx and attempt < len(backoffs): + import asyncio + + await asyncio.sleep(backoffs[attempt]) + continue + self._raise_stream_error(resp2, after_payment=True) + + async def _aiter_and_archive( + self, + response: httpx.Response, + body: dict[str, Any], + cost_usd: float, + *, + streaming: bool = True, + ) -> AsyncIterator[ChatCompletionChunk]: + """Async mirror of :meth:`LLMClient._iter_and_archive`. Writes the + assembled ``chat.completion`` response to ``~/.blockrun/data/`` and + the cost row to ``~/.blockrun/cost_log.jsonl`` once the stream + finishes — only for paid calls (cost_usd > 0).""" + assembled_id: str | None = None + assembled_model: str | None = None + assembled_created: int = 0 + content_parts: list[str] = [] + finish_reason: str | None = None + usage_dict: dict[str, Any] | None = None + + async for chunk in self._aiter_sse_chunks(response): + if chunk.choices: + choice = chunk.choices[0] + content = stream_choice_content(choice) + if content: + content_parts.append(content) + fr = stream_choice_finish_reason(choice) + if fr: + finish_reason = fr + if assembled_id is None: + _id, _model, _created = chunk_meta(chunk) + if _id: + assembled_id = _id + assembled_model = _model + assembled_created = _created + _usage = chunk_usage_dict(chunk) + if _usage is not None: + usage_dict = _usage + # Race-free per-call x402 charge — see LLMClient._iter_and_archive. + chunk.cost_usd = cost_usd + yield chunk + + if cost_usd > 0: + from .cache import save_to_cache + + response_data: dict[str, Any] = { + "id": assembled_id or "stream", + "object": "chat.completion", + "created": assembled_created or int(__import__("time").time()), + "model": assembled_model or body.get("model"), + "choices": [ + { + "index": 0, + "message": { + "role": "assistant", + "content": "".join(content_parts), + }, + "finish_reason": finish_reason, + } + ], + "stream": streaming, + } + if usage_dict: + response_data["usage"] = usage_dict + try: + save_to_cache( + "/v1/chat/completions", + body, + response_data, + cost_usd=cost_usd, + **self._billing_meta(), + ) + except Exception: + pass + self._log_transaction("/v1/chat/completions", body, response_data, cost_usd) + + @staticmethod + async def _aiter_sse_chunks(response: httpx.Response) -> AsyncIterator[ChatCompletionChunk]: + """Async variant of :meth:`LLMClient._iter_sse_chunks`.""" + async for raw_line in response.aiter_lines(): + if not raw_line or not raw_line.startswith("data: "): + continue + payload = raw_line[6:].strip() + if payload == "[DONE]": + return + try: + chunk_dict = _json.loads(payload) + except Exception: + continue + try: + yield ChatCompletionChunk(**chunk_dict) + except Exception: + yield ChatCompletionChunk.model_construct(**chunk_dict) + + # Reuse the sync helpers — Python class-attribute lookup binds them + # correctly to whatever self is passed when the bound method is called. + _sign_payment_from_response = LLMClient._sign_payment_from_response + _raise_stream_error = LLMClient._raise_stream_error + + async def _request_with_payment(self, endpoint: str, body: dict[str, Any]) -> ChatResponse: + """Make async request with automatic payment handling.""" + url = f"{self.api_url}{endpoint}" + req_headers = {"Content-Type": "application/json", "User-Agent": _get_user_agent()} + + response = await self._client.post(url, json=body, headers=req_headers) + + # Auto-retry on transient server errors + if response.status_code in (502, 503): + import asyncio + + await asyncio.sleep(1) + response = await self._client.post(url, json=body, headers=req_headers) + + if response.status_code == 402: + # Account rail: a 402 is the account being out of credit, not a + # challenge to sign. Nothing here can sign, so say so plainly. + raise_for_api_key_402(response, self.api_key) + # See the sync path: past this point the payment is settled. + try: + return await self._handle_payment_and_retry(url, body, response) + except (httpx.HTTPError, APIError) as exc: + _mark_settled(exc) + raise + + if response.status_code != 200: + try: + error_body = response.json() + except Exception: + error_body = {"error": "Request failed"} + raise APIError( + f"API error: {response.status_code}", + response.status_code, + sanitize_error_response(error_body), + retry_after=retry_after_of(response), + ) + + # 200 on first attempt => no payment required (free / cached). Charge $0. + chat_response = ChatResponse(**response.json()) + chat_response.cost_usd = 0.0 + return chat_response + + async def _handle_payment_and_retry( + self, + url: str, + body: dict[str, Any], + response: httpx.Response, + ) -> ChatResponse: + """Handle 402 response asynchronously.""" + # Get payment required header (x402 library uses lowercase) + payment_header = response.headers.get("payment-required") + if not payment_header: + try: + resp_body = response.json() + if "x402" in resp_body: + payment_header = resp_body + except Exception: + pass + + if not payment_header: + raise PaymentError("402 response but no payment requirements found") + + if isinstance(payment_header, str): + payment_required = parse_payment_required(payment_header) + else: + payment_required = payment_header + + details = extract_payment_details(payment_required) + + # Enforce the spend limit on the QUOTE, before signing. This handler + # computes its cost_usd only after the paid POST returns (it prefers the + # price echoed on the response), which is far too late to refuse. + _enforce_spend_limits( + self, + float(details.get("amount", 0)) / 1e6, + body.get("model") if isinstance(body, dict) else None, + ) + + # Create signed payment payload (v2 format) + # SECURITY: Signing happens locally - only the signature is sent to server + resource = details.get("resource") or {} + _warn_if_clamped(body, resource.get("description")) + # Pass through extensions from server (for Bazaar discovery) + extensions = payment_required.get("extensions", {}) + payment_payload = create_payment_payload( + account=self.account, + recipient=details["recipient"], + amount=details["amount"], + network=details.get("network", "eip155:84532" if self.is_testnet() else "eip155:8453"), + resource_url=validate_resource_url( + resource.get("url", f"{self.api_url}/v1/chat/completions"), self.api_url + ), + resource_description=resource.get("description", "BlockRun AI API call"), + max_timeout_seconds=details.get("maxTimeoutSeconds", 300), + extra=details.get("extra"), + extensions=extensions, + asset=details.get("asset"), + ) + + # Retry with payment (x402 library expects PAYMENT-SIGNATURE header) + # Use longer timeout for Live Search requests + is_search_request = "search_parameters" in body or body.get("search") is True + request_timeout = self.search_timeout if is_search_request else self.timeout + + payment_headers = { + "Content-Type": "application/json", + "User-Agent": _get_user_agent(), + "PAYMENT-SIGNATURE": payment_payload, + } + + # Retry with payment, with one automatic retry on 502/503 + retry_response = await self._client.post( + url, json=body, headers=payment_headers, timeout=request_timeout + ) + if retry_response.status_code in (502, 503): + import asyncio + + await asyncio.sleep(1) + retry_response = await self._client.post( + url, json=body, headers=payment_headers, timeout=request_timeout + ) + + if retry_response.status_code == 402: + # Account rail: a 402 is the account being out of credit, not a + # challenge to sign. Nothing here can sign, so say so plainly. + raise_for_api_key_402(retry_response, self.api_key) + raise PaymentError("Payment was rejected. Check your wallet balance.") + + if retry_response.status_code != 200: + try: + error_body = retry_response.json() + except Exception: + error_body = {"error": "Request failed"} + raise APIError( + f"{paid_request_error_prefix(retry_response.headers)}: {retry_response.status_code}", + retry_response.status_code, + sanitize_error_response(error_body), + retry_after=retry_after_of(retry_response), + ) + + # Extract cost and save locally + price_info = {} + try: + resp_body = response.json() + price_info = resp_body.get("price", {}) + except Exception: + pass + cost_usd = ( + float(price_info.get("amount", 0)) + if price_info + else float(details.get("amount", 0)) / 1e6 + ) + self._last_call_cost = cost_usd + settlement = self._capture_settlement(retry_response) + + response_data = retry_response.json() + # Per-call real charge + settlement (see sync _handle_payment_and_retry). + chat_response = ChatResponse(**response_data) + chat_response.cost_usd = cost_usd + if settlement: + chat_response.settlement = dict(settlement) + from .cache import save_to_cache + + save_to_cache( + "/v1/chat/completions", + body, + response_data, + cost_usd=cost_usd, + **self._billing_meta(), + ) + self._log_transaction("/v1/chat/completions", body, response_data, cost_usd) + + return chat_response + + async def _request_with_payment_raw( + self, endpoint: str, body: dict[str, Any] + ) -> dict[str, Any]: + """Make async request with automatic payment handling, returning raw JSON.""" + from .cache import get_cached, save_to_cache + + # Check cache first + cached = get_cached(endpoint, body) + if cached is not None: + return cached + + url = f"{self.api_url}{endpoint}" + req_headers = {"Content-Type": "application/json", "User-Agent": _get_user_agent()} + + response = await self._client.post(url, json=body, headers=req_headers) + + # Auto-retry on transient server errors + if response.status_code in (502, 503): + import asyncio + + await asyncio.sleep(1) + response = await self._client.post(url, json=body, headers=req_headers) + + if response.status_code == 402: + # Account rail: a 402 is the account being out of credit, not a + # challenge to sign. Nothing here can sign, so say so plainly. + raise_for_api_key_402(response, self.api_key) + try: + result = await self._handle_payment_and_retry_raw(url, body, response) + except (httpx.HTTPError, APIError) as exc: + _mark_settled(exc) + raise + save_to_cache( + endpoint, + body, + result, + cost_usd=self._last_call_cost, + **self._billing_meta(), + ) + self._log_transaction(endpoint, body, result, self._last_call_cost) + return result + + if response.status_code != 200: + try: + error_body = response.json() + except Exception: + error_body = {"error": "Request failed"} + raise APIError( + f"API error: {response.status_code}", + response.status_code, + sanitize_error_response(error_body), + retry_after=retry_after_of(response), + ) + + return response.json() + + async def _handle_payment_and_retry_raw( + self, + url: str, + body: dict[str, Any], + response: httpx.Response, + ) -> dict[str, Any]: + """Handle 402 response asynchronously for raw endpoints.""" + payment_header = response.headers.get("payment-required") + if not payment_header: + try: + resp_body = response.json() + if "x402" in resp_body: + payment_header = resp_body + except Exception: + pass + + if not payment_header: + raise PaymentError("402 response but no payment requirements found") + + if isinstance(payment_header, str): + payment_required = parse_payment_required(payment_header) + else: + payment_required = payment_header + + details = extract_payment_details(payment_required) + + resource = details.get("resource") or {} + _warn_if_clamped(body, resource.get("description")) + extensions = payment_required.get("extensions", {}) + payment_payload = create_payment_payload( + account=self.account, + recipient=details["recipient"], + amount=details["amount"], + network=details.get("network", "eip155:84532" if self.is_testnet() else "eip155:8453"), + resource_url=validate_resource_url(resource.get("url", url), self.api_url), + resource_description=resource.get("description", "BlockRun AI API call"), + max_timeout_seconds=details.get("maxTimeoutSeconds", 300), + extra=details.get("extra"), + extensions=extensions, + asset=details.get("asset"), + ) + + payment_headers = { + "Content-Type": "application/json", + "User-Agent": _get_user_agent(), + "PAYMENT-SIGNATURE": payment_payload, + } + + # Retry with payment, with one automatic retry on 502/503 + retry_response = await self._client.post( + url, json=body, headers=payment_headers, timeout=self.timeout + ) + if retry_response.status_code in (502, 503): + import asyncio + + await asyncio.sleep(1) + retry_response = await self._client.post( + url, json=body, headers=payment_headers, timeout=self.timeout + ) + + if retry_response.status_code == 402: + # Account rail: a 402 is the account being out of credit, not a + # challenge to sign. Nothing here can sign, so say so plainly. + raise_for_api_key_402(retry_response, self.api_key) + raise PaymentError("Payment was rejected. Check your wallet balance.") + + if retry_response.status_code != 200: + try: + error_body = retry_response.json() + except Exception: + error_body = {"error": "Request failed"} + raise APIError( + f"{paid_request_error_prefix(retry_response.headers)}: {retry_response.status_code}", + retry_response.status_code, + sanitize_error_response(error_body), + retry_after=retry_after_of(retry_response), + ) + + cost_usd = float(details.get("amount", 0)) / 1e6 + self._last_call_cost = cost_usd + self._capture_settlement(retry_response) + + return retry_response.json() + + async def _get_with_payment_raw( + self, endpoint: str, params: dict[str, Any] | None = None + ) -> dict[str, Any]: + """Async GET with x402 payment handling, returning raw JSON.""" + from .cache import get_cached, save_to_cache + + cache_key_body = params or {} + cached = get_cached(endpoint, cache_key_body) + if cached is not None: + return cached + + url = f"{self.api_url}{endpoint}" + req_headers = {"User-Agent": _get_user_agent()} + + response = await self._client.get(url, params=params, headers=req_headers) + + if response.status_code in (502, 503): + import asyncio + + await asyncio.sleep(1) + response = await self._client.get(url, params=params, headers=req_headers) + + if response.status_code == 402: + # Account rail: a 402 is the account being out of credit, not a + # challenge to sign. Nothing here can sign, so say so plainly. + raise_for_api_key_402(response, self.api_key) + result = await self._handle_get_payment_and_retry(url, params, response) + save_to_cache( + endpoint, + cache_key_body, + result, + cost_usd=self._last_call_cost, + **self._billing_meta(), + ) + self._log_transaction(endpoint, cache_key_body, result, self._last_call_cost) + return result + + if response.status_code != 200: + try: + error_body = response.json() + except Exception: + error_body = {"error": "Request failed"} + raise APIError( + f"API error: {response.status_code}", + response.status_code, + sanitize_error_response(error_body), + retry_after=retry_after_of(response), + ) + + return response.json() + + async def _handle_get_payment_and_retry( + self, + url: str, + params: dict[str, Any] | None, + response: httpx.Response, + ) -> dict[str, Any]: + """Handle 402 response asynchronously for GET endpoints.""" + payment_header = response.headers.get("payment-required") + if not payment_header: + try: + resp_body = response.json() + if "x402" in resp_body: + payment_header = resp_body + except Exception: + pass + + if not payment_header: + raise PaymentError("402 response but no payment requirements found") + + if isinstance(payment_header, str): + payment_required = parse_payment_required(payment_header) + else: + payment_required = payment_header + + details = extract_payment_details(payment_required) + + resource = details.get("resource") or {} + extensions = payment_required.get("extensions", {}) + payment_payload = create_payment_payload( + account=self.account, + recipient=details["recipient"], + amount=details["amount"], + network=details.get("network", "eip155:84532" if self.is_testnet() else "eip155:8453"), + resource_url=validate_resource_url(resource.get("url", url), self.api_url), + resource_description=resource.get("description", "BlockRun AI API call"), + max_timeout_seconds=details.get("maxTimeoutSeconds", 300), + extra=details.get("extra"), + extensions=extensions, + asset=details.get("asset"), + ) + + payment_headers = { + "User-Agent": _get_user_agent(), + "PAYMENT-SIGNATURE": payment_payload, + } + + retry_response = await self._client.get( + url, params=params, headers=payment_headers, timeout=self.timeout + ) + if retry_response.status_code in (502, 503): + import asyncio + + await asyncio.sleep(1) + retry_response = await self._client.get( + url, params=params, headers=payment_headers, timeout=self.timeout + ) + + if retry_response.status_code == 402: + # Account rail: a 402 is the account being out of credit, not a + # challenge to sign. Nothing here can sign, so say so plainly. + raise_for_api_key_402(retry_response, self.api_key) + raise PaymentError("Payment was rejected. Check your wallet balance.") + + if retry_response.status_code != 200: + try: + error_body = retry_response.json() + except Exception: + error_body = {"error": "Request failed"} + raise APIError( + f"{paid_request_error_prefix(retry_response.headers)}: {retry_response.status_code}", + retry_response.status_code, + sanitize_error_response(error_body), + retry_after=retry_after_of(retry_response), + ) + + cost_usd = float(details.get("amount", 0)) / 1e6 + self._last_call_cost = cost_usd + self._capture_settlement(retry_response) + + return retry_response.json() + + async def image_edit( + self, + prompt: str, + image: str | list[str], + *, + model: str = "openai/gpt-image-2", + mask: str | None = None, + size: str = "1024x1024", + n: int = 1, + ) -> ImageResponse: + """Async image editing (img2img). ``image`` may be a single data URI or + a list of 1-4 data URIs for multi-image fusion (openai/* up to 4, + google/* up to 3).""" + body: dict[str, Any] = { + "model": model, + "prompt": prompt, + "image": image, + "size": size, + "n": n, + } + if mask is not None: + body["mask"] = mask + + data = await self._request_with_payment_raw("/v1/images/image2image", body) + return ImageResponse(**data) + + async def search( + self, + query: str, + *, + sources: list[str] | None = None, + max_results: int = 10, + from_date: str | None = None, + to_date: str | None = None, + ) -> SearchResult: + """Async standalone search.""" + body: dict[str, Any] = { + "query": query, + "max_results": max_results, + } + if sources is not None: + body["sources"] = sources + if from_date is not None: + body["from_date"] = from_date + if to_date is not None: + body["to_date"] = to_date + + data = await self._request_with_payment_raw("/v1/search", body) + return SearchResult(**data) + + # ── Prediction Markets (Powered by Predexon) ──────────────────────────── + + async def pm(self, path: str, **params: Any) -> dict[str, Any]: + """Async query Predexon prediction market data (GET). Powered by Predexon.""" + return await self._get_with_payment_raw(f"/v1/pm/{path}", params or None) + + async def pm_query(self, path: str, query: dict[str, Any]) -> dict[str, Any]: + """Async structured query for Predexon data (POST). Powered by Predexon.""" + return await self._request_with_payment_raw(f"/v1/pm/{path}", query) + + async def pm_markets(self, **params: Any) -> dict[str, Any]: + """RETIRED — ``/v1/pm/markets`` no longer exists. + + Predexon sunset market matching on 2026-07-20 and the whole + canonical layer went with it, so this path returns 410 upstream. + Use ``pm("markets/search", q=...)`` for cross-venue lookups. + + Kept as a raising stub rather than deleted so upgrading does not + break imports or attribute access; it raises before any network + I/O, so you never pay a round trip to learn it is gone. + + :raises RetiredEndpointError: always. + """ + raise RetiredEndpointError( + "/v1/pm/markets was sunset by Predexon on 2026-07-20 (upstream 410). Use pm('markets/search', q=...) for cross-venue lookups." + ) + + async def pm_listings(self, **params: Any) -> dict[str, Any]: + """RETIRED — ``/v1/pm/markets/listings`` no longer exists. + + Predexon sunset market matching on 2026-07-20 and the whole + canonical layer went with it, so this path returns 410 upstream. + Use ``pm("markets/search", q=...)`` for cross-venue lookups. + + Kept as a raising stub rather than deleted so upgrading does not + break imports or attribute access; it raises before any network + I/O, so you never pay a round trip to learn it is gone. + + :raises RetiredEndpointError: always. + """ + raise RetiredEndpointError( + "/v1/pm/markets/listings was sunset by Predexon on 2026-07-20 (upstream 410). Use pm('markets/search', q=...) for cross-venue lookups." + ) + + async def pm_outcome(self, predexon_id: str) -> dict[str, Any]: + """RETIRED — ``/v1/pm/outcomes/{predexon_id}`` no longer exists. + + Predexon sunset market matching on 2026-07-20 and the whole + canonical layer went with it, so this path returns 410 upstream. + Use ``pm("markets/search", q=...)`` for cross-venue lookups. + + Kept as a raising stub rather than deleted so upgrading does not + break imports or attribute access; it raises before any network + I/O, so you never pay a round trip to learn it is gone. + + :raises RetiredEndpointError: always. + """ + raise RetiredEndpointError( + "/v1/pm/outcomes/{predexon_id} was sunset by Predexon on 2026-07-20 (upstream 410). Use pm('markets/search', q=...) for cross-venue lookups." + ) + + async def pm_polymarket_markets(self, **params: Any) -> dict[str, Any]: + """List Polymarket markets (Predexon v2). Tier 1 ($0.001/call).""" + return await self.pm("polymarket/markets", **params) + + async def pm_polymarket_events(self, **params: Any) -> dict[str, Any]: + """List Polymarket events (Predexon v2). Tier 1 ($0.001/call).""" + return await self.pm("polymarket/events", **params) + + async def pm_polymarket_markets_keyset(self, **params: Any) -> dict[str, Any]: + """Polymarket markets with cursor-based keyset pagination. Tier 1 ($0.001/call).""" + return await self.pm("polymarket/markets/keyset", **params) + + async def pm_polymarket_events_keyset(self, **params: Any) -> dict[str, Any]: + """Polymarket events with cursor-based keyset pagination. Tier 1 ($0.001/call).""" + return await self.pm("polymarket/events/keyset", **params) + + async def pm_polymarket_positions(self, **params: Any) -> dict[str, Any]: + """Polymarket open positions (per-wallet, market-level PnL). + Tier 1 ($0.001/call).""" + return await self.pm("polymarket/positions", **params) + + async def pm_polymarket_trades(self, **params: Any) -> dict[str, Any]: + """Recent Polymarket trades. Tier 1 ($0.001/call).""" + return await self.pm("polymarket/trades", **params) + + async def pm_polymarket_leaderboard(self, **params: Any) -> dict[str, Any]: + """Polymarket trader leaderboard. Tier 1 ($0.001/call).""" + return await self.pm("polymarket/leaderboard", **params) + + async def pm_kalshi_markets(self, **params: Any) -> dict[str, Any]: + """List Kalshi markets. Tier 1 ($0.001/call).""" + return await self.pm("kalshi/markets", **params) + + async def pm_limitless_markets(self, **params: Any) -> dict[str, Any]: + """List Limitless markets. Tier 1 ($0.001/call).""" + return await self.pm("limitless/markets", **params) + + async def pm_sports_categories(self) -> dict[str, Any]: + """List available sports categories. Tier 1 ($0.001/call). + + .. warning:: + Upstream is returning 500 for every ``sports/*`` path as of + 2026-08-04. The route still resolves, so this keeps working the + moment Predexon restores it, but do not build on it yet. + """ + return await self.pm("sports/categories") + + async def pm_sports_markets(self, **params: Any) -> dict[str, Any]: + """List sports markets grouped by game. Tier 1 ($0.001/call). + + .. warning:: + Upstream is returning 500 for every ``sports/*`` path as of + 2026-08-04. The route still resolves, so this keeps working the + moment Predexon restores it, but do not build on it yet. + """ + return await self.pm("sports/markets", **params) + + async def pm_wallet_identity(self, wallet: str) -> dict[str, Any]: + """Identity + profile for one wallet. Tier 2 ($0.005/call).""" + return await self.pm(f"polymarket/wallet/identity/{wallet}") + + async def pm_wallet_identities(self, addresses: list[str]) -> dict[str, Any]: + """Bulk identity for up to 200 wallet addresses. Tier 2 ($0.005/call).""" + return await self.pm_query("polymarket/wallet/identities", {"addresses": addresses}) + + async def pm_wallet_cluster(self, address: str) -> dict[str, Any]: + """Wallet-cluster discovery (on-chain transfers + identity proofs). + Tier 2 ($0.005/call).""" + return await self.pm(f"polymarket/wallet/{address}/cluster") + + # ── DefiLlama (DeFi protocols / TVL / yields / prices) ────────────────── + + async def defi(self, path: str, **params: Any) -> dict[str, Any]: + """Async query DefiLlama DeFi data (GET). $0.005/call ($0.001 for prices).""" + return await self._get_with_payment_raw(f"/v1/defillama/{path}", params or None) + + async def defi_protocols(self) -> dict[str, Any]: + """Async: all DeFi protocols with TVL ($0.005/call).""" + return await self.defi("protocols") + + async def defi_protocol(self, slug: str) -> dict[str, Any]: + """Async: single protocol details + historical TVL ($0.005/call).""" + return await self.defi(f"protocol/{slug}") + + async def defi_chains(self) -> dict[str, Any]: + """Async: current TVL of every chain ($0.005/call).""" + return await self.defi("chains") + + async def defi_yields(self, **params: Any) -> dict[str, Any]: + """Async: yield pools with APY/TVL ($0.005/call).""" + return await self.defi("yields", **params) + + async def defi_prices(self, coins: list[str] | str) -> dict[str, Any]: + """Async: token price lookup ($0.001/call).""" + joined = ",".join(coins) if isinstance(coins, list) else coins + return await self.defi(f"prices/{joined}") + + # ── 0x DEX (swap quotes + gasless) — free passthrough ─────────────────── + + async def dex( + self, + path: str, + *, + method: str = "GET", + body: dict[str, Any] | None = None, + **params: Any, + ) -> dict[str, Any]: + """Async query the 0x Swap / Gasless APIs (free passthrough).""" + endpoint = f"/v1/zerox/{path}" + if method.upper() == "POST": + return await self._request_with_payment_raw(endpoint, body or {}) + return await self._get_with_payment_raw(endpoint, params or None) + + async def dex_price(self, **params: Any) -> dict[str, Any]: + """Async: indicative Permit2 swap price (free).""" + return await self.dex("price", **params) + + async def dex_quote(self, **params: Any) -> dict[str, Any]: + """Async: firm Permit2 swap quote (free).""" + return await self.dex("quote", **params) + + async def dex_gasless_price(self, **params: Any) -> dict[str, Any]: + """Async: gasless indicative price quote (free).""" + return await self.dex("gasless/price", **params) + + async def dex_gasless_quote(self, **params: Any) -> dict[str, Any]: + """Async: gasless firm quote — returns trade.eip712 to sign (free).""" + return await self.dex("gasless/quote", **params) + + async def dex_gasless_submit(self, body: dict[str, Any]) -> dict[str, Any]: + """Async: submit a signed gasless trade (free).""" + return await self.dex("gasless/submit", method="POST", body=body) + + async def dex_gasless_status(self, trade_hash: str) -> dict[str, Any]: + """Async: poll a gasless trade's status (free).""" + return await self.dex(f"gasless/status/{trade_hash}") + + async def dex_chains(self) -> dict[str, Any]: + """Async: chains where the Swap API is supported (free).""" + return await self.dex("swap/chains") + + async def dex_gasless_chains(self) -> dict[str, Any]: + """Async: chains where the Gasless API is supported (free).""" + return await self.dex("gasless/chains") + + # ── Modal Sandbox (pay-per-call cloud compute) ─────────────────────────── + + async def modal(self, path: str, body: dict[str, Any] | None = None) -> dict[str, Any]: + """Async call the Modal sandbox compute API (POST passthrough).""" + return await self._request_with_payment_raw(f"/v1/modal/{path}", body or {}) + + async def modal_sandbox_create(self, **body: Any) -> dict[str, Any]: + """Async: create a sandbox ($0.01 CPU / $0.05 GPU).""" + return await self.modal("sandbox/create", body) + + async def modal_sandbox_exec( + self, sandbox_id: str, command: list[str], **body: Any + ) -> dict[str, Any]: + """Async: execute a command in a sandbox ($0.001).""" + return await self.modal( + "sandbox/exec", {"sandbox_id": sandbox_id, "command": command, **body} + ) + + async def modal_sandbox_status(self, sandbox_id: str) -> dict[str, Any]: + """Async: check a sandbox's status ($0.001).""" + return await self.modal("sandbox/status", {"sandbox_id": sandbox_id}) + + async def modal_sandbox_terminate(self, sandbox_id: str) -> dict[str, Any]: + """Async: terminate a sandbox ($0.001).""" + return await self.modal("sandbox/terminate", {"sandbox_id": sandbox_id}) + + async def list_models(self) -> list[dict[str, Any]]: + """List available LLM models asynchronously.""" + response = await self._client.get(f"{self.api_url}/v1/models") + + if response.status_code != 200: + try: + error_body = response.json() + except Exception: + error_body = {"error": "Request failed"} + raise APIError( + f"Failed to list models: {response.status_code}", + response.status_code, + sanitize_error_response(error_body), + retry_after=retry_after_of(response), + ) + + return response.json().get("data", []) + + async def list_image_models(self) -> list[dict[str, Any]]: + """List available image generation models asynchronously. + + ``/v1/images/models`` was deprecated server-side; this filters the + unified ``/v1/models`` catalog by ``categories: ["image"]`` so existing + callers keep working. + """ + models = await self.list_models() + return [m for m in models if "image" in (m.get("categories") or [])] + + async def list_all_models(self) -> list[dict[str, Any]]: + """ + List all available models (chat, image, music, etc.) asynchronously. + + Returns: + List of all model information dicts with ``type`` set per category. + """ + all_models = await self.list_models() + for m in all_models: + cats = m.get("categories") or [] + if "chat" in cats: + m["type"] = "llm" + elif "image" in cats: + m["type"] = "image" + elif "music" in cats or "audio" in cats: + m["type"] = "music" + else: + m["type"] = cats[0] if cats else "llm" + return all_models + + @property + def payment_mode(self) -> str: + """Which rail this client pays on: ``"apikey"`` or ``"wallet"``. + + Worth checking once at startup when both a key and a wallet are + configured in the environment: it is the difference between + spending credit and spending USDC.""" + return payment_mode(self) + + def get_wallet_address(self) -> str: + # No address on the account rail: payment comes from prepaid + # credit, so there is nothing to return but the empty string. + if self.api_key: + return "" + """Get the wallet address.""" + return self.account.address + + def is_testnet(self) -> bool: + """Check if client is configured for testnet.""" + return "testnet.blockrun.ai" in self.api_url + + def _billing_meta(self) -> dict[str, str | None]: + """Billing metadata for cost-log entries.""" + return { + "wallet": self.account.address, + "network": _detect_network(self.api_url), + "client_kind": type(self).__name__, + } + + def _log_transaction( + self, + endpoint: str, + body: dict[str, Any], + response: Any, + cost_usd: float, + ) -> None: + """Async-client twin of :meth:`LLMClient._log_transaction`.""" + logger = self._tx_logger + if logger is None: + return + settlement = self._last_settlement + self._last_settlement = None + try: + logger.log( + endpoint=endpoint, + request=body, + response=response, + cost_usd=cost_usd, + model=(body.get("model") if isinstance(body, dict) else None), + wallet=self.account.address, + network=_detect_network(self.api_url), + client_kind=type(self).__name__, + settlement=settlement, + ) + except Exception: + pass + + async def get_balance(self) -> float: + # Returning 0 would be the worst available answer: it is + # indistinguishable from an empty wallet, and an agent gating on it + # would stop calling a well-funded account. + if self.api_key: + raise wallet_only("get_balance") + """ + Get USDC balance on Base network. + + Automatically detects mainnet vs testnet based on API URL: + - Mainnet: Base (Chain ID 8453) + - Testnet: Base Sepolia (Chain ID 84532) + + Returns: + float: USDC balance (6 decimal places normalized) + + Example: + balance = await client.get_balance() + print(f"Balance: ${balance:.2f} USDC") + """ + # USDC contracts + # Mainnet: Base + # Testnet: Base Sepolia + if self.is_testnet(): + usdc_contract = "0x036CbD53842c5426634e7929541eC2318f3dCF7e" + rpcs = [ + "https://sepolia.base.org", + "https://base-sepolia-rpc.publicnode.com", + ] + else: + usdc_contract = "0x833589fCD6eDb6E08f4c7C32D4f71b54bdA02913" + rpcs = [ + "https://base.publicnode.com", + "https://mainnet.base.org", + "https://base.meowrpc.com", + ] + + # balanceOf(address) function selector + selector = "0x70a08231" + # Pad wallet address to 32 bytes + padded_address = self.account.address[2:].lower().zfill(64) + data = selector + padded_address + + payload = { + "jsonrpc": "2.0", + "method": "eth_call", + "params": [{"to": usdc_contract, "data": data}, "latest"], + "id": 1, + } + + last_error = None + async with httpx.AsyncClient(timeout=10) as http_client: + for rpc in rpcs: + try: + response = await http_client.post(rpc, json=payload) + result = response.json().get("result", "0x0") + # Convert from hex and normalize (USDC has 6 decimals) + balance_raw = int(result, 16) + return balance_raw / 1_000_000 + except Exception as e: + last_error = e + continue + + # If all RPCs failed, raise the last error + raise last_error or Exception("All RPCs failed") + + async def close(self): + """Close the async HTTP client.""" + await self._client.aclose() + + async def __aenter__(self): + return self + + async def __aexit__(self, exc_type, exc_val, exc_tb): + await self.close() + + +# ============================================================================= +# Testnet Convenience Functions +# ============================================================================= + + +def testnet_client(private_key: str | None = None, **kwargs) -> LLMClient: + """ + Create a testnet LLM client for development and testing. + + This is a convenience function that creates an LLMClient configured + for the BlockRun testnet (Base Sepolia). + + Args: + private_key: Base Sepolia wallet private key (or set BLOCKRUN_WALLET_KEY env var) + **kwargs: Additional arguments passed to LLMClient + + Returns: + LLMClient configured for testnet + + Example: + from blockrun_llm import testnet_client + + client = testnet_client() # Uses BLOCKRUN_WALLET_KEY + response = client.chat("openai/gpt-oss-20b", "Hello!") + + Testnet Setup: + 1. Get testnet ETH from https://www.alchemy.com/faucets/base-sepolia + 2. Get testnet USDC from https://faucet.circle.com/ + 3. Use your wallet with testnet funds + + Available Testnet Models: + - openai/gpt-oss-20b + - openai/gpt-oss-120b + """ + return LLMClient( + private_key=private_key, + api_url=LLMClient.TESTNET_API_URL, + **kwargs, + ) + + +async def async_testnet_client(private_key: str | None = None, **kwargs) -> AsyncLLMClient: + """ + Create an async testnet LLM client for development and testing. + + This is a convenience function that creates an AsyncLLMClient configured + for the BlockRun testnet (Base Sepolia). + + Args: + private_key: Base Sepolia wallet private key (or set BLOCKRUN_WALLET_KEY env var) + **kwargs: Additional arguments passed to AsyncLLMClient + + Returns: + AsyncLLMClient configured for testnet + + Example: + from blockrun_llm import async_testnet_client + + async with async_testnet_client() as client: + response = await client.chat("openai/gpt-oss-20b", "Hello!") + """ + return AsyncLLMClient( + private_key=private_key, + api_url=AsyncLLMClient.TESTNET_API_URL, + **kwargs, + ) diff --git a/blockrun_llm/image.py b/blockrun_llm/image.py new file mode 100644 index 0000000..60c14bf --- /dev/null +++ b/blockrun_llm/image.py @@ -0,0 +1,458 @@ +""" +BlockRun Image Client - Generate images via x402 micropayments. + +SECURITY NOTE - Private Key Handling: +===================================== +Your private key NEVER leaves your machine. Here's what happens: + +1. Key stays local - only used to sign an EIP-712 typed data message +2. Only the SIGNATURE is sent in the PAYMENT-SIGNATURE header +3. BlockRun verifies the signature on-chain via Coinbase CDP facilitator +4. Your actual private key is NEVER transmitted to any server + +This is the same security model as signing any blockchain transaction. + +Usage: + from blockrun_llm import ImageClient + + # Initialize with private key from env (BLOCKRUN_WALLET_KEY) + client = ImageClient() + + # Generate an image + result = client.generate("A cute cat wearing a space helmet") + print(result.data[0].url) + + # With specific model + result = client.generate("prompt", model="google/nano-banana-pro") +""" + +from __future__ import annotations + +import os +from typing import Any + +import httpx +from dotenv import load_dotenv +from eth_account import Account + +from .apikey import ( + api_key_base_url, + auth_headers, + missing_credential_error, + payment_mode, + raise_for_api_key_402, + resolve_api_key, +) +from .jobs import poll_until_completed +from .tx_log import paid_request_error_prefix +from .types import APIError, ImageResponse, PaymentError, retry_after_of +from .validation import ( + build_payment_rejected_error, + sanitize_error_response, + validate_api_url, + validate_private_key, + validate_resource_url, +) +from .x402 import create_payment_payload, extract_payment_details, parse_payment_required + +# Load environment variables +load_dotenv() + + +class ImageClient: + """ + BlockRun Image Generation Client. + + Generate images using Nano Banana (Google Gemini), DALL-E 3, + GPT Image 1, or CogView-4 (Zhipu AI) + with automatic x402 micropayments on Base chain. + """ + + DEFAULT_API_URL = "https://blockrun.ai/api" + DEFAULT_MODEL = "google/nano-banana" + DEFAULT_SIZE = "1024x1024" + + # Image generation slow-path polling. Models like ``openai/gpt-image-2`` + # and ``openai/dall-e-3`` routinely exceed the gateway's 30s inline + # window and come back as ``202 + poll_url`` instead of the finished + # image. The client replays the same PAYMENT-SIGNATURE on every poll; + # settlement only happens on the first completed poll, so giving up + # before then costs the caller nothing. + IMAGE_POLL_INTERVAL_SECONDS = 5.0 + IMAGE_POLL_BUDGET_SECONDS = 300.0 + + def __init__( + self, + private_key: str | None = None, + api_url: str | None = None, + timeout: float = 200.0, # gpt-image-2 at >=1536px can take ~180s server-side; 200s gives buffer + ): + """ + Initialize the BlockRun Image client. + + Args: + private_key: EVM wallet private key (or set BLOCKRUN_WALLET_KEY env var) + api_url: API endpoint URL (default: https://blockrun.ai/api) + timeout: Request timeout in seconds (default: 200 for images — gpt-image-2 at large sizes can run ~180s server-side) + + Raises: + ValueError: If no private key is provided or found in env + """ + # Get private key from param, environment, or ~/.blockrun/.session file + from .wallet import load_wallet + + # Account rail first, and before any wallet variable is read: an API + # key is a complete credential on its own, so demanding a private key + # alongside it would make every API-key user invent a wallet they + # never use. See apikey.py for the precedence rule. + api_key = resolve_api_key(private_key) + key = ( + None + if api_key + else ( + private_key + or os.environ.get("BLOCKRUN_WALLET_KEY") + or os.environ.get("BASE_CHAIN_WALLET_KEY") + or load_wallet() # Loads from ~/.blockrun/.session + ) + ) + if not api_key and not key: + raise missing_credential_error() + + # Validate private key format + if key: + validate_private_key(key) + + # Initialize wallet account (key stays local, never transmitted) + self.api_key = api_key + # No wallet on the account rail: nothing is signed locally. + self.account = Account.from_key(key) if key else None + + # Validate and set API URL + # BLOCKRUN_API_URL names an x402 gateway; an API-key client must not + # follow it and hand the key to a host set up for another rail. + api_url_raw = ( + api_key_base_url(api_url) + if api_key + else (api_url or os.environ.get("BLOCKRUN_API_URL") or self.DEFAULT_API_URL) + ) + validate_api_url(api_url_raw) + self.api_url = api_url_raw.rstrip("/") + + self.timeout = timeout + + # HTTP client + self._client = httpx.Client(headers=auth_headers(api_key), timeout=timeout) + + def generate( + self, + prompt: str, + *, + model: str | None = None, + size: str | None = None, + n: int = 1, + **kwargs: Any, + ) -> ImageResponse: + """ + Generate an image from a text prompt. + + Args: + prompt: Text description of the image to generate + model: Model ID (default: "google/nano-banana") + Options: "google/nano-banana", "google/nano-banana-pro", + "openai/dall-e-3", "openai/gpt-image-1", + "openai/gpt-image-2", "zai/cogview-4", + "xai/grok-imagine-image", "xai/grok-imagine-image-pro", + "black-forest/flux-1.1-pro" + size: Image size (default: "1024x1024") + n: Number of images to generate (default: 1) + + Returns: + ImageResponse with generated image URLs + + Example: + result = client.generate("A sunset over mountains") + print(result.data[0].url) # Image URL or data URL + + Raises: + TypeError: If unexpected keyword arguments are passed + """ + if kwargs: + unsupported = ", ".join(sorted(kwargs.keys())) + hint = ( + " `quality` is Solana-only (SolanaLLMClient.image) — the Base gateway " + "has no such field and would silently ignore it." + if "quality" in kwargs + else "" + ) + raise TypeError( + f"generate() got unexpected keyword argument(s): {unsupported}. " + f"Valid parameters are: prompt, model, size, n.{hint}" + ) + + # Build request body + body: dict[str, Any] = { + "model": model or self.DEFAULT_MODEL, + "prompt": prompt, + "size": size or self.DEFAULT_SIZE, + "n": n, + } + + # Make request (with automatic payment handling) + return self._request_with_payment("/v1/images/generations", body) + + def edit( + self, + prompt: str, + image: str | list[str], + *, + model: str | None = None, + mask: str | None = None, + size: str | None = None, + n: int = 1, + **kwargs: Any, + ) -> ImageResponse: + """ + Edit an image using img2img, or fuse multiple source images. + + Args: + prompt: Text description of the desired edit + image: A single base64 "data:image/...;base64,..." data URI, or a + list of 1-4 such data URIs to fuse multiple sources (e.g. a + reference photo + a brand logo). Plain URLs are not accepted — + the source must be a data URI. + model: Model ID (default: "openai/gpt-image-2") + Edit-supported: "openai/gpt-image-1", "openai/gpt-image-2", + "google/nano-banana", "google/nano-banana-pro" + Multi-image caps: openai/* up to 4, google/* up to 3. + mask: Optional base64-encoded mask image (OpenAI gpt-image-* only; + cannot be combined with multiple source images). + size: Image size (default: "1024x1024") + n: Number of images to generate (default: 1) + + Returns: + ImageResponse with edited image URLs + + Example: + # Single-image edit + result = client.edit( + "Make the sky purple", + image="data:image/png;base64,..." + ) + + # Multi-image fusion (Nano Banana) + result = client.edit( + "Place the logo on the t-shirt", + image=["data:image/png;base64,...", "data:image/png;base64,..."], + model="google/nano-banana", + ) + print(result.data[0].url) + + Raises: + TypeError: If unexpected keyword arguments are passed + """ + if kwargs: + unsupported = ", ".join(sorted(kwargs.keys())) + hint = ( + " `quality` is Solana-only (SolanaLLMClient.image_edit) — the Base " + "gateway has no such field and would silently ignore it." + if "quality" in kwargs + else "" + ) + raise TypeError( + f"edit() got unexpected keyword argument(s): {unsupported}. " + f"Valid parameters are: prompt, image, model, mask, size, n.{hint}" + ) + + body: dict[str, Any] = { + "model": model or "openai/gpt-image-2", + "prompt": prompt, + "image": image, + "size": size or self.DEFAULT_SIZE, + "n": n, + } + if mask is not None: + body["mask"] = mask + + return self._request_with_payment("/v1/images/image2image", body) + + def _request_with_payment(self, endpoint: str, body: dict[str, Any]) -> ImageResponse: + """ + Make a request with automatic x402 payment handling. + + 1. Send initial request + 2. If 402, parse payment requirements + 3. Sign payment locally + 4. Retry with X-Payment header + """ + url = f"{self.api_url}{endpoint}" + + # First attempt (will likely return 402) + response = self._client.post( + url, + json=body, + headers={"Content-Type": "application/json"}, + ) + + # Handle 402 Payment Required + if response.status_code == 402: + # Account rail: a 402 is the account being out of credit, not a + # challenge to sign. Nothing here can sign, so say so plainly. + raise_for_api_key_402(response, self.api_key) + return self._handle_payment_and_retry(url, body, response) + + # Account rail: the key already paid, so a slow model returns its async + # envelope on the FIRST post. The wallet rail only ever sees a 202 after + # the signed retry, so without this branch every slow model raised + # "API error: 202" for API-key callers. + if self.api_key and response.status_code == 202: + return self._poll_until_completed(response, None) + + # Handle other errors + if response.status_code != 200: + try: + error_body = response.json() + except Exception: + error_body = {"error": "Request failed"} + raise APIError( + f"API error: {response.status_code}", + response.status_code, + sanitize_error_response(error_body), + retry_after=retry_after_of(response), + ) + + # Parse successful response + return ImageResponse(**response.json()) + + def _handle_payment_and_retry( + self, + url: str, + body: dict[str, Any], + response: httpx.Response, + ) -> ImageResponse: + """Handle 402 response: parse requirements, sign payment, retry.""" + # Get payment required header (x402 library uses lowercase) + payment_header = response.headers.get("payment-required") + if not payment_header: + # Try to get from response body + try: + resp_body = response.json() + if "x402" in resp_body: + payment_header = resp_body + except Exception: + pass + + if not payment_header: + raise PaymentError("402 response but no payment requirements found") + + # Parse payment requirements + if isinstance(payment_header, str): + payment_required = parse_payment_required(payment_header) + else: + payment_required = payment_header + + # Extract payment details + details = extract_payment_details(payment_required) + + # Create signed payment payload (v2 format) + resource = details.get("resource") or {} + # Pass through extensions from server (for Bazaar discovery) + extensions = payment_required.get("extensions", {}) + payment_payload = create_payment_payload( + account=self.account, + recipient=details["recipient"], + amount=details["amount"], + network=details.get("network", "eip155:8453"), + resource_url=validate_resource_url( + resource.get("url", f"{self.api_url}/v1/images/generations"), self.api_url + ), + resource_description=resource.get("description", "BlockRun Image Generation"), + max_timeout_seconds=details.get("maxTimeoutSeconds", 300), + extra=details.get("extra"), + asset=details.get("asset"), + extensions=extensions, + ) + + # Retry with payment (x402 library expects PAYMENT-SIGNATURE header) + retry_response = self._client.post( + url, + json=body, + headers={ + "Content-Type": "application/json", + "PAYMENT-SIGNATURE": payment_payload, + }, + ) + + # Check for errors + if retry_response.status_code == 402: + # Account rail: a 402 is the account being out of credit, not a + # challenge to sign. Nothing here can sign, so say so plainly. + raise_for_api_key_402(retry_response, self.api_key) + raise build_payment_rejected_error(retry_response) + + if retry_response.status_code == 200: + return ImageResponse(**retry_response.json()) + + if retry_response.status_code == 202: + # Slow-path async flow — gateway returned a job stub with a + # poll_url. Replay the same signature on each poll until the + # upstream finishes; settlement happens on the completed poll. + return self._poll_until_completed(retry_response, payment_payload) + + try: + error_body = retry_response.json() + except Exception: + error_body = {"error": "Request failed"} + raise APIError( + f"{paid_request_error_prefix(retry_response.headers)}: {retry_response.status_code}", + retry_response.status_code, + sanitize_error_response(error_body), + retry_after=retry_after_of(retry_response), + ) + + def _poll_until_completed( + self, + submit_resp: httpx.Response, + payment_payload: str | None, + ) -> ImageResponse: + """Poll the gateway's ``poll_url`` until the upstream returns the + finished image. Settlement happens on the first ``status=completed`` + poll, so timeout = no spend. Shared with music — see jobs.py.""" + data = poll_until_completed( + self._client, + submit_resp, + payment_payload, + api_url=self.api_url, + api_key=self.api_key, + interval_seconds=self.IMAGE_POLL_INTERVAL_SECONDS, + budget_seconds=self.IMAGE_POLL_BUDGET_SECONDS, + label="Image", + ) + return ImageResponse(**data) + + @property + def payment_mode(self) -> str: + """Which rail this client pays on: ``"apikey"`` or ``"wallet"``. + + Worth checking once at startup when both a key and a wallet are + configured in the environment: it is the difference between + spending credit and spending USDC.""" + return payment_mode(self) + + def get_wallet_address(self) -> str: + # No address on the account rail: payment comes from prepaid + # credit, so there is nothing to return but the empty string. + if self.api_key: + return "" + """Get the wallet address being used for payments.""" + return self.account.address + + def close(self): + """Close the HTTP client.""" + self._client.close() + + def __enter__(self): + return self + + def __exit__(self, exc_type, exc_val, exc_tb): + self.close() diff --git a/blockrun_llm/jobs.py b/blockrun_llm/jobs.py new file mode 100644 index 0000000..dccc924 --- /dev/null +++ b/blockrun_llm/jobs.py @@ -0,0 +1,123 @@ +"""Polling for the gateway's async media jobs. + +A slow generation answers ``202`` with a ``poll_url`` and settles on the first +poll that observes ``completed``, so a poll that times out has cost nothing. +Images and music share this loop: the same statuses, the same settlement rule, +the same two rails. One copy, so the two clients cannot drift apart on how a +job ends. +""" + +from __future__ import annotations + +import time +from typing import Any + +import httpx + +from .apikey import raise_for_api_key_402, resolve_poll_url +from .types import APIError, retry_after_of +from .validation import build_payment_rejected_error, sanitize_error_response + + +def absolute_poll_url(url: str, api_url: str, api_key: str | None) -> str: + """Resolve a relative ``poll_url`` against the configured API host. + + Server-returned poll URLs look like ``/api/v1/images/generations/``; + ``api_url`` already ends with ``/api`` on the wallet rail, and the account + rail serves the same route without that prefix. + """ + if url.startswith(("http://", "https://")): + return url + return resolve_poll_url(url, api_url, api_key) + + +def poll_until_completed( + client: httpx.Client, + submit_resp: httpx.Response, + payment_payload: str | None, + *, + api_url: str, + api_key: str | None, + interval_seconds: float, + budget_seconds: float, + label: str, +) -> dict[str, Any]: + """Poll ``poll_url`` until the job completes; return the completed body. + + ``payment_payload`` is the create's PAYMENT-SIGNATURE on the wallet rail + (the job is bound to that wallet and settles against it) and ``None`` on + the account rail, where the key rides on the client's default headers. + ``label`` names the product in errors ("Image", "Music"). + """ + try: + submit_data = submit_resp.json() + except Exception: + submit_data = {} + + poll_url_rel = submit_data.get("poll_url") + job_id = submit_data.get("id") + if not poll_url_rel: + raise APIError("Slow-path 202 missing poll_url", 202, {"response": submit_data}) + + poll_url = absolute_poll_url(poll_url_rel, api_url, api_key) + poll_headers = {"PAYMENT-SIGNATURE": payment_payload} if payment_payload else {} + deadline = time.monotonic() + budget_seconds + last_status = submit_data.get("status", "queued") + + while time.monotonic() < deadline: + time.sleep(interval_seconds) + + poll_resp = client.get(poll_url, headers=poll_headers) + try: + poll_data = poll_resp.json() + except Exception: + poll_data = {} + last_status = poll_data.get("status", last_status) + + if poll_resp.status_code == 402: + # Account rail: a 402 is the account being out of credit, not a + # challenge to sign. Nothing here can sign, so say so plainly. + raise_for_api_key_402(poll_resp, api_key) + # Settlement failed on this poll — surface the gateway reason. + raise build_payment_rejected_error(poll_resp) + + if last_status == "failed": + raise APIError( + f"{label} generation failed upstream: {poll_data.get('error', 'unknown')}", + poll_resp.status_code, + sanitize_error_response(poll_data if isinstance(poll_data, dict) else {}), + retry_after=retry_after_of(poll_resp), + ) + + if poll_resp.status_code == 200 and last_status == "completed": + tx_hash = poll_resp.headers.get("x-payment-receipt") + if tx_hash and "txHash" not in poll_data: + poll_data["txHash"] = tx_hash + return poll_data + + if poll_resp.status_code in (202, 504): + # 202 = still queued/in_progress; 504 = transient upstream + # hiccup. Both retriable inside the budget. + continue + + if poll_resp.status_code != 200: + try: + error_body = poll_resp.json() + except Exception: + error_body = {"error": "Request failed"} + raise APIError( + f"{label} poll failed: HTTP {poll_resp.status_code}", + poll_resp.status_code, + sanitize_error_response(error_body), + retry_after=retry_after_of(poll_resp), + ) + + raise APIError( + ( + f"{label} generation did not complete within {budget_seconds:.0f}s " + f"(last status: {last_status}). Settlement only happens on " + "completion, so no payment was taken." + ), + 504, + {"id": job_id, "last_status": last_status}, + ) diff --git a/blockrun_llm/music.py b/blockrun_llm/music.py new file mode 100644 index 0000000..e94b869 --- /dev/null +++ b/blockrun_llm/music.py @@ -0,0 +1,346 @@ +""" +BlockRun Music Client - Generate music tracks via x402 micropayments. + +SECURITY NOTE - Private Key Handling: +===================================== +Your private key NEVER leaves your machine. Here's what happens: + +1. Key stays local - only used to sign an EIP-712 typed data message +2. Only the SIGNATURE is sent in the PAYMENT-SIGNATURE header +3. BlockRun verifies the signature on-chain via Coinbase CDP facilitator + +Usage: + from blockrun_llm import MusicClient + + client = MusicClient() # Uses BLOCKRUN_WALLET_KEY from env + + # Generate an instrumental track + result = client.generate("upbeat synthwave with neon pads") + print(result.data[0].url) # CDN URL — download within 24h + + # With lyrics + result = client.generate( + "upbeat pop song", + instrumental=False, + lyrics="Hello world, this is my song...", + ) + +Pricing: $0.1575/track +Note: Generated URLs expire in ~24h — download immediately if needed. +""" + +from __future__ import annotations + +import os +from typing import Any + +import httpx +from dotenv import load_dotenv +from eth_account import Account + +from .apikey import ( + api_key_base_url, + auth_headers, + missing_credential_error, + payment_mode, + raise_for_api_key_402, + resolve_api_key, +) +from .jobs import poll_until_completed +from .tx_log import paid_request_error_prefix +from .types import APIError, MusicResponse, PaymentError, retry_after_of +from .validation import ( + sanitize_error_response, + validate_api_url, + validate_private_key, +) +from .x402 import create_payment_payload, extract_payment_details, parse_payment_required + +load_dotenv() + + +class MusicClient: + """ + BlockRun Music Generation Client. + + Generate full-length ~3 minute music tracks using MiniMax Music 2.5+ + with automatic x402 micropayments on Base chain. + + Pricing: $0.1575/track + """ + + DEFAULT_API_URL = "https://blockrun.ai/api" + DEFAULT_MODEL = "minimax/music-2.5+" + DEFAULT_TIMEOUT = 210.0 # music gen takes 1-3 min + # A track takes one to three minutes and the gateway answers 202 + + # poll_url at once, so the wait happens here, poll by poll. Settlement is + # on the completed poll: a budget that runs out has cost nothing. + MUSIC_POLL_INTERVAL_SECONDS = 5.0 + MUSIC_POLL_BUDGET_SECONDS = 300.0 + + def __init__( + self, + private_key: str | None = None, + api_url: str | None = None, + timeout: float = 210.0, + ): + """ + Initialize the BlockRun Music client. + + Args: + private_key: EVM wallet private key (or set BLOCKRUN_WALLET_KEY env var) + api_url: API endpoint URL (default: https://blockrun.ai/api) + timeout: Request timeout in seconds (default: 210 for music generation) + """ + from .wallet import load_wallet + + # Account rail first, and before any wallet variable is read: an API + # key is a complete credential on its own, so demanding a private key + # alongside it would make every API-key user invent a wallet they + # never use. See apikey.py for the precedence rule. + api_key = resolve_api_key(private_key) + key = ( + None + if api_key + else ( + private_key + or os.environ.get("BLOCKRUN_WALLET_KEY") + or os.environ.get("BASE_CHAIN_WALLET_KEY") + or load_wallet() + ) + ) + if not api_key and not key: + raise missing_credential_error() + + if key: + validate_private_key(key) + self.api_key = api_key + # No wallet on the account rail: nothing is signed locally. + self.account = Account.from_key(key) if key else None + + # BLOCKRUN_API_URL names an x402 gateway; an API-key client must not + # follow it and hand the key to a host set up for another rail. + api_url_raw = ( + api_key_base_url(api_url) + if api_key + else (api_url or os.environ.get("BLOCKRUN_API_URL") or self.DEFAULT_API_URL) + ) + validate_api_url(api_url_raw) + self.api_url = api_url_raw.rstrip("/") + + self.timeout = timeout + self._client = httpx.Client(headers=auth_headers(api_key), timeout=timeout) + + def generate( + self, + prompt: str, + *, + model: str | None = None, + instrumental: bool = True, + lyrics: str | None = None, + ) -> MusicResponse: + """ + Generate a music track from a text prompt. + + Takes 1-3 minutes. Returns a CDN URL valid for ~24h. + + Args: + prompt: Music style, mood, or description. + E.g. "upbeat synthwave with neon pads", "chill lo-fi beats", + "epic orchestral film score" + model: Model ID (default: "minimax/music-2.5+") + Options: "minimax/music-2.5+", "minimax/music-2.5" + instrumental: Generate without vocals (default: True) + lyrics: Custom lyrics — cannot be used with instrumental=True + + Returns: + MusicResponse with track URL, duration, and optional lyrics + + Raises: + ValueError: If both instrumental=True and lyrics are provided + PaymentError: If wallet has insufficient balance + APIError: If the API returns an error + + Example: + result = client.generate("chill lo-fi beats with piano") + print(result.data[0].url) # Download this — expires in 24h + + Example with lyrics: + result = client.generate( + "upbeat pop", instrumental=False, + lyrics="Hello world, this is my song..." + ) + """ + if instrumental and lyrics and lyrics.strip(): + raise ValueError("Cannot specify lyrics when instrumental is True") + + body: dict[str, Any] = { + "model": model or self.DEFAULT_MODEL, + "prompt": prompt, + "instrumental": instrumental, + } + if lyrics and lyrics.strip(): + body["lyrics"] = lyrics.strip() + + return self._request_with_payment("/v1/audio/generations", body) + + def _request_with_payment(self, endpoint: str, body: dict[str, Any]) -> MusicResponse: + """Make a request with automatic x402 payment handling.""" + url = f"{self.api_url}{endpoint}" + + response = self._client.post( + url, + json=body, + headers={"Content-Type": "application/json"}, + ) + + if response.status_code == 402: + # Account rail: a 402 is the account being out of credit, not a + # challenge to sign. Nothing here can sign, so say so plainly. + raise_for_api_key_402(response, self.api_key) + return self._handle_payment_and_retry(url, body, response) + + # Account rail: the key already paid, so the job's 202 comes on the + # FIRST post. Music is never fast enough to finish inline, so without + # this branch every API-key music request raised "API error: 202". + if self.api_key and response.status_code == 202: + return self._poll_until_completed(response, None) + + if response.status_code != 200: + try: + error_body = response.json() + except Exception: + error_body = {"error": "Request failed"} + raise APIError( + f"API error: {response.status_code}", + response.status_code, + sanitize_error_response(error_body), + retry_after=retry_after_of(response), + ) + + return MusicResponse(**response.json()) + + def _handle_payment_and_retry( + self, + url: str, + body: dict[str, Any], + response: httpx.Response, + ) -> MusicResponse: + """Handle 402 response: parse requirements, sign payment, retry.""" + payment_header = response.headers.get("payment-required") + if not payment_header: + try: + resp_body = response.json() + if "x402" in resp_body: + payment_header = resp_body + except Exception: + pass + + if not payment_header: + raise PaymentError("402 response but no payment requirements found") + + if isinstance(payment_header, str): + payment_required = parse_payment_required(payment_header) + else: + payment_required = payment_header + + details = extract_payment_details(payment_required) + resource = details.get("resource") or {} + extensions = payment_required.get("extensions", {}) + + payment_payload = create_payment_payload( + account=self.account, + recipient=details["recipient"], + amount=details["amount"], + network=details.get("network", "eip155:8453"), + resource_url=resource.get("url", f"{self.api_url}/v1/audio/generations"), + resource_description=resource.get("description", "BlockRun Music Generation"), + max_timeout_seconds=details.get("maxTimeoutSeconds", 300), + extra=details.get("extra"), + asset=details.get("asset"), + extensions=extensions, + ) + + retry_response = self._client.post( + url, + json=body, + headers={ + "Content-Type": "application/json", + "PAYMENT-SIGNATURE": payment_payload, + }, + ) + + if retry_response.status_code == 402: + # Account rail: a 402 is the account being out of credit, not a + # challenge to sign. Nothing here can sign, so say so plainly. + raise_for_api_key_402(retry_response, self.api_key) + raise PaymentError("Payment was rejected. Check your wallet balance.") + + if retry_response.status_code == 202: + # The signed create is queued: replay the same signature on each + # poll — the job is bound to this wallet — and settle on completion. + return self._poll_until_completed(retry_response, payment_payload) + + if retry_response.status_code != 200: + try: + error_body = retry_response.json() + except Exception: + error_body = {"error": "Request failed"} + raise APIError( + f"{paid_request_error_prefix(retry_response.headers)}: {retry_response.status_code}", + retry_response.status_code, + sanitize_error_response(error_body), + retry_after=retry_after_of(retry_response), + ) + + data = retry_response.json() + # Attach tx hash from response header + tx_hash = retry_response.headers.get("x-payment-receipt") or retry_response.headers.get( + "X-Payment-Receipt" + ) + if tx_hash: + data["txHash"] = tx_hash + + return MusicResponse(**data) + + def _poll_until_completed( + self, submit_resp: httpx.Response, payment_payload: str | None + ) -> MusicResponse: + data = poll_until_completed( + self._client, + submit_resp, + payment_payload, + api_url=self.api_url, + api_key=self.api_key, + interval_seconds=self.MUSIC_POLL_INTERVAL_SECONDS, + budget_seconds=self.MUSIC_POLL_BUDGET_SECONDS, + label="Music", + ) + return MusicResponse(**data) + + @property + def payment_mode(self) -> str: + """Which rail this client pays on: ``"apikey"`` or ``"wallet"``. + + Worth checking once at startup when both a key and a wallet are + configured in the environment: it is the difference between + spending credit and spending USDC.""" + return payment_mode(self) + + def get_wallet_address(self) -> str: + # No address on the account rail: payment comes from prepaid + # credit, so there is nothing to return but the empty string. + if self.api_key: + return "" + """Get the wallet address being used for payments.""" + return self.account.address + + def close(self): + """Close the HTTP client.""" + self._client.close() + + def __enter__(self): + return self + + def __exit__(self, exc_type, exc_val, exc_tb): + self.close() diff --git a/blockrun_llm/phone.py b/blockrun_llm/phone.py new file mode 100644 index 0000000..52db7dd --- /dev/null +++ b/blockrun_llm/phone.py @@ -0,0 +1,367 @@ +""" +BlockRun Phone Client - Twilio-backed phone lookup + number provisioning via x402. + +Endpoints (all under /v1/phone/...): + POST /lookup $0.01 Carrier + line type lookup + POST /lookup/fraud $0.05 Carrier + SIM-swap / call-forwarding fraud signals + POST /numbers/buy $5.00 Provision a US/CA number (30-day lease, bound to wallet) + POST /numbers/renew $5.00 Extend an existing number by 30 days + POST /numbers/list $0.001 List the wallet's active numbers + POST /numbers/release free Release a provisioned number (still goes through x402 + so the backend can identify the wallet) + +After buying a number you can use it as the `from_` caller-ID in VoiceClient.call(). + +Usage: + from blockrun_llm import PhoneClient + + client = PhoneClient() + + # Lookup a number + info = client.lookup("+14155552671") + print(info) + + # Buy a number (US, optional area code) + bought = client.buy_number(country="US", area_code="415") + print(bought["phone_number"], bought["expires_at"]) + + # List your active numbers + print(client.list_numbers()) + + # Renew / release + client.renew_number(bought["phone_number"]) + client.release_number(bought["phone_number"]) + +SECURITY NOTE: your private key never leaves your machine. Only EIP-712 +signatures are sent in the PAYMENT-SIGNATURE header. +""" + +from __future__ import annotations + +import os +from typing import Any + +import httpx +from dotenv import load_dotenv +from eth_account import Account +from typing_extensions import Self + +from .apikey import ( + api_key_base_url, + auth_headers, + missing_credential_error, + payment_mode, + raise_for_api_key_402, + resolve_api_key, +) +from .tx_log import paid_request_error_prefix +from .types import APIError, PaymentError, retry_after_of +from .validation import ( + sanitize_error_response, + validate_api_url, + validate_private_key, +) +from .x402 import ( + create_payment_payload, + extract_payment_details, + parse_payment_required, +) + +load_dotenv() + + +# Mirrors src/lib/twilio.ts PHONE_PRICES on the backend (settled USDC amount). +PHONE_PRICES: dict[str, float] = { + "lookup": 0.01, + "lookup/fraud": 0.05, + "numbers/buy": 5.00, + "numbers/renew": 5.00, + "numbers/list": 0.001, + "numbers/release": 0.0, +} + + +class PhoneClient: + """ + BlockRun Phone Client. + + Wraps the `/v1/phone/*` x402 endpoints. Use this for phone-number lookup + (carrier + fraud) and for provisioning the caller-ID numbers required by + VoiceClient.call(). + """ + + DEFAULT_API_URL = "https://blockrun.ai/api" + DEFAULT_TIMEOUT = 60.0 + + def __init__( + self, + private_key: str | None = None, + api_url: str | None = None, + timeout: float = DEFAULT_TIMEOUT, + ): + from .wallet import load_wallet + + # Account rail first, and before any wallet variable is read: an API + # key is a complete credential on its own, so demanding a private key + # alongside it would make every API-key user invent a wallet they + # never use. See apikey.py for the precedence rule. + api_key = resolve_api_key(private_key) + key = ( + None + if api_key + else ( + private_key + or os.environ.get("BLOCKRUN_WALLET_KEY") + or os.environ.get("BASE_CHAIN_WALLET_KEY") + or load_wallet() + ) + ) + if not api_key and not key: + raise missing_credential_error() + + if key: + validate_private_key(key) + self.api_key = api_key + # No wallet on the account rail: nothing is signed locally. + self.account = Account.from_key(key) if key else None + + # BLOCKRUN_API_URL names an x402 gateway; an API-key client must not + # follow it and hand the key to a host set up for another rail. + api_url_raw = ( + api_key_base_url(api_url) + if api_key + else (api_url or os.environ.get("BLOCKRUN_API_URL") or self.DEFAULT_API_URL) + ) + validate_api_url(api_url_raw) + self.api_url = api_url_raw.rstrip("/") + + self.timeout = timeout + self._client = httpx.Client(headers=auth_headers(api_key), timeout=timeout) + + # ------------------------------------------------------------------ Lookup + + def lookup(self, phone_number: str) -> dict[str, Any]: + """ + Carrier + line-type lookup. ~$0.01. + + Args: + phone_number: E.164 number (e.g. "+14155552671"). + + Returns: + Twilio Lookup payload with carrier, line_type_intelligence, etc. + """ + self._require_e164(phone_number) + return self._request("lookup", {"phoneNumber": phone_number.strip()}) + + def lookup_fraud(self, phone_number: str) -> dict[str, Any]: + """ + Lookup + fraud signals (SIM swap, call forwarding). ~$0.05. + + Args: + phone_number: E.164 number. + + Returns: + Lookup payload including SIM-swap + call-forwarding intelligence. + """ + self._require_e164(phone_number) + return self._request("lookup/fraud", {"phoneNumber": phone_number.strip()}) + + # ------------------------------------------------------------- Provisioning + + def buy_number( + self, + country: str = "US", + area_code: str | None = None, + ) -> dict[str, Any]: + """ + Provision a dedicated phone number for 30 days. $5.00. + + Args: + country: ISO country code, "US" or "CA" (default "US"). + area_code: Optional 3-digit area-code hint. Availability not guaranteed — + the backend falls back to any number in the country if the area + code can't be matched. + + Returns: + Dict with: + - phone_number (str): the E.164 number you now own + - expires_at (str): ISO-8601 expiry (30 days out) + - chain (str): "base" | "solana" + - message (str): human-readable note + - txHash (str, optional): on-chain payment receipt + + Note: payment is settled only after Twilio confirms the purchase, so + failed purchases do NOT charge your wallet. + """ + if country not in ("US", "CA"): + raise ValueError("country must be 'US' or 'CA'") + body: dict[str, Any] = {"country": country} + if area_code is not None: + if not (isinstance(area_code, str) and area_code.isdigit() and len(area_code) == 3): + raise ValueError("area_code must be a 3-digit string, e.g. '415'") + body["areaCode"] = area_code + return self._request("numbers/buy", body) + + def renew_number(self, phone_number: str) -> dict[str, Any]: + """ + Extend an existing provisioned number by 30 days. $5.00. + + Args: + phone_number: E.164 number your wallet owns. + + Returns: + Dict with phone_number, new expires_at, and txHash. + + Raises: + APIError(403): wallet doesn't own this number or it has expired. + """ + self._require_e164(phone_number) + return self._request("numbers/renew", {"phoneNumber": phone_number.strip()}) + + def list_numbers(self) -> dict[str, Any]: + """ + List the wallet's active phone numbers. ~$0.001. + + Returns: + Dict with: + - numbers: list of {phone_number, chain, expires_at, active} + - count: int + - txHash: str + """ + return self._request("numbers/list", {}) + + def release_number(self, phone_number: str) -> dict[str, Any]: + """ + Release a provisioned number back to the Twilio pool. Free, but the + request still flows through x402 so the backend can verify ownership. + + Args: + phone_number: E.164 number your wallet owns. + + Returns: + Dict with {released: True, phone_number}. + """ + self._require_e164(phone_number) + return self._request("numbers/release", {"phoneNumber": phone_number.strip()}) + + # ---------------------------------------------------------------- Internals + + @staticmethod + def _require_e164(value: str) -> None: + if not value or not isinstance(value, str): + raise ValueError("phone_number is required (E.164 format, e.g. '+14155552671')") + v = value.strip() + if not v.startswith("+") or not v[1:].isdigit() or not (8 <= len(v) <= 16): + raise ValueError(f"phone_number must be E.164 (e.g. '+14155552671'), got {value!r}") + + def _request(self, path: str, body: dict[str, Any]) -> dict[str, Any]: + url = f"{self.api_url}/v1/phone/{path}" + response = self._client.post(url, json=body, headers={"Content-Type": "application/json"}) + if response.status_code == 402: + # Account rail: a 402 is the account being out of credit, not a + # challenge to sign. Nothing here can sign, so say so plainly. + raise_for_api_key_402(response, self.api_key) + return self._handle_payment_and_retry(url, body, response) + return self._unwrap(response) + + def _handle_payment_and_retry( + self, + url: str, + body: dict[str, Any], + response: httpx.Response, + ) -> dict[str, Any]: + payment_header: Any = response.headers.get("payment-required") + if not payment_header: + try: + resp_body = response.json() + if "x402" in resp_body or "accepts" in resp_body: + payment_header = resp_body + except Exception: + pass + if not payment_header: + raise PaymentError("402 response but no payment requirements found") + + if isinstance(payment_header, str): + payment_required = parse_payment_required(payment_header) + else: + payment_required = payment_header + + details = extract_payment_details(payment_required) + resource = details.get("resource") or {} + extensions = payment_required.get("extensions", {}) + + payment_payload = create_payment_payload( + account=self.account, + recipient=details["recipient"], + amount=details["amount"], + network=details.get("network", "eip155:8453"), + resource_url=resource.get("url", url), + resource_description=resource.get("description", "BlockRun Phone"), + max_timeout_seconds=details.get("maxTimeoutSeconds", 300), + extra=details.get("extra"), + asset=details.get("asset"), + extensions=extensions, + ) + + retry = self._client.post( + url, + json=body, + headers={ + "Content-Type": "application/json", + "PAYMENT-SIGNATURE": payment_payload, + }, + ) + if retry.status_code == 402: + # Account rail: a 402 is the account being out of credit, not a + # challenge to sign. Nothing here can sign, so say so plainly. + raise_for_api_key_402(retry, self.api_key) + raise PaymentError("Payment was rejected. Check your wallet balance.") + data = self._unwrap(retry, after_payment=True) + tx_hash = retry.headers.get("x-payment-receipt") or retry.headers.get("X-Payment-Receipt") + if tx_hash and isinstance(data, dict): + data.setdefault("txHash", tx_hash) + return data + + @staticmethod + def _unwrap(response: httpx.Response, *, after_payment: bool = False) -> dict[str, Any]: + if response.status_code == 200: + return response.json() + try: + error_body = response.json() + except Exception: + error_body = {"error": "Request failed"} + prefix = paid_request_error_prefix(response.headers) if after_payment else "API error" + raise APIError( + f"{prefix}: {response.status_code}", + response.status_code, + sanitize_error_response(error_body), + retry_after=retry_after_of(response), + ) + + # ------------------------------------------------------------------ Helpers + + @property + def payment_mode(self) -> str: + """Which rail this client pays on: ``"apikey"`` or ``"wallet"``. + + Worth checking once at startup when both a key and a wallet are + configured in the environment: it is the difference between + spending credit and spending USDC.""" + return payment_mode(self) + + def get_wallet_address(self) -> str: + # No address on the account rail: payment comes from prepaid + # credit, so there is nothing to return but the empty string. + if self.api_key: + return "" + """Return the EVM wallet address used for payments.""" + return self.account.address + + def close(self) -> None: + self._client.close() + + def __enter__(self) -> Self: + return self + + def __exit__(self, exc_type, exc_val, exc_tb) -> None: + self.close() diff --git a/blockrun_llm/portrait.py b/blockrun_llm/portrait.py new file mode 100644 index 0000000..d31f46b --- /dev/null +++ b/blockrun_llm/portrait.py @@ -0,0 +1,355 @@ +""" +BlockRun Portrait Client — enroll Virtual Portraits via x402 micropayments. + +A Virtual Portrait is an AI-generated character image registered as a +face/character reference asset. After enrollment ($0.01 USDC, one-time, +no KYC), you get back a `ta_xxxxxxxx` asset id that can be passed as +`real_face_asset_id` to `VideoClient.generate()` on Seedance 2.0 or 2.0-fast +to keep the same character across multiple videos. + +For a *real* person's likeness, use `RealFaceClient` instead — it enrolls +a real face for $0.01 via a brief on-phone liveness check (no KYC) and +yields a `ta_` id usable the same way. Virtual Portraits are for +AI-generated personas, mascots, avatars, and virtual spokespeople. + +SECURITY NOTE - Private Key Handling: +===================================== +Your private key NEVER leaves your machine. The key is used locally to +sign an EIP-3009 USDC transfer authorization; only the signature is +transmitted in the PAYMENT-SIGNATURE header. + +Usage: + from blockrun_llm import PortraitClient + + client = PortraitClient() # Uses BLOCKRUN_WALLET_KEY from env + + portrait = client.enroll( + name="My Spokesperson", + image_url="https://example.com/character.jpg", + ) + print(portrait.asset_id) # ta_abcdef1234567890 + print(portrait.settlement.tx_hash) # 0x9f3a… + + # List wallet's enrolled portraits (free, no payment) + listing = client.list_portraits() + for p in listing.portraits: + print(p.assetId, p.name) +""" + +from __future__ import annotations + +import os +from typing import Any + +import httpx +from dotenv import load_dotenv +from eth_account import Account + +from .apikey import ( + api_key_base_url, + auth_headers, + missing_credential_error, + payment_mode, + raise_for_api_key_402, + resolve_api_key, + wallet_only, +) +from .tx_log import paid_request_error_prefix +from .types import ( + APIError, + PaymentError, + PortraitEnrollment, + PortraitList, + retry_after_of, +) +from .validation import ( + raise_api_error, + sanitize_error_response, + validate_api_url, + validate_private_key, + validate_resource_url, +) +from .x402 import ( + create_payment_payload, + extract_payment_details, + parse_payment_required, +) + +load_dotenv() + + +# Hard limits enforced upstream; mirror locally to fail fast. +_MAX_NAME_LEN = 64 + + +class PortraitClient: + """ + BlockRun Virtual Portrait Client. + + Wraps `POST /v1/portrait/enroll` ($0.01 USDC, one-time) and the free + `GET /v1/wallet/
/portraits` listing endpoint. + + The enrollment endpoint settles AFTER the portrait is successfully + registered upstream, so failed enrollments (content filter, network + error) return 502 with no charge. + """ + + DEFAULT_API_URL = "https://blockrun.ai/api" + ENROLL_ENDPOINT = "/v1/portrait/enroll" + + def __init__( + self, + private_key: str | None = None, + api_url: str | None = None, + timeout: float = 60.0, + ): + """ + Args: + private_key: EVM wallet private key (or set BLOCKRUN_WALLET_KEY). + api_url: API endpoint URL (default https://blockrun.ai/api). + timeout: Per-HTTP-call timeout in seconds. + """ + from .wallet import load_wallet + + # Account rail first, and before any wallet variable is read: an API + # key is a complete credential on its own, so demanding a private key + # alongside it would make every API-key user invent a wallet they + # never use. See apikey.py for the precedence rule. + api_key = resolve_api_key(private_key) + key = ( + None + if api_key + else ( + private_key + or os.environ.get("BLOCKRUN_WALLET_KEY") + or os.environ.get("BASE_CHAIN_WALLET_KEY") + or load_wallet() + ) + ) + if not api_key and not key: + raise missing_credential_error() + + if key: + validate_private_key(key) + self.api_key = api_key + # No wallet on the account rail: nothing is signed locally. + self.account = Account.from_key(key) if key else None + + # BLOCKRUN_API_URL names an x402 gateway; an API-key client must not + # follow it and hand the key to a host set up for another rail. + api_url_raw = ( + api_key_base_url(api_url) + if api_key + else (api_url or os.environ.get("BLOCKRUN_API_URL") or self.DEFAULT_API_URL) + ) + validate_api_url(api_url_raw) + self.api_url = api_url_raw.rstrip("/") + + self.timeout = timeout + self._client = httpx.Client(headers=auth_headers(api_key), timeout=timeout) + + # ------------------------------------------------------------------ + # Enrollment ($0.01 USDC) + # ------------------------------------------------------------------ + + def enroll(self, name: str, image_url: str) -> PortraitEnrollment: + """ + Enroll a Virtual Portrait. Costs $0.01 USDC on Base, one-time. + + Args: + name: Display name (1-64 chars). + image_url: Public `https://` URL pointing to a JPG/PNG/WEBP + image (max 10 MB). Server-side fetched at enrollment time. + + Returns: + PortraitEnrollment with the `ta_xxxxxxxx` asset id, settlement + tx hash, and usage hints. + + Raises: + ValueError: If name or image_url fails local validation. + PaymentError: If wallet balance is insufficient or the payment + is rejected. + APIError: For 4xx/5xx upstream errors (502 = enrollment failed, + no payment was taken — safe to retry with a different image). + """ + if not name or not name.strip(): + raise ValueError("name is required (1-64 chars)") + if len(name) > _MAX_NAME_LEN: + raise ValueError(f"name must be {_MAX_NAME_LEN} chars or fewer (got {len(name)})") + if not image_url or not image_url.lower().startswith(("https://", "http://")): + raise ValueError("image_url must be an http(s) URL") + + body: dict[str, Any] = { + "name": name, + "image_url": image_url, + } + return self._post_with_payment(self.ENROLL_ENDPOINT, body) + + # ------------------------------------------------------------------ + # Listing (free, rate-limited) + # ------------------------------------------------------------------ + + def list_portraits(self, wallet_address: str | None = None) -> PortraitList: + """ + List portraits enrolled by a wallet. Free, but rate-limited to + ~20 requests / hour / IP (shared with the wallet-reconciliation + bucket). + + Args: + wallet_address: Wallet to query. Defaults to the client's own + address. + + Returns: + PortraitList with the wallet address and each portrait's + asset id, name, image url, and enrollment tx hash. + """ + # Keyed by wallet, so the account rail has no default to fall back on: + # say which argument is missing instead of an AttributeError on None. + if not wallet_address and self.account is None: + raise wallet_only("list_portraits") + addr = wallet_address or self.account.address + url = f"{self.api_url}/v1/wallet/{addr}/portraits" + resp = self._client.get(url) + raise_for_api_key_402(resp, self.api_key) + if resp.status_code == 429: + try: + error_body = resp.json() + except Exception: + error_body = {"error": "Rate limit exceeded"} + raise APIError( + "Rate limit exceeded on portrait listing", + resp.status_code, + sanitize_error_response(error_body), + retry_after=retry_after_of(resp), + ) + if resp.status_code != 200: + self._raise_api_error(resp, "Portrait listing failed") + return PortraitList(**resp.json()) + + # ------------------------------------------------------------------ + # Internal: x402 paid POST + # ------------------------------------------------------------------ + + def _post_with_payment(self, endpoint: str, body: dict[str, Any]) -> PortraitEnrollment: + url = f"{self.api_url}{endpoint}" + + resp = self._client.post( + url, + json=body, + headers={"Content-Type": "application/json"}, + ) + + if resp.status_code == 402: + # Account rail: a 402 is the account being out of credit, not a + # challenge to sign. Nothing here can sign, so say so plainly. + raise_for_api_key_402(resp, self.api_key) + return self._handle_payment_and_retry(url, body, resp) + + if resp.status_code != 200: + self._raise_api_error(resp, "Enrollment failed") + + return PortraitEnrollment(**resp.json()) + + def _handle_payment_and_retry( + self, + url: str, + body: dict[str, Any], + response: httpx.Response, + ) -> PortraitEnrollment: + payment_header = response.headers.get("payment-required") + if not payment_header: + try: + resp_body = response.json() + if isinstance(resp_body, dict) and ( + "x402Version" in resp_body or "accepts" in resp_body + ): + payment_header = resp_body + except Exception: + pass + if not payment_header: + raise PaymentError("402 response but no payment requirements found") + + payment_required = ( + parse_payment_required(payment_header) + if isinstance(payment_header, str) + else payment_header + ) + details = extract_payment_details(payment_required) + resource = details.get("resource") or {} + extensions = payment_required.get("extensions", {}) + + payment_payload = create_payment_payload( + account=self.account, + recipient=details["recipient"], + amount=details["amount"], + network=details.get("network", "eip155:8453"), + resource_url=validate_resource_url(resource.get("url", url), self.api_url), + resource_description=resource.get( + "description", "BlockRun Virtual Portrait Enrollment" + ), + max_timeout_seconds=details.get("maxTimeoutSeconds", 300), + extra=details.get("extra"), + asset=details.get("asset"), + extensions=extensions, + ) + + retry = self._client.post( + url, + json=body, + headers={ + "Content-Type": "application/json", + "PAYMENT-SIGNATURE": payment_payload, + }, + ) + + if retry.status_code == 402: + # Account rail: a 402 is the account being out of credit, not a + # challenge to sign. Nothing here can sign, so say so plainly. + raise_for_api_key_402(retry, self.api_key) + raise PaymentError("Payment was rejected. Check your wallet balance.") + + if retry.status_code == 502: + # Enrollment failed upstream — no payment was taken per the spec. + self._raise_api_error( + retry, + "Portrait enrollment failed upstream (no payment taken — safe to retry)", + ) + + if retry.status_code != 200: + self._raise_api_error(retry, f"Enrollment: {paid_request_error_prefix(retry.headers)}") + + return PortraitEnrollment(**retry.json()) + + def _raise_api_error(self, resp: httpx.Response, prefix: str) -> None: + raise_api_error(resp, prefix) + + # ------------------------------------------------------------------ + # Utilities + # ------------------------------------------------------------------ + + @property + def payment_mode(self) -> str: + """Which rail this client pays on: ``"apikey"`` or ``"wallet"``. + + Worth checking once at startup when both a key and a wallet are + configured in the environment: it is the difference between + spending credit and spending USDC.""" + return payment_mode(self) + + def get_wallet_address(self) -> str: + # No address on the account rail: payment comes from prepaid + # credit, so there is nothing to return but the empty string. + if self.api_key: + return "" + """Return the wallet address used for payments.""" + return self.account.address + + def close(self): + """Close the underlying HTTP client.""" + self._client.close() + + def __enter__(self): + return self + + def __exit__(self, exc_type, exc_val, exc_tb): + self.close() diff --git a/blockrun_llm/price.py b/blockrun_llm/price.py new file mode 100644 index 0000000..ad24761 --- /dev/null +++ b/blockrun_llm/price.py @@ -0,0 +1,358 @@ +""" +BlockRun Price Client - Pyth-backed market data via x402. + +Backend endpoints (payment gating mirrors CategoryConfig.paid in +blockrun/src/lib/pyth-handler.ts — crypto/fx/commodity are free across +price+history+list; only usstock and stocks/{market} charge): + + GET /v1/crypto/price/{symbol} (free) + GET /v1/crypto/history/{symbol}?... (free) + GET /v1/crypto/list?q=&limit= (free) + GET /v1/fx/price/{symbol} (free) + GET /v1/fx/history/{symbol}?... (free) + GET /v1/fx/list (free) + GET /v1/commodity/price/{symbol} (free) + GET /v1/commodity/history/{symbol}?... (free) + GET /v1/commodity/list (free) + GET /v1/usstock/price/{symbol} (paid — legacy alias for stocks/us) + GET /v1/usstock/history/{symbol}?... (paid) + GET /v1/usstock/list (free) + GET /v1/stocks/{market}/price/{symbol} (paid — market ∈ {us,hk,jp,kr,gb,de,fr,nl,ie,lu,cn,ca}) + GET /v1/stocks/{market}/history/{symbol} (paid) + GET /v1/stocks/{market}/list (free) + +Usage: + from blockrun_llm import PriceClient + + p = PriceClient() + btc = p.price("crypto", "BTC-USD") + aapl = p.price("stocks", "AAPL", market="us") + bars = p.history("stocks", "AAPL", resolution="D", from_ts=1700000000, to_ts=1710000000, market="us") + symbols = p.list_symbols("crypto", q="sol") +""" + +from __future__ import annotations + +import os +from typing import Any, Literal + +import httpx +from dotenv import load_dotenv +from eth_account import Account +from typing_extensions import Self + +from .apikey import ( + api_key_base_url, + auth_headers, + payment_mode, + raise_for_api_key_402, + resolve_api_key, +) +from .tx_log import paid_request_error_prefix +from .types import ( + APIError, + PaymentError, + PriceHistoryResponse, + PricePoint, + SymbolListResponse, + retry_after_of, +) +from .validation import ( + sanitize_error_response, + validate_api_url, + validate_private_key, +) +from .x402 import create_payment_payload, extract_payment_details, parse_payment_required + +load_dotenv() + +Category = Literal["crypto", "fx", "commodity", "usstock", "stocks"] +Resolution = Literal["1", "5", "15", "60", "240", "D", "W", "M"] +Session = Literal["pre", "post", "on"] +Market = Literal["us", "hk", "jp", "kr", "gb", "de", "fr", "nl", "ie", "lu", "cn", "ca"] + + +class PriceClient: + """ + BlockRun Pyth-backed market data client. + + Free endpoints (crypto/fx/commodity price) work without a wallet but a + wallet is still required at construction time so paid endpoints (stocks, + history) work seamlessly. If you only need free data, set + ``require_wallet=False``. + """ + + DEFAULT_API_URL = "https://blockrun.ai/api" + DEFAULT_TIMEOUT = 30.0 + + def __init__( + self, + private_key: str | None = None, + api_url: str | None = None, + timeout: float = DEFAULT_TIMEOUT, + require_wallet: bool = True, + ): + from .wallet import load_wallet + + # Account rail first, and before any wallet variable is read: an API + # key is a complete credential on its own, so demanding a private key + # alongside it would make every API-key user invent a wallet they + # never use. See apikey.py for the precedence rule. + api_key = resolve_api_key(private_key) + key = ( + None + if api_key + else ( + private_key + or os.environ.get("BLOCKRUN_WALLET_KEY") + or os.environ.get("BASE_CHAIN_WALLET_KEY") + or load_wallet() + ) + ) + if not api_key and not key and require_wallet: + raise ValueError( + "Private key required for paid endpoints. Either:\n" + " 1. Pass private_key parameter\n" + " 2. Set BLOCKRUN_WALLET_KEY environment variable\n" + " 3. Place key in ~/.blockrun/.session\n" + " 4. Pass require_wallet=False if only using free endpoints." + ) + + if key: + validate_private_key(key) + self.api_key = api_key + # No wallet on the account rail: nothing is signed locally. + self.account = Account.from_key(key) if key else None + + # BLOCKRUN_API_URL names an x402 gateway; an API-key client must not + # follow it and hand the key to a host set up for another rail. + api_url_raw = ( + api_key_base_url(api_url) + if api_key + else (api_url or os.environ.get("BLOCKRUN_API_URL") or self.DEFAULT_API_URL) + ) + validate_api_url(api_url_raw) + self.api_url = api_url_raw.rstrip("/") + + self.timeout = timeout + self._client = httpx.Client(headers=auth_headers(api_key), timeout=timeout) + + # ───────── Price ───────── + + def price( + self, + category: Category, + symbol: str, + *, + market: Market | None = None, + session: Session | None = None, + ) -> PricePoint: + """ + Fetch a realtime price quote. + + For ``stocks`` category the ``market`` param is required. + """ + endpoint = self._category_path(category, market, "price", symbol) + params: dict[str, Any] = {} + if session is not None: + params["session"] = session + data = self._get_with_payment(endpoint, params=params) + return PricePoint( + symbol=data.get("symbol", symbol.upper()), + price=data["price"], + publish_time=data.get("publishTime"), + confidence=data.get("confidence"), + feed_id=data.get("feedId"), + **{ + k: v + for k, v in data.items() + if k not in {"symbol", "price", "publishTime", "confidence", "feedId"} + }, + ) + + def history( + self, + category: Category, + symbol: str, + *, + resolution: Resolution = "D", + from_ts: int, + to_ts: int, + market: Market | None = None, + session: Session | None = None, + ) -> PriceHistoryResponse: + """ + Fetch OHLC bars between two Unix timestamps (seconds). + """ + endpoint = self._category_path(category, market, "history", symbol) + params: dict[str, Any] = { + "resolution": resolution, + "from": from_ts, + "to": to_ts, + } + if session is not None: + params["session"] = session + data = self._get_with_payment(endpoint, params=params) + return PriceHistoryResponse( + symbol=data.get("symbol", symbol.upper()), + resolution=data.get("resolution", resolution), + bars=data.get("bars", []), + **{k: v for k, v in data.items() if k not in {"symbol", "resolution", "bars"}}, + ) + + def list_symbols( + self, + category: Category, + *, + q: str | None = None, + limit: int = 100, + market: Market | None = None, + ) -> SymbolListResponse: + """ + List available symbols in a category (free discovery endpoint). + """ + endpoint = self._category_path(category, market, "list", None) + params: dict[str, Any] = {"limit": limit} + if q: + params["q"] = q + data = self._get_with_payment(endpoint, params=params) + # Backend returns either a bare array or an object with "symbols". + if isinstance(data, list): + return SymbolListResponse(symbols=data, count=len(data)) + return SymbolListResponse( + symbols=data.get("symbols", data.get("feeds", [])), + count=data.get("count"), + **{k: v for k, v in data.items() if k not in {"symbols", "feeds", "count"}}, + ) + + # ───────── Internals ───────── + + def _category_path( + self, + category: Category, + market: str | None, + kind: str, + symbol: str | None, + ) -> str: + if category == "stocks": + if not market: + raise ValueError("market is required for category='stocks' (e.g. market='us')") + base = f"/v1/stocks/{market}" + elif category in ("crypto", "fx", "commodity", "usstock"): + base = f"/v1/{category}" + else: + raise ValueError(f"Unknown category: {category}") + if symbol is None: + return f"{base}/{kind}" + return f"{base}/{kind}/{symbol.upper()}" + + def _get_with_payment(self, endpoint: str, *, params: dict[str, Any] | None = None) -> Any: + url = f"{self.api_url}{endpoint}" + response = self._client.get(url, params=params) + if response.status_code == 402: + # Account rail: a 402 is the account being out of credit, not a + # challenge to sign. Nothing here can sign, so say so plainly. + raise_for_api_key_402(response, self.api_key) + if self.account is None: + raise PaymentError( + f"{endpoint} returned 402 Payment Required but no wallet is configured." + ) + return self._pay_and_retry(url, params, response) + if response.status_code != 200: + try: + error_body = response.json() + except Exception: + error_body = {"error": "Request failed"} + raise APIError( + f"API error: {response.status_code}", + response.status_code, + sanitize_error_response(error_body), + retry_after=retry_after_of(response), + ) + return response.json() + + def _pay_and_retry( + self, + url: str, + params: dict[str, Any] | None, + response: httpx.Response, + ) -> Any: + payment_header: Any = response.headers.get("payment-required") + if not payment_header: + try: + resp_body = response.json() + payment_header = resp_body.get("x402") or resp_body + except Exception: + pass + if not payment_header: + raise PaymentError("402 response but no payment requirements found") + + if isinstance(payment_header, str): + payment_required = parse_payment_required(payment_header) + else: + payment_required = payment_header + + details = extract_payment_details(payment_required) + resource = details.get("resource") or {} + extensions = payment_required.get("extensions", {}) + + payment_payload = create_payment_payload( + account=self.account, + recipient=details["recipient"], + amount=details["amount"], + network=details.get("network", "eip155:8453"), + resource_url=resource.get("url", url), + resource_description=resource.get("description", "BlockRun Price Data"), + max_timeout_seconds=details.get("maxTimeoutSeconds", 300), + extra=details.get("extra"), + asset=details.get("asset"), + extensions=extensions, + ) + + retry = self._client.get( + url, + params=params, + headers={"PAYMENT-SIGNATURE": payment_payload}, + ) + if retry.status_code == 402: + # Account rail: a 402 is the account being out of credit, not a + # challenge to sign. Nothing here can sign, so say so plainly. + raise_for_api_key_402(retry, self.api_key) + raise PaymentError("Payment was rejected. Check your wallet balance.") + if retry.status_code != 200: + try: + error_body = retry.json() + except Exception: + error_body = {"error": "Request failed"} + raise APIError( + f"{paid_request_error_prefix(retry.headers)}: {retry.status_code}", + retry.status_code, + sanitize_error_response(error_body), + retry_after=retry_after_of(retry), + ) + return retry.json() + + @property + def payment_mode(self) -> str: + """Which rail this client pays on: ``"apikey"`` or ``"wallet"``. + + Worth checking once at startup when both a key and a wallet are + configured in the environment: it is the difference between + spending credit and spending USDC.""" + return payment_mode(self) + + def get_wallet_address(self) -> str | None: + # No address on the account rail: payment comes from prepaid + # credit, so there is nothing to return but the empty string. + if self.api_key: + return "" + return self.account.address if self.account else None + + def close(self) -> None: + self._client.close() + + def __enter__(self) -> Self: + return self + + def __exit__(self, exc_type, exc_val, exc_tb) -> None: + self.close() diff --git a/blockrun_llm/realface.py b/blockrun_llm/realface.py new file mode 100644 index 0000000..c9b9899 --- /dev/null +++ b/blockrun_llm/realface.py @@ -0,0 +1,526 @@ +""" +BlockRun RealFace Client — enroll a real person's face via x402 micropayments. + +A RealFace registers a *real person's* likeness as a face/character reference +asset. Unlike a Virtual Portrait (AI-generated character, see PortraitClient), +RealFace proves the enroller is the same person in the photo via a brief +on-phone liveness check (nod + blink, ~1 minute). **No KYC** — no government +ID, no account login, just the liveness step. After enrollment ($0.01 USDC, +one-time) you get back a `ta_xxxxxxxx` asset id that can be passed as +`real_face_asset_id` to `VideoClient.generate()` on Seedance 2.0 / 2.0-fast +to keep the same person across multiple videos. + +The flow is three steps: + + 1. init() — FREE. Returns a group_id + an h5_link the real + person scans on their phone. + 2. (phone liveness) — The rights-holder opens h5_link, allows camera, + nods + blinks. ~60 seconds. Nothing goes to BlockRun. + 3. enroll() — $0.01 USDC. Uploads the face photo, matches it + against the live capture, returns the ta_xxx asset. + +Use status() (or the wait_for_active() helper) between steps 2 and 3 to detect +when the person has finished the phone check. + +SECURITY NOTE - Private Key Handling: +===================================== +Your private key NEVER leaves your machine. The key is used locally to +sign an EIP-3009 USDC transfer authorization; only the signature is +transmitted in the PAYMENT-SIGNATURE header. + +Usage: + from blockrun_llm import RealFaceClient + + client = RealFaceClient() # Uses BLOCKRUN_WALLET_KEY from env + + # 1. Start enrollment (free). Render init.h5_link as a QR for the person. + init = client.init(name="Jane — Q3 spokesperson") + print(init.h5_link) # show as QR; they scan + do the liveness check + + # 2. Wait until they finish the phone liveness check (polls status). + client.wait_for_active(init.group_id) + + # 3. Finalize ($0.01 USDC) with the person's face photo. + rf = client.enroll( + name="Jane — Q3 spokesperson", + image_url="https://example.com/jane.jpg", + group_id=init.group_id, + ) + print(rf.asset_id) # ta_abcdef1234567890 + print(rf.settlement.tx_hash) # 0x9f3a… + + # List the wallet's enrolled RealFaces (free, no payment) + listing = client.list_realfaces() + for r in listing.realfaces: + print(r.assetId, r.name) +""" + +from __future__ import annotations + +import os +import re +import time +from typing import Any + +import httpx +from dotenv import load_dotenv +from eth_account import Account + +from .apikey import ( + api_key_base_url, + auth_headers, + missing_credential_error, + payment_mode, + raise_for_api_key_402, + resolve_api_key, + wallet_only, +) +from .tx_log import paid_request_error_prefix +from .types import ( + APIError, + PaymentError, + RealFaceEnrollment, + RealFaceInit, + RealFaceList, + RealFaceStatus, + retry_after_of, +) +from .validation import ( + raise_api_error, + sanitize_error_response, + validate_api_url, + validate_private_key, + validate_resource_url, +) +from .x402 import ( + create_payment_payload, + extract_payment_details, + parse_payment_required, +) + +load_dotenv() + + +# Hard limits enforced upstream; mirror locally to fail fast. +_MAX_NAME_LEN = 64 +# Upstream group ids look like "legacy_rf_8137"; validate to fail fast. +_GROUP_ID_RE = re.compile(r"^legacy_rf_\d+$") + + +class RealFaceClient: + """ + BlockRun RealFace Client. + + Wraps the three-step real-person enrollment flow: + - `POST /v1/realface/init` (free, rate-limited) + - `GET /v1/realface/status` (free, rate-limited) + - `POST /v1/realface/enroll` ($0.01 USDC, one-time) + plus the free `GET /v1/wallet/
/realfaces` listing endpoint. + + The enroll endpoint settles AFTER the asset is successfully matched and + registered upstream, so failed enrollments (group not active, face + mismatch, network error) return an error with no charge. + """ + + DEFAULT_API_URL = "https://blockrun.ai/api" + INIT_ENDPOINT = "/v1/realface/init" + STATUS_ENDPOINT = "/v1/realface/status" + ENROLL_ENDPOINT = "/v1/realface/enroll" + + def __init__( + self, + private_key: str | None = None, + api_url: str | None = None, + timeout: float = 60.0, + ): + """ + Args: + private_key: EVM wallet private key (or set BLOCKRUN_WALLET_KEY). + api_url: API endpoint URL (default https://blockrun.ai/api). + timeout: Per-HTTP-call timeout in seconds. + """ + from .wallet import load_wallet + + # Account rail first, and before any wallet variable is read: an API + # key is a complete credential on its own, so demanding a private key + # alongside it would make every API-key user invent a wallet they + # never use. See apikey.py for the precedence rule. + api_key = resolve_api_key(private_key) + key = ( + None + if api_key + else ( + private_key + or os.environ.get("BLOCKRUN_WALLET_KEY") + or os.environ.get("BASE_CHAIN_WALLET_KEY") + or load_wallet() + ) + ) + if not api_key and not key: + raise missing_credential_error() + + if key: + validate_private_key(key) + self.api_key = api_key + # No wallet on the account rail: nothing is signed locally. + self.account = Account.from_key(key) if key else None + + # BLOCKRUN_API_URL names an x402 gateway; an API-key client must not + # follow it and hand the key to a host set up for another rail. + api_url_raw = ( + api_key_base_url(api_url) + if api_key + else (api_url or os.environ.get("BLOCKRUN_API_URL") or self.DEFAULT_API_URL) + ) + validate_api_url(api_url_raw) + self.api_url = api_url_raw.rstrip("/") + + self.timeout = timeout + self._client = httpx.Client(headers=auth_headers(api_key), timeout=timeout) + + # ------------------------------------------------------------------ + # Step 1: init (free, rate-limited) + # ------------------------------------------------------------------ + + def init(self, name: str, group_id: str | None = None) -> RealFaceInit: + """ + Start (or refresh) a RealFace enrollment. Free, but rate-limited to + ~10 calls / hour / IP (each call creates an upstream session). + + Args: + name: Display name for the asset group (1-64 chars). + group_id: If set, refresh the h5_link for this existing group + instead of creating a new one. Use when the original 120s + H5 session expired before the person finished scanning. + + Returns: + RealFaceInit with `group_id`, the `h5_link` to give the real + person (render as a QR), and `expires_in_seconds`. + + Raises: + ValueError: If name or group_id fails local validation. + APIError: For 4xx/5xx upstream errors (429 = rate limited). + """ + if not name or not name.strip(): + raise ValueError("name is required (1-64 chars)") + if len(name) > _MAX_NAME_LEN: + raise ValueError(f"name must be {_MAX_NAME_LEN} chars or fewer (got {len(name)})") + if group_id is not None and not _GROUP_ID_RE.match(group_id): + raise ValueError("group_id must look like 'legacy_rf_'") + + body: dict[str, Any] = {"name": name} + if group_id: + body["groupId"] = group_id + + url = f"{self.api_url}{self.INIT_ENDPOINT}" + resp = self._client.post(url, json=body, headers={"Content-Type": "application/json"}) + raise_for_api_key_402(resp, self.api_key) + if resp.status_code != 200: + self._raise_api_error(resp, "RealFace init failed") + return RealFaceInit(**resp.json()) + + # ------------------------------------------------------------------ + # Step 2 helper: status / wait_for_active (free, rate-limited) + # ------------------------------------------------------------------ + + def status(self, group_id: str) -> RealFaceStatus: + """ + Poll the state of a RealFace asset group. Free, but rate-limited. + + Args: + group_id: The `legacy_rf_…` id returned by init(). + + Returns: + RealFaceStatus; `ready_to_finalize` is True once the real person + has completed the phone liveness check (status == "active"). + + Raises: + ValueError: If group_id fails local validation. + APIError: For 4xx/5xx upstream errors. + """ + if not group_id or not _GROUP_ID_RE.match(group_id): + raise ValueError("group_id must look like 'legacy_rf_'") + + url = f"{self.api_url}{self.STATUS_ENDPOINT}" + resp = self._client.get(url, params={"groupId": group_id}) + raise_for_api_key_402(resp, self.api_key) + if resp.status_code != 200: + self._raise_api_error(resp, "RealFace status check failed") + return RealFaceStatus(**resp.json()) + + def wait_for_active( + self, + group_id: str, + timeout_seconds: float = 180.0, + poll_interval_seconds: float = 4.0, + ) -> RealFaceStatus: + """ + Block until the group is active (the real person finished the phone + liveness check), then return its status. Convenience wrapper around + repeated status() polling. + + Args: + group_id: The `legacy_rf_…` id returned by init(). + timeout_seconds: Give up after this long (default 180s; the H5 + session itself expires ~120s after each init/refresh). + poll_interval_seconds: Seconds between status checks (default 4s, + matching the studio UI; keep >=3s to respect rate limits). + + Returns: + RealFaceStatus with `ready_to_finalize == True`. + + Raises: + ValueError: If group_id fails local validation. + TimeoutError: If the group is not active within timeout_seconds. + APIError: For 4xx/5xx upstream errors during polling. + """ + if not group_id or not _GROUP_ID_RE.match(group_id): + raise ValueError("group_id must look like 'legacy_rf_'") + if poll_interval_seconds <= 0: + raise ValueError("poll_interval_seconds must be positive") + + deadline = time.monotonic() + timeout_seconds + while True: + state = self.status(group_id) + if state.ready_to_finalize: + return state + if time.monotonic() + poll_interval_seconds >= deadline: + raise TimeoutError( + f"RealFace group {group_id} not active after {timeout_seconds:.0f}s " + f"(last status: {state.status!r}). The person may not have finished the " + f"phone liveness check; call init(group_id=…) to refresh an expired h5_link." + ) + time.sleep(poll_interval_seconds) + + # ------------------------------------------------------------------ + # Step 3: enroll ($0.01 USDC) + # ------------------------------------------------------------------ + + def enroll(self, name: str, image_url: str, group_id: str) -> RealFaceEnrollment: + """ + Finalize a RealFace enrollment. Costs $0.01 USDC on Base, one-time. + + Requires the real person to have already completed the phone liveness + check (group status == "active"; use wait_for_active() to block on it). + + Args: + name: Display name (1-64 chars). + image_url: Public `https://` URL pointing to a JPG/PNG/WEBP photo + of the same person (max 10 MB). Server-side fetched and + matched against the live H5 capture. + group_id: The `legacy_rf_…` id returned by init(). + + Returns: + RealFaceEnrollment with the `ta_xxxxxxxx` asset id, settlement tx + hash, and usage hints. + + Raises: + ValueError: If any argument fails local validation. + PaymentError: If wallet balance is insufficient or the payment + is rejected. + APIError: For upstream errors. No payment is taken on these: + 425 = group not active yet (do the phone check first), + 422 = face did not match the live capture (try a clearer + photo), 502 = upstream upload/status failure. + """ + if not name or not name.strip(): + raise ValueError("name is required (1-64 chars)") + if len(name) > _MAX_NAME_LEN: + raise ValueError(f"name must be {_MAX_NAME_LEN} chars or fewer (got {len(name)})") + if not image_url or not image_url.lower().startswith(("https://", "http://")): + raise ValueError("image_url must be an http(s) URL") + if not group_id or not _GROUP_ID_RE.match(group_id): + raise ValueError("group_id must look like 'legacy_rf_'") + + body: dict[str, Any] = { + "name": name, + "image_url": image_url, + "group_id": group_id, + } + return self._post_with_payment(self.ENROLL_ENDPOINT, body) + + # ------------------------------------------------------------------ + # Listing (free, rate-limited) + # ------------------------------------------------------------------ + + def list_realfaces(self, wallet_address: str | None = None) -> RealFaceList: + """ + List RealFaces enrolled by a wallet. Free, but rate-limited to + ~20 requests / hour / IP (shared with the wallet-reconciliation + bucket). + + Args: + wallet_address: Wallet to query. Defaults to the client's own + address. + + Returns: + RealFaceList with the wallet address and each RealFace's asset id, + name, image url, and enrollment tx hash. + """ + # Keyed by wallet, so the account rail has no default to fall back on: + # say which argument is missing instead of an AttributeError on None. + if not wallet_address and self.account is None: + raise wallet_only("list_realfaces") + addr = wallet_address or self.account.address + url = f"{self.api_url}/v1/wallet/{addr}/realfaces" + resp = self._client.get(url) + if resp.status_code == 429: + try: + error_body = resp.json() + except Exception: + error_body = {"error": "Rate limit exceeded"} + raise APIError( + "Rate limit exceeded on RealFace listing", + resp.status_code, + sanitize_error_response(error_body), + retry_after=retry_after_of(resp), + ) + raise_for_api_key_402(resp, self.api_key) + if resp.status_code != 200: + self._raise_api_error(resp, "RealFace listing failed") + return RealFaceList(**resp.json()) + + # ------------------------------------------------------------------ + # Internal: x402 paid POST + # ------------------------------------------------------------------ + + def _post_with_payment(self, endpoint: str, body: dict[str, Any]) -> RealFaceEnrollment: + url = f"{self.api_url}{endpoint}" + + resp = self._client.post( + url, + json=body, + headers={"Content-Type": "application/json"}, + ) + + if resp.status_code == 402: + # Account rail: a 402 is the account being out of credit, not a + # challenge to sign. Nothing here can sign, so say so plainly. + raise_for_api_key_402(resp, self.api_key) + return self._handle_payment_and_retry(url, body, resp) + + if resp.status_code != 200: + self._raise_api_error(resp, "RealFace enrollment failed") + + return RealFaceEnrollment(**resp.json()) + + def _handle_payment_and_retry( + self, + url: str, + body: dict[str, Any], + response: httpx.Response, + ) -> RealFaceEnrollment: + payment_header = response.headers.get("payment-required") + if not payment_header: + try: + resp_body = response.json() + if isinstance(resp_body, dict) and ( + "x402Version" in resp_body or "accepts" in resp_body + ): + payment_header = resp_body + except Exception: + pass + if not payment_header: + raise PaymentError("402 response but no payment requirements found") + + payment_required = ( + parse_payment_required(payment_header) + if isinstance(payment_header, str) + else payment_header + ) + details = extract_payment_details(payment_required) + resource = details.get("resource") or {} + extensions = payment_required.get("extensions", {}) + + payment_payload = create_payment_payload( + account=self.account, + recipient=details["recipient"], + amount=details["amount"], + network=details.get("network", "eip155:8453"), + resource_url=validate_resource_url(resource.get("url", url), self.api_url), + resource_description=resource.get("description", "BlockRun RealFace Enrollment"), + max_timeout_seconds=details.get("maxTimeoutSeconds", 300), + extra=details.get("extra"), + asset=details.get("asset"), + extensions=extensions, + ) + + retry = self._client.post( + url, + json=body, + headers={ + "Content-Type": "application/json", + "PAYMENT-SIGNATURE": payment_payload, + }, + ) + + if retry.status_code == 402: + # Account rail: a 402 is the account being out of credit, not a + # challenge to sign. Nothing here can sign, so say so plainly. + raise_for_api_key_402(retry, self.api_key) + raise PaymentError("Payment was rejected. Check your wallet balance.") + + if retry.status_code == 425: + # Group is not active — the person hasn't finished the phone + # liveness check. No payment was taken. + self._raise_api_error( + retry, + "RealFace group not active yet — the person must finish the phone " + "liveness check first (no payment taken)", + ) + + if retry.status_code == 422: + # Face did not match the live H5 capture. No payment was taken. + self._raise_api_error( + retry, + "RealFace match failed — the photo did not match the live capture " + "(try a clearer front-facing photo of the same person; no payment taken)", + ) + + if retry.status_code == 502: + # Upstream upload / status failure — no payment was taken. + self._raise_api_error( + retry, + "RealFace enrollment failed upstream (no payment taken — safe to retry)", + ) + + if retry.status_code != 200: + self._raise_api_error( + retry, f"RealFace enrollment: {paid_request_error_prefix(retry.headers)}" + ) + + return RealFaceEnrollment(**retry.json()) + + def _raise_api_error(self, resp: httpx.Response, prefix: str) -> None: + raise_api_error(resp, prefix) + + # ------------------------------------------------------------------ + # Utilities + # ------------------------------------------------------------------ + + @property + def payment_mode(self) -> str: + """Which rail this client pays on: ``"apikey"`` or ``"wallet"``. + + Worth checking once at startup when both a key and a wallet are + configured in the environment: it is the difference between + spending credit and spending USDC.""" + return payment_mode(self) + + def get_wallet_address(self) -> str: + # No address on the account rail: payment comes from prepaid + # credit, so there is nothing to return but the empty string. + if self.api_key: + return "" + """Return the wallet address used for payments.""" + return self.account.address + + def close(self): + """Close the underlying HTTP client.""" + self._client.close() + + def __enter__(self): + return self + + def __exit__(self, exc_type, exc_val, exc_tb): + self.close() diff --git a/blockrun_llm/router.py b/blockrun_llm/router.py new file mode 100644 index 0000000..711dc27 --- /dev/null +++ b/blockrun_llm/router.py @@ -0,0 +1,115 @@ +""" +Smart Router for BlockRun LLM SDK + +Thin compatibility shim over :mod:`blockrun_llm.router_core` — the Python port +of `@blockrun/router-core `_, the +same routing engine the TypeScript SDK and the BlockRun gateway run. + +Routing decisions are local and deterministic (<1ms, no extra model call): the +core classifies the task shape, applies capability constraints as hard filters, +and ranks an ordered candidate portfolio; :mod:`blockrun_llm.router_adapter` +then resolves that ranking against the live catalog. + +Until 1.10.1 this module carried its own hand-maintained tier tables and a +14-dimension scorer. Those have been replaced by the shared core, so tier +configuration now lives in :data:`blockrun_llm.router_core.DEFAULT_ROUTING_CONFIG` +(and, for the SDK-only ``free`` profile, in +:data:`blockrun_llm.router_adapter.FREE_TIERS`). + +Usage: + from blockrun_llm import LLMClient + + client = LLMClient() + result = client.smart_chat("What is 2+2?") + print(result.response) # '4' + print(result.model) # 'google/gemini-3.5-flash' + print(f"Saved {result.routing.savings * 100:.0f}%") +""" + +from __future__ import annotations + +from collections.abc import Mapping + +from .router_adapter import ( + BASE_MINIMUM_PAYMENT_USD, + FREE_TIERS, + ResolvedRoutingDecision, + route_with_catalog, +) +from .router_core import DEFAULT_ROUTING_CONFIG +from .router_core import classify_by_rules as _classify_by_rules +from .router_core.types import ModelPricing, ScoringResult, Tier, TierConfig +from .types import RoutingProfile + +#: Back-compat alias — this module used to define its own decision TypedDict. +RoutingDecision = ResolvedRoutingDecision + +__all__ = [ + "DEFAULT_ROUTING_CONFIG", + "FREE_TIERS", + "ModelPricing", + "ResolvedRoutingDecision", + "RoutingDecision", + "RoutingProfile", + "ScoringResult", + "Tier", + "TierConfig", + "classify_by_rules", + "route", +] + + +def classify_by_rules( + prompt: str, + system_prompt: str | None = None, + estimated_tokens: int | None = None, +) -> ScoringResult: + """Classify a prompt into a tier with the shared 15-dimension scorer. + + ``estimated_tokens`` defaults to the ~4-chars-per-token estimate the router + itself uses. + """ + if estimated_tokens is None: + full_text = f"{system_prompt or ''} {prompt}" + estimated_tokens = -(-len(full_text) // 4) # ceil + return _classify_by_rules( + prompt, system_prompt, estimated_tokens, DEFAULT_ROUTING_CONFIG["scoring"] + ) + + +def route( + prompt: str, + system_prompt: str | None, + max_output_tokens: int, + model_pricing: Mapping[str, ModelPricing], + routing_profile: RoutingProfile = "auto", + *, + minimum_payment_usd: float = BASE_MINIMUM_PAYMENT_USD, +) -> ResolvedRoutingDecision: + """ + Route a request to the cheapest capable model. + + Args: + prompt: User message + system_prompt: Optional system prompt + max_output_tokens: Max tokens to generate + model_pricing: Dict of model_id -> {"input_price": x, "output_price": y, + "flat_price": z}, as built from ``/v1/models`` + routing_profile: "free" | "eco" | "auto" | "premium" + minimum_payment_usd: x402 per-request floor applied to the cost + estimate; defaults to the Base chain's $0.002 + + Returns: + The routing decision: selected ``model``, the ordered ``fallbacks`` + chain, ``tier``, ``confidence``, ``method``, ``reasoning``, cost + metadata, plus the portfolio's ``candidates`` / ``candidate_scores`` / + ``task_type`` when the portfolio strategy ran. + """ + return route_with_catalog( + prompt, + system_prompt, + max_output_tokens, + model_pricing, + routing_profile=routing_profile, + minimum_payment_usd=minimum_payment_usd, + ) diff --git a/blockrun_llm/router_adapter.py b/blockrun_llm/router_adapter.py new file mode 100644 index 0000000..6aaf7c7 --- /dev/null +++ b/blockrun_llm/router_adapter.py @@ -0,0 +1,385 @@ +""" +Host glue between the BlockRun catalog and :mod:`blockrun_llm.router_core`. + +Python port of the TypeScript SDK's ``src/router-adapter.ts``. Router Core is +deliberately product-neutral, so everything BlockRun-specific lives here: + +* catalog id resolution (the router's ``free/*`` namespace vs the gateway's + ``nvidia/*`` ids), +* the x402 per-request payment floors used for cost metadata, +* capacity filtering against the full conversation, not just the last message, +* the SDK-only ``free`` routing profile, which Router Core does not model. +""" + +from __future__ import annotations + +import math +from collections.abc import Mapping, Sequence +from typing import Any + +from .router_core import ( + DEFAULT_MODEL_CAPABILITIES, + DEFAULT_ROUTING_CONFIG, + calculate_model_cost, + filter_candidates_by_capacity, + get_fallback_chain, + route, +) +from .router_core.types import ( + Capacity, + ModelPricing, + RouterOptions, + RoutingConfig, + RoutingDecision, + TierConfig, +) + + +class ResolvedRoutingDecision(RoutingDecision, total=False): + """A Router Core decision resolved against the live BlockRun catalog. + + Adds ``fallbacks`` — the remaining candidates in ranked order, which + ``chat()`` walks when an upstream fails transiently. + """ + + fallbacks: list[str] + + +#: Virtual model ids that select a routing profile instead of a concrete model. +AUTO_ROUTING_PROFILES: Mapping[str, str] = { + "blockrun/auto": "auto", + "blockrun/eco": "eco", + "blockrun/premium": "premium", +} + +# x402 per-request payment floors, used only for cost METADATA (the real charge +# is always the gateway's 402 quote). Free models settle at $0 and are never +# floored. +BASE_MINIMUM_PAYMENT_USD = 0.002 +SOLANA_MINIMUM_PAYMENT_USD = 0.001 + +#: The BlockRun free tier is a gateway concept, not a Router Core profile: the +#: core's tiers rank paid models by task affinity, and its evidence candidates +#: are paid ids. ``routing_profile="free"`` therefore routes on the rules +#: strategy over this tier table, and the adapter additionally drops any +#: candidate the catalog does not price at $0. +#: +#: Refreshed 2026-08-31. Every id below was verified by asking the gateway for +#: it twice and reading back the ``model`` field of the reply, because a 200 is +#: not proof: blockrun server-redirects a retired free id to a live one, so a +#: dead rung answers normally while quietly serving something else. That is the +#: shape that defeats a host's ``/exclude``, and it is why the previous table +#: went stale unnoticed. Substituting on 2026-08-31, hence absent here: +#: ``step-3.7-flash``, ``nemotron-nano-9b-v2``, ``nemotron-nano-12b-v2-vl`` and +#: ``mistral-nemotron`` (retired upstream 2026-08-30), plus ``nemotron-3-ultra-550b`` +#: and ``nemotron-3-nano-omni-30b-a3b-reasoning`` — the latter two still list at +#: $0 in ``/v1/models`` but both answer as ``nemotron-3-nano-30b``. +#: +#: The table is no longer NVIDIA-only: ``cohere/north-mini-code`` and +#: ``poolside/laguna-xs-2.1`` serve at $0 and carry the free coding load. +#: ``gpt-oss-120b/20b`` stay out — proxy-only ids with no catalog price, under +#: the NVIDIA free tier's prompt-retention policy. +FREE_TIERS: dict[str, TierConfig] = { + "SIMPLE": { + # Fastest free model (~121 tok/s), and latency is the only axis that + # separates free rungs — they all cost $0. + "primary": "nvidia/nemotron-3-nano-30b", # 131K ctx + "fallback": [ + "nvidia/nemotron-3.5-lightning", + "nvidia/llama-3.2-11b-vision", + "poolside/laguna-xs-2.1", + ], + }, + "MEDIUM": { + "primary": "nvidia/nemotron-3.5-lightning", # 1M ctx — free tier flagship + "fallback": [ + "nvidia/nemotron-3-nano-30b", + "poolside/laguna-xs-2.1", + "cohere/north-mini-code", + ], + }, + "COMPLEX": { + # Only free model above 256K, so it absorbs long inputs; the vision rung + # behind it absorbs multi-modal ones. + "primary": "nvidia/nemotron-3.5-lightning", + "fallback": [ + "cohere/north-mini-code", # 256K ctx + "nvidia/nemotron-3-nano-30b", + "nvidia/llama-3.2-11b-vision", # free vision + ], + }, + "REASONING": { + "primary": "nvidia/nemotron-3.5-lightning", + "fallback": [ + "nvidia/nemotron-3-nano-30b", + "cohere/north-mini-code", + "poolside/laguna-xs-2.1", + ], + }, +} + + +def build_model_pricing(models: list[dict[str, Any]]) -> dict[str, ModelPricing]: + """Build the router's pricing map from a ``/v1/models`` payload. + + Shared by every client (Base and Solana, sync and async) so the four copies + cannot drift. Rows the catalog marks unavailable are skipped: a model that + cannot serve a request must not win routing, since every call to it would + fail with a non-transient error. + + The catalog uses the nested ``pricing.input`` / ``pricing.output`` shape; + older snapshots used top-level ``inputPrice`` / ``outputPrice``. Both are + accepted so the SDK keeps working through backend transitions. + """ + pricing: dict[str, ModelPricing] = {} + for model in models: + if model.get("available") is False: + continue + model_id = model.get("id", "") + if not model_id: + continue + block = model.get("pricing") or {} + input_price = block.get("input", model.get("inputPrice", model.get("input_price", 0))) + output_price = block.get("output", model.get("outputPrice", model.get("output_price", 0))) + flat_price = block.get("flat", model.get("flatPrice", model.get("flat_price", 0))) + pricing[model_id] = { + "input_price": float(input_price or 0), + "output_price": float(output_price or 0), + "flat_price": float(flat_price or 0), + } + return pricing + + +def routing_profile_for_model(model: str) -> str | None: + """Map a ``blockrun/auto``-style virtual model id to a routing profile.""" + return AUTO_ROUTING_PROFILES.get(model.lower()) + + +def _is_free(pricing: ModelPricing | None) -> bool: + if pricing is None: + return False + return ( + pricing.get("input_price", 0) == 0 + and pricing.get("output_price", 0) == 0 + and not pricing.get("flat_price") + ) + + +def _capacity(model_id: str) -> Capacity | None: + capabilities = DEFAULT_MODEL_CAPABILITIES.get(model_id) + if capabilities is None: + return None + return { + "context_window": capabilities["context_window"], + "max_output": capabilities["max_output_tokens"], + } + + +def _free_config(config: RoutingConfig) -> RoutingConfig: + """A rules-only config whose every profile lands on the free tier table.""" + free_config: RoutingConfig = dict(config) # type: ignore[assignment] + free_config["strategy"] = "rules" + free_config["tiers"] = FREE_TIERS + free_config["eco_tiers"] = FREE_TIERS + free_config["premium_tiers"] = FREE_TIERS + free_config["agentic_tiers"] = FREE_TIERS + # Promotions promote paid models; they must never leak into the free tier. + free_config["promotions"] = [] + return free_config + + +def routing_text(messages: Sequence[Mapping[str, Any]]) -> dict[str, Any]: + """Extract the routing view of a chat transcript. + + Returns ``prompt`` (last user text), ``system_prompt``, ``conversation_chars`` + (the FULL transcript size — capacity checks must see the whole conversation, + not just the last user message) and ``has_vision``. + """ + system_parts = [ + message["content"] + for message in messages + if message.get("role") == "system" and isinstance(message.get("content"), str) + ] + system_prompt = "\n".join(system_parts) or None + + last_user = next( + ( + message["content"] + for message in reversed(list(messages)) + if message.get("role") == "user" and isinstance(message.get("content"), str) + ), + None, + ) + last_text = next( + ( + message["content"] + for message in reversed(list(messages)) + if isinstance(message.get("content"), str) + ), + None, + ) + + conversation_chars = 0 + has_vision = False + for message in messages: + content = message.get("content") + if isinstance(content, str): + conversation_chars += len(content) + elif isinstance(content, list): + for part in content: + if not isinstance(part, Mapping): + continue + if part.get("type") in ("image_url", "image"): + has_vision = True + text = part.get("text") + if isinstance(text, str): + conversation_chars += len(text) + + return { + "prompt": last_user if last_user is not None else (last_text or ""), + "system_prompt": system_prompt, + "conversation_chars": conversation_chars, + "has_vision": has_vision, + } + + +def route_with_catalog( + prompt: str, + system_prompt: str | None, + max_output_tokens: int, + model_pricing: Mapping[str, ModelPricing], + *, + routing_profile: str | None = None, + requires_structured_output: bool = False, + tools: Sequence[Mapping[str, Any]] | None = None, + tool_choice: Any = None, + minimum_payment_usd: float = SOLANA_MINIMUM_PAYMENT_USD, + conversation_chars: int | None = None, + has_vision: bool = False, + config: RoutingConfig | None = None, + now: Any = None, +) -> ResolvedRoutingDecision: + """Route a request and resolve the ranking against the live catalog. + + ``routing_profile`` accepts Router Core's ``"eco" | "auto" | "premium"`` + plus the SDK-only ``"free"``. + """ + tool_list = list(tools or []) + if tool_choice == "none": + requires_tools: bool | None = False + elif tool_choice == "required" or isinstance(tool_choice, Mapping): + requires_tools = True + else: + requires_tools = None + + is_free_profile = routing_profile == "free" + active_config = config or DEFAULT_ROUTING_CONFIG + if is_free_profile: + active_config = _free_config(active_config) + core_profile = None if is_free_profile else routing_profile + + options: RouterOptions = { + "config": active_config, + "model_pricing": model_pricing, + "routing_profile": core_profile, # type: ignore[typeddict-item] + "has_tools": len(tool_list) > 0, + "tool_count": len(tool_list), + "tool_names": [ + tool.get("function", {}).get("name", "") + for tool in tool_list + if isinstance(tool.get("function"), Mapping) + ], + "has_vision": has_vision, + "requires_structured_output": requires_structured_output, + } + if requires_tools is not None: + options["requires_tools"] = requires_tools + if now is not None: + options["now"] = now + + decision = route(prompt, system_prompt, max_output_tokens, options) + + # Turn the ranking into a gateway-callable list. The ranking is trusted + # as-is — including ids withheld from /v1/models (e.g. moonshot/kimi-k2.7), + # which the gateway serves by direct id — with one exception: the router + # names its free tier `free/`, a namespace resolved by ClawRouter's + # proxy. The gateway's ids are `nvidia/`, and an unmapped `free/*` id + # draws a hard 400 (non-transient, so the fallback chain would never + # engage). Map `free/*` to its catalog-listed `nvidia/*` id and drop it when + # there is none (the proxy-only gpt-oss pair). + tier_configs = decision.get("tier_configs") or active_config["tiers"] + ranked = decision.get("candidates") or [ + decision["model"], + *get_fallback_chain(decision["tier"], tier_configs), + ] + callable_models: list[str] = [] + for model_id in ranked: + if not model_id.startswith("free/"): + resolved: str | None = model_id + else: + nvidia_id = f"nvidia/{model_id[5:]}" + resolved = nvidia_id if nvidia_id in model_pricing else None + if resolved and resolved not in callable_models: + callable_models.append(resolved) + + if is_free_profile: + # Belt and braces: the free profile must never emit a billable model, + # even if a host config or promotion smuggles one into the tier table. + free_only = [ + model_id for model_id in callable_models if _is_free(model_pricing.get(model_id)) + ] + if free_only: + callable_models = free_only + + # Capacity check against the FULL conversation, not just the routing prompt + # — an agent transcript can be 100x the last user message, and a context + # overflow is a non-transient 400 the fallback chain won't save. Models + # unknown to the capability snapshot are kept (benefit of the doubt). + estimated_input_tokens = math.ceil( + max(conversation_chars or 0, len(f"{system_prompt or ''} {prompt}")) / 4 + ) + fitting = filter_candidates_by_capacity( + callable_models, estimated_input_tokens, max_output_tokens, _capacity + ) + available_candidates = fitting if fitting else callable_models + + # If nothing survived (a chain of proxy-only free ids), call the router's + # pick as-is so the gateway's real error surfaces rather than an invented + # one here. + model = available_candidates[0] if available_candidates else decision["model"] + + costs = calculate_model_cost( + model, model_pricing, estimated_input_tokens, max_output_tokens, routing_profile + ) + # Free models settle at $0 (no payment is signed) — never floor them up to + # the paid minimum. Detected from the catalog pricing, because Router Core's + # calculate_model_cost applies its own internal floor even to $0 models. + entry = model_pricing.get(model) + is_free = _is_free(entry) if entry is not None else False + cost_estimate = 0.0 if is_free else max(costs["cost_estimate"], minimum_payment_usd) + baseline_cost = costs["baseline_cost"] + if routing_profile == "premium" or baseline_cost <= 0: + savings = 0.0 + elif entry is not None: + savings = max(0.0, (baseline_cost - cost_estimate) / baseline_cost) + else: + savings = decision["savings"] + + resolved_decision: ResolvedRoutingDecision = dict(decision) # type: ignore[assignment] + resolved_decision["baseline_cost"] = baseline_cost + resolved_decision["cost_estimate"] = cost_estimate + resolved_decision["savings"] = savings + resolved_decision["model"] = model + if model != decision["model"]: + resolved_decision["reasoning"] = f"{decision['reasoning']} | catalog fallback: {model}" + resolved_decision["candidates"] = available_candidates + if "candidate_scores" in decision: + resolved_decision["candidate_scores"] = [ + score + for score in decision["candidate_scores"] + if score["model"] in available_candidates + ] + # `fallbacks` is the SDK's runtime retry chain: every remaining candidate in + # ranked order, which chat() walks on a transient upstream failure. + resolved_decision["fallbacks"] = available_candidates[1:] + return resolved_decision diff --git a/blockrun_llm/router_core/__init__.py b/blockrun_llm/router_core/__init__.py new file mode 100644 index 0000000..e27de28 --- /dev/null +++ b/blockrun_llm/router_core/__init__.py @@ -0,0 +1,121 @@ +""" +Router Core — deterministic, constraint-first model routing. + +Python port of `@blockrun/router-core `_ +(upstream commit ``5ee7c23``), the same routing engine the TypeScript SDK and +the BlockRun gateway use. The package is deliberately product-neutral: task +classification, hard capability filtering, portfolio scoring, ordered +fallbacks, and routing configuration. It contains no wallet, gateway client, +proxy server, agent loop, payment handling or telemetry transport — the SDK +supplies those through :mod:`blockrun_llm.router_adapter`. + +Hosts provide request capabilities and current model pricing, and may override +model capability and performance observations without adding a network call on +the routing hot path. + +Usage:: + + from blockrun_llm.router_core import DEFAULT_ROUTING_CONFIG, route + + decision = route(prompt, system_prompt, max_output_tokens, { + "config": DEFAULT_ROUTING_CONFIG, + "model_pricing": pricing, + "has_tools": True, + "requires_tools": True, + }) + +Routing is local and deterministic for identical inputs, configuration, model +metadata, and time. +""" + +from __future__ import annotations + +from .config import DEFAULT_ROUTING_CONFIG +from .model_capabilities import DEFAULT_MODEL_CAPABILITIES +from .model_profiles import HISTORICAL_MODEL_PROFILES, LIVE_MODEL_PROFILES +from .portfolio import PortfolioStrategy, classify_task +from .rules import classify_by_rules +from .selector import ( + calculate_model_cost, + filter_by_exclude_list, + filter_by_tool_calling, + filter_by_vision, + filter_candidates_by_capacity, + get_fallback_chain, + get_fallback_chain_filtered, +) +from .strategy import ( + RouterStrategy, + RulesStrategy, + apply_unavailable_models, + get_strategy, + register_strategy, +) +from .tool_intent import infer_tool_requirement +from .types import ( + Capacity, + ModelCapabilities, + ModelPerformanceProfile, + ModelPricing, + RouterOptions, + RoutingConfig, + RoutingDecision, + RoutingProfile, + TaskType, + Tier, + TierConfig, +) + +# Registered here instead of in strategy.py so PortfolioStrategy can reuse the +# stable RulesStrategy without introducing a module cycle. +register_strategy(PortfolioStrategy()) + + +def route( + prompt: str, + system_prompt: str | None, + max_output_tokens: int, + options: RouterOptions, +) -> RoutingDecision: + """Route a request to the cheapest capable model. + + Delegates to the configured strategy (``PortfolioStrategy`` by default). + """ + strategy = get_strategy(options["config"].get("strategy") or "portfolio") + return strategy.route(prompt, system_prompt, max_output_tokens, options) + + +__all__ = [ + "DEFAULT_MODEL_CAPABILITIES", + "DEFAULT_ROUTING_CONFIG", + "HISTORICAL_MODEL_PROFILES", + "LIVE_MODEL_PROFILES", + "Capacity", + "ModelCapabilities", + "ModelPerformanceProfile", + "ModelPricing", + "PortfolioStrategy", + "RouterOptions", + "RouterStrategy", + "RoutingConfig", + "RoutingDecision", + "RoutingProfile", + "RulesStrategy", + "TaskType", + "Tier", + "TierConfig", + "apply_unavailable_models", + "calculate_model_cost", + "classify_by_rules", + "classify_task", + "filter_by_exclude_list", + "filter_by_tool_calling", + "filter_by_vision", + "filter_candidates_by_capacity", + "get_fallback_chain", + "get_fallback_chain_filtered", + "get_strategy", + "infer_tool_requirement", + "register_strategy", + "route", +] diff --git a/blockrun_llm/router_core/_js.py b/blockrun_llm/router_core/_js.py new file mode 100644 index 0000000..7dc0bc7 --- /dev/null +++ b/blockrun_llm/router_core/_js.py @@ -0,0 +1,82 @@ +""" +Small JavaScript-semantics helpers used by the Router Core port. + +The router is a line-by-line port of ``@blockrun/router-core``. A handful of +JS behaviours differ from their obvious Python equivalents in ways that change +routing output, so they are isolated here instead of being approximated at +each call site: + +* ``Number.prototype.toFixed`` rounds half away from zero on the exact binary + value; Python's format spec rounds half to even. +* ``Date.parse`` accepts a bare ``YYYY-MM-DD`` (UTC midnight) and a trailing + ``Z``; ``datetime.fromisoformat`` before 3.11 accepts neither combination. +* Template literals stringify booleans as ``true`` / ``false``, and the + reasoning strings the router emits are asserted on by hosts and tests. + +Ported regexes are compiled with ``re.ASCII`` so ``\\b``, ``\\w``, ``\\d`` and +``\\s`` keep JavaScript's ASCII-only meaning. Without it a pattern like +``\\b(?:urgent|fast)\\b`` silently stops matching inside CJK text, because +Python treats the surrounding Han characters as word characters while +JavaScript does not. +""" + +from __future__ import annotations + +import math +import re +from datetime import datetime, timezone +from decimal import ROUND_HALF_DOWN, ROUND_HALF_UP, Decimal + +#: Flag set applied to every ported regex (see module docstring). +JS_FLAGS = re.ASCII +JS_FLAGS_I = re.ASCII | re.IGNORECASE + + +def js_regex(pattern: str, *, ignorecase: bool = False, multiline: bool = False) -> re.Pattern[str]: + """Compile ``pattern`` with JavaScript-compatible flag semantics.""" + flags = JS_FLAGS_I if ignorecase else JS_FLAGS + if multiline: + flags |= re.MULTILINE + return re.compile(pattern, flags) + + +def to_fixed(value: float, digits: int) -> str: + """Port of ``Number.prototype.toFixed`` (round half away from zero).""" + if not math.isfinite(value): # NaN / Infinity, which toFixed passes through + return str(value) + quantum = Decimal(1).scaleb(-digits) + # toFixed resolves a tie to the larger integer, which is away from zero for + # positives and toward zero for negatives. + rounding = ROUND_HALF_UP if value >= 0 else ROUND_HALF_DOWN + return str(Decimal(value).quantize(quantum, rounding=rounding)) + + +def js_bool(value: bool) -> str: + """Port of JS template-literal boolean stringification.""" + return "true" if value else "false" + + +def parse_date(value: str) -> datetime | None: + """Port of ``Date.parse`` for the ISO forms the router config uses. + + Returns an aware UTC datetime, or ``None`` when the value is unparseable + (``Date.parse`` yields ``NaN``, which the portfolio scorer treats as "no + observation" rather than propagating a NaN score). + """ + text = value.strip() + if not text: + return None + if text.endswith(("Z", "z")): + text = f"{text[:-1]}+00:00" + try: + parsed = datetime.fromisoformat(text) + except ValueError: + return None + return parsed if parsed.tzinfo is not None else parsed.replace(tzinfo=timezone.utc) + + +def as_utc(value: object | None) -> datetime: + """Normalize a caller-supplied ``now`` to an aware UTC datetime.""" + if isinstance(value, datetime): + return value if value.tzinfo is not None else value.replace(tzinfo=timezone.utc) + return datetime.now(timezone.utc) diff --git a/blockrun_llm/router_core/config.py b/blockrun_llm/router_core/config.py new file mode 100644 index 0000000..dc784fd --- /dev/null +++ b/blockrun_llm/router_core/config.py @@ -0,0 +1,1351 @@ +""" +Default Routing Config + +Python port of ``@blockrun/router-core`` ``config.ts`` (upstream commit +``5ee7c23``, 2026-08-30 — the same pin the TypeScript SDK +bundles). V3.5: every tier rung is a public catalog id. + +All routing parameters as a module constant. Hosts override by passing their +own ``RoutingConfig`` in ``RouterOptions["config"]``. + +Scoring uses 15 weighted dimensions with sigmoid confidence calibration. +Keys are snake_case; ``dimension_weights`` keys stay camelCase because they +are dimension *names* emitted by the classifier, not config fields. +""" + +from __future__ import annotations + +from .types import RoutingConfig + +DEFAULT_ROUTING_CONFIG: RoutingConfig = { + "version": "3.5", + "strategy": "portfolio", + "portfolio": { + "auto": { + "quality": 0.47, + "capability": 0.2, + "cost": 0.18, + "speed": 0.07, + "reliability": 0.03, + "legacy": 0.05, + }, + "eco": { + "quality": 0.36, + "capability": 0.2, + "cost": 0.28, + "speed": 0.1, + "reliability": 0.04, + "legacy": 0.02, + }, + "premium": { + "quality": 0.58, + "capability": 0.2, + "cost": 0.08, + "speed": 0.06, + "reliability": 0.06, + "legacy": 0.02, + }, + "high_stakes_boost": {"quality": 0.08, "reliability": 0.05}, + "latency_sensitive_speed_boost": 0.08, + "affinity_floor_gap": {"auto": 0.1, "eco": 0.22, "premium": 0.05}, + }, + "classifier": { + "llm_model": "google/gemini-2.5-flash", + "llm_max_tokens": 10, + "llm_temperature": 0, + "prompt_truncation_chars": 500, + "cache_ttl_ms": 3_600_000, # 1 hour + }, + "scoring": { + "token_count_thresholds": {"simple": 50, "complex": 500}, + # Multilingual keywords: EN + ZH + JA + RU + DE + ES + PT + KO + AR + "code_keywords": [ + # English + "function", + "class", + "import", + "def", + "SELECT", + "async", + "await", + "const", + "let", + "var", + "return", + "```", + # Chinese + "函数", + "类", + "导入", + "定义", + "查询", + "异步", + "等待", + "常量", + "变量", + "返回", + # Japanese + "関数", + "クラス", + "インポート", + "非同期", + "定数", + "変数", + # Russian + "функция", + "класс", + "импорт", + "определ", + "запрос", + "асинхронный", + "ожидать", + "константа", + "переменная", + "вернуть", + # German + "funktion", + "klasse", + "importieren", + "definieren", + "abfrage", + "asynchron", + "erwarten", + "konstante", + "variable", + "zurückgeben", + # Spanish + "función", + "clase", + "importar", + "definir", + "consulta", + "asíncrono", + "esperar", + "constante", + "variable", + "retornar", + # Portuguese + "função", + "classe", + "importar", + "definir", + "consulta", + "assíncrono", + "aguardar", + "constante", + "variável", + "retornar", + # Korean + "함수", + "클래스", + "가져오기", + "정의", + "쿼리", + "비동기", + "대기", + "상수", + "변수", + "반환", + # Arabic + "دالة", + "فئة", + "استيراد", + "تعريف", + "استعلام", + "غير متزامن", + "انتظار", + "ثابت", + "متغير", + "إرجاع", + ], + "reasoning_keywords": [ + # English + "prove", + "theorem", + "derive", + "step by step", + "chain of thought", + "formally", + "mathematical", + "proof", + "logically", + # Chinese + "证明", + "定理", + "推导", + "逐步", + "思维链", + "形式化", + "数学", + "逻辑", + # Japanese + "証明", + "定理", + "導出", + "ステップバイステップ", + "論理的", + # Russian + "доказать", + "докажи", + "доказательств", + "теорема", + "вывести", + "шаг за шагом", + "пошагово", + "поэтапно", + "цепочка рассуждений", + "рассуждени", + "формально", + "математически", + "логически", + # German + "beweisen", + "beweis", + "theorem", + "ableiten", + "schritt für schritt", + "gedankenkette", + "formal", + "mathematisch", + "logisch", + # Spanish + "demostrar", + "teorema", + "derivar", + "paso a paso", + "cadena de pensamiento", + "formalmente", + "matemático", + "prueba", + "lógicamente", + # Portuguese + "provar", + "teorema", + "derivar", + "passo a passo", + "cadeia de pensamento", + "formalmente", + "matemático", + "prova", + "logicamente", + # Korean + "증명", + "정리", + "도출", + "단계별", + "사고의 연쇄", + "형식적", + "수학적", + "논리적", + # Arabic + "إثبات", + "نظرية", + "اشتقاق", + "خطوة بخطوة", + "سلسلة التفكير", + "رسمياً", + "رياضي", + "برهان", + "منطقياً", + ], + "simple_keywords": [ + # English + "what is", + "define", + "translate", + "hello", + "yes or no", + "capital of", + "how old", + "who is", + "when was", + # Chinese + "什么是", + "定义", + "翻译", + "你好", + "是否", + "首都", + "多大", + "谁是", + "何时", + # Japanese + "とは", + "定義", + "翻訳", + "こんにちは", + "はいかいいえ", + "首都", + "誰", + # Russian + "что такое", + "определение", + "перевести", + "переведи", + "привет", + "да или нет", + "столица", + "сколько лет", + "кто такой", + "когда", + "объясни", + # German + "was ist", + "definiere", + "übersetze", + "hallo", + "ja oder nein", + "hauptstadt", + "wie alt", + "wer ist", + "wann", + "erkläre", + # Spanish + "qué es", + "definir", + "traducir", + "hola", + "sí o no", + "capital de", + "cuántos años", + "quién es", + "cuándo", + # Portuguese + "o que é", + "definir", + "traduzir", + "olá", + "sim ou não", + "capital de", + "quantos anos", + "quem é", + "quando", + # Korean + "무엇", + "정의", + "번역", + "안녕하세요", + "예 또는 아니오", + "수도", + "누구", + "언제", + # Arabic + "ما هو", + "تعريف", + "ترجم", + "مرحبا", + "نعم أو لا", + "عاصمة", + "من هو", + "متى", + ], + "technical_keywords": [ + # English + "algorithm", + "optimize", + "architecture", + "distributed", + "kubernetes", + "microservice", + "database", + "infrastructure", + # Chinese + "算法", + "优化", + "架构", + "分布式", + "微服务", + "数据库", + "基础设施", + # Japanese + "アルゴリズム", + "最適化", + "アーキテクチャ", + "分散", + "マイクロサービス", + "データベース", + # Russian + "алгоритм", + "оптимизировать", + "оптимизаци", + "оптимизируй", + "архитектура", + "распределённый", + "микросервис", + "база данных", + "инфраструктура", + # German + "algorithmus", + "optimieren", + "architektur", + "verteilt", + "kubernetes", + "mikroservice", + "datenbank", + "infrastruktur", + # Spanish + "algoritmo", + "optimizar", + "arquitectura", + "distribuido", + "microservicio", + "base de datos", + "infraestructura", + # Portuguese + "algoritmo", + "otimizar", + "arquitetura", + "distribuído", + "microsserviço", + "banco de dados", + "infraestrutura", + # Korean + "알고리즘", + "최적화", + "아키텍처", + "분산", + "마이크로서비스", + "데이터베이스", + "인프라", + # Arabic + "خوارزمية", + "تحسين", + "بنية", + "موزع", + "خدمة مصغرة", + "قاعدة بيانات", + "بنية تحتية", + ], + "creative_keywords": [ + # English + "story", + "poem", + "compose", + "brainstorm", + "creative", + "imagine", + "write a", + # Chinese + "故事", + "诗", + "创作", + "头脑风暴", + "创意", + "想象", + "写一个", + # Japanese + "物語", + "詩", + "作曲", + "ブレインストーム", + "創造的", + "想像", + # Russian + "история", + "рассказ", + "стихотворение", + "сочинить", + "сочини", + "мозговой штурм", + "творческий", + "представить", + "придумай", + "напиши", + # German + "geschichte", + "gedicht", + "komponieren", + "brainstorming", + "kreativ", + "vorstellen", + "schreibe", + "erzählung", + # Spanish + "historia", + "poema", + "componer", + "lluvia de ideas", + "creativo", + "imaginar", + "escribe", + # Portuguese + "história", + "poema", + "compor", + "criativo", + "imaginar", + "escreva", + # Korean + "이야기", + "시", + "작곡", + "브레인스토밍", + "창의적", + "상상", + "작성", + # Arabic + "قصة", + "قصيدة", + "تأليف", + "عصف ذهني", + "إبداعي", + "تخيل", + "اكتب", + ], + # New dimension keyword lists (multilingual) + "imperative_verbs": [ + # English + "build", + "create", + "implement", + "design", + "develop", + "construct", + "generate", + "deploy", + "configure", + "set up", + # Chinese + "构建", + "创建", + "实现", + "设计", + "开发", + "生成", + "部署", + "配置", + "设置", + # Japanese + "構築", + "作成", + "実装", + "設計", + "開発", + "生成", + "デプロイ", + "設定", + # Russian + "построить", + "построй", + "создать", + "создай", + "реализовать", + "реализуй", + "спроектировать", + "разработать", + "разработай", + "сконструировать", + "сгенерировать", + "сгенерируй", + "развернуть", + "разверни", + "настроить", + "настрой", + # German + "erstellen", + "bauen", + "implementieren", + "entwerfen", + "entwickeln", + "konstruieren", + "generieren", + "bereitstellen", + "konfigurieren", + "einrichten", + # Spanish + "construir", + "crear", + "implementar", + "diseñar", + "desarrollar", + "generar", + "desplegar", + "configurar", + # Portuguese + "construir", + "criar", + "implementar", + "projetar", + "desenvolver", + "gerar", + "implantar", + "configurar", + # Korean + "구축", + "생성", + "구현", + "설계", + "개발", + "배포", + "설정", + # Arabic + "بناء", + "إنشاء", + "تنفيذ", + "تصميم", + "تطوير", + "توليد", + "نشر", + "إعداد", + ], + "constraint_indicators": [ + # English + "under", + "at most", + "at least", + "within", + "no more than", + "o(", + "maximum", + "minimum", + "limit", + "budget", + # Chinese + "不超过", + "至少", + "最多", + "在内", + "最大", + "最小", + "限制", + "预算", + # Japanese + "以下", + "最大", + "最小", + "制限", + "予算", + # Russian + "не более", + "не менее", + "как минимум", + "в пределах", + "максимум", + "минимум", + "ограничение", + "бюджет", + # German + "höchstens", + "mindestens", + "innerhalb", + "nicht mehr als", + "maximal", + "minimal", + "grenze", + "budget", + # Spanish + "como máximo", + "al menos", + "dentro de", + "no más de", + "máximo", + "mínimo", + "límite", + "presupuesto", + # Portuguese + "no máximo", + "pelo menos", + "dentro de", + "não mais que", + "máximo", + "mínimo", + "limite", + "orçamento", + # Korean + "이하", + "이상", + "최대", + "최소", + "제한", + "예산", + # Arabic + "على الأكثر", + "على الأقل", + "ضمن", + "لا يزيد عن", + "أقصى", + "أدنى", + "حد", + "ميزانية", + ], + "output_format_keywords": [ + # English + "json", + "yaml", + "xml", + "table", + "csv", + "markdown", + "schema", + "format as", + "structured", + # Chinese + "表格", + "格式化为", + "结构化", + # Japanese + "テーブル", + "フォーマット", + "構造化", + # Russian + "таблица", + "форматировать как", + "структурированный", + # German + "tabelle", + "formatieren als", + "strukturiert", + # Spanish + "tabla", + "formatear como", + "estructurado", + # Portuguese + "tabela", + "formatar como", + "estruturado", + # Korean + "테이블", + "형식", + "구조화", + # Arabic + "جدول", + "تنسيق", + "منظم", + ], + "reference_keywords": [ + # English + "above", + "below", + "previous", + "following", + "the docs", + "the api", + "the code", + "earlier", + "attached", + # Chinese + "上面", + "下面", + "之前", + "接下来", + "文档", + "代码", + "附件", + # Japanese + "上記", + "下記", + "前の", + "次の", + "ドキュメント", + "コード", + # Russian + "выше", + "ниже", + "предыдущий", + "следующий", + "документация", + "код", + "ранее", + "вложение", + # German + "oben", + "unten", + "vorherige", + "folgende", + "dokumentation", + "der code", + "früher", + "anhang", + # Spanish + "arriba", + "abajo", + "anterior", + "siguiente", + "documentación", + "el código", + "adjunto", + # Portuguese + "acima", + "abaixo", + "anterior", + "seguinte", + "documentação", + "o código", + "anexo", + # Korean + "위", + "아래", + "이전", + "다음", + "문서", + "코드", + "첨부", + # Arabic + "أعلاه", + "أدناه", + "السابق", + "التالي", + "الوثائق", + "الكود", + "مرفق", + ], + "negation_keywords": [ + # English + "don't", + "do not", + "avoid", + "never", + "without", + "except", + "exclude", + "no longer", + # Chinese + "不要", + "避免", + "从不", + "没有", + "除了", + "排除", + # Japanese + "しないで", + "避ける", + "決して", + "なしで", + "除く", + # Russian + "не делай", + "не надо", + "нельзя", + "избегать", + "никогда", + "без", + "кроме", + "исключить", + "больше не", + # German + "nicht", + "vermeide", + "niemals", + "ohne", + "außer", + "ausschließen", + "nicht mehr", + # Spanish + "no hagas", + "evitar", + "nunca", + "sin", + "excepto", + "excluir", + # Portuguese + "não faça", + "evitar", + "nunca", + "sem", + "exceto", + "excluir", + # Korean + "하지 마", + "피하다", + "절대", + "없이", + "제외", + # Arabic + "لا تفعل", + "تجنب", + "أبداً", + "بدون", + "باستثناء", + "استبعاد", + ], + "domain_specific_keywords": [ + # English + "quantum", + "fpga", + "vlsi", + "risc-v", + "asic", + "photonics", + "genomics", + "proteomics", + "topological", + "homomorphic", + "zero-knowledge", + "lattice-based", + # Chinese + "量子", + "光子学", + "基因组学", + "蛋白质组学", + "拓扑", + "同态", + "零知识", + "格密码", + # Japanese + "量子", + "フォトニクス", + "ゲノミクス", + "トポロジカル", + # Russian + "квантовый", + "фотоника", + "геномика", + "протеомика", + "топологический", + "гомоморфный", + "с нулевым разглашением", + "на основе решёток", + # German + "quanten", + "photonik", + "genomik", + "proteomik", + "topologisch", + "homomorph", + "zero-knowledge", + "gitterbasiert", + # Spanish + "cuántico", + "fotónica", + "genómica", + "proteómica", + "topológico", + "homomórfico", + # Portuguese + "quântico", + "fotônica", + "genômica", + "proteômica", + "topológico", + "homomórfico", + # Korean + "양자", + "포토닉스", + "유전체학", + "위상", + "동형", + # Arabic + "كمي", + "ضوئيات", + "جينوميات", + "طوبولوجي", + "تماثلي", + ], + # Agentic task keywords - file ops, execution, multi-step, iterative work + # Pruned: removed overly common words like "then", "first", "run", "test", "build" + "agentic_task_keywords": [ + # English - File operations (clearly agentic) + "read file", + "read the file", + "look at", + "check the", + "open the", + "edit", + "modify", + "update the", + "change the", + "write to", + "create file", + # English - Execution (specific commands only) + "execute", + "deploy", + "install", + "npm", + "pip", + "compile", + # English - Multi-step patterns (specific only) + "after that", + "and also", + "once done", + "step 1", + "step 2", + # English - Iterative work + "fix", + "debug", + "until it works", + "keep trying", + "iterate", + "make sure", + "verify", + "confirm", + # Chinese (keep specific ones) + "读取文件", + "查看", + "打开", + "编辑", + "修改", + "更新", + "创建", + "执行", + "部署", + "安装", + "第一步", + "第二步", + "修复", + "调试", + "直到", + "确认", + "验证", + # Spanish + "leer archivo", + "editar", + "modificar", + "actualizar", + "ejecutar", + "desplegar", + "instalar", + "paso 1", + "paso 2", + "arreglar", + "depurar", + "verificar", + # Portuguese + "ler arquivo", + "editar", + "modificar", + "atualizar", + "executar", + "implantar", + "instalar", + "passo 1", + "passo 2", + "corrigir", + "depurar", + "verificar", + # Korean + "파일 읽기", + "편집", + "수정", + "업데이트", + "실행", + "배포", + "설치", + "단계 1", + "단계 2", + "디버그", + "확인", + # Arabic + "قراءة ملف", + "تحرير", + "تعديل", + "تحديث", + "تنفيذ", + "نشر", + "تثبيت", + "الخطوة 1", + "الخطوة 2", + "إصلاح", + "تصحيح", + "تحقق", + ], + # Dimension weights (sum to 1.0) + "dimension_weights": { + "tokenCount": 0.08, + "codePresence": 0.15, + "reasoningMarkers": 0.18, + "technicalTerms": 0.1, + "creativeMarkers": 0.05, + "simpleIndicators": 0.02, # Reduced from 0.12 to make room for agenticTask + "multiStepPatterns": 0.12, + "questionComplexity": 0.05, + "imperativeVerbs": 0.03, + "constraintCount": 0.04, + "outputFormat": 0.03, + "referenceComplexity": 0.02, + "negationComplexity": 0.01, + "domainSpecificity": 0.02, + "agenticTask": 0.04, # Reduced - agentic signals influence tier selection, not dominate it + }, + # Tier boundaries on weighted score axis + "tier_boundaries": { + "simple_medium": 0.0, + "medium_complex": 0.3, # Raised from 0.18 - prevent simple tasks from reaching expensive COMPLEX tier + "complex_reasoning": 0.5, # Raised from 0.4 - reserve for true reasoning tasks + }, + # Sigmoid steepness for confidence calibration + "confidence_steepness": 12, + # Below this confidence → ambiguous (null tier) + "confidence_threshold": 0.7, + }, + # ─── Tier chains ─── + # + # Catalog refresh 2026-08-29 (V3.5). Every chain below names only models + # the public catalog lists (GET https://blockrun.ai/api/v1/models). Ids the + # gateway withholds (`hidden: true`) — kimi-k2.5/k2.6/k2.7, the grok-4-fast + # and grok-4-1-fast pairs, grok-4-0709, claude-opus-4.6, gemini-3-pro-preview, + # the whole `free/*` namespace — were removed everywhere, including fallback + # rungs, so a routed model is always one a user can find on blockrun.ai/models. + # + # Primaries moved only where portfolio.ts already carries calibration + # evidence for the successor (Sonnet 5 over Sonnet 4.6, GPT-5 Mini for + # agentic MEDIUM, Gemini 3.5 Flash where Kimi K2.7 was). Newcomers with no + # trajectory evidence yet (gemini-3.6-flash, glm-5.3, glm-5.3-flash, + # gpt-5.6-luna, grok-4.3, minimax-m3, qwen3.7-plus) enter as fallback rungs; + # promotion waits for a calibration run, because version recency is not a + # quality signal. + # + # Latency figures in comments are the 2026-08-29 gateway probe + # (model-profiles.generated.json); prices are the catalog list. + # Auto (balanced) tier configs - current default smart routing + "tiers": { + "SIMPLE": { + "primary": "google/gemini-2.5-flash", # $0.30/$2.50 — 60% retention (best) in the 2026-03 run; still the fastest quality answer + "fallback": [ + "google/gemini-3-flash-preview", # $0.50/$3 — GPQA 5/6 in the 2026-07 calibration + "google/gemini-3.5-flash-lite", # $0.30/$2.50, 1M ctx, thinking mode — same price as 2.5 Flash, newer generation + "deepseek/deepseek-chat", # $0.14/$0.28, 1M ctx + "google/gemini-3.1-flash-lite", # $0.25/$1.50, 1M ctx + "openai/gpt-5.6-luna", # $0.20/$1.20, 1M ctx — GPT-5.6 cost tier (cut 2026-07-30) + "openai/gpt-5.4-nano", # $0.20/$1.25, 1M ctx + "google/gemini-2.5-flash-lite", # $0.10/$0.40 + "nvidia/nemotron-3.5-lightning", # FREE backstop — NVIDIA free tier (probed 2026-08-30) + ], + }, + "MEDIUM": { + # Was moonshot/kimi-k2.7 (hidden 2026-08). Gemini 3.5 Flash is the + # calibrated successor: MGSM 5/5, GPQA 4/6, extraction band (portfolio.ts). + "primary": "google/gemini-3.5-flash", # $1.50/$9, 1M ctx, vision + tools + "fallback": [ + "google/gemini-3.6-flash", # $1.50/$7.50 — newest Flash, output 17% cheaper than 3.5; awaiting calibration + "zai/glm-5.3-flash", # $0.15/$0.50, 1M ctx, vision + tools verified live 2026-08-27 + "openai/gpt-5.6-terra", # $2/$12, 1M ctx — GPT-5.6 balanced tier + "google/gemini-3-flash-preview", # $0.50/$3 + "deepseek/deepseek-chat", # $0.14/$0.28 + "google/gemini-2.5-flash", # $0.30/$2.50 + "minimax/minimax-m3", # $0.30/$1.20, 1M ctx + "google/gemini-3.1-flash-lite", # $0.25/$1.50 + "openai/gpt-5.6-luna", # $0.20/$1.20 + "google/gemini-2.5-flash-lite", # $0.10/$0.40 + ], + }, + "COMPLEX": { + "primary": "google/gemini-3.1-pro", # $2/$12 — proven long-context flagship (portfolio.ts long_context lead) + "fallback": [ + "google/gemini-3.6-flash", # $1.50/$7.50 — Pro-level quality at Flash price (Google's claim; uncalibrated here) + "google/gemini-3.5-flash", # $1.50/$9 — calibrated + "anthropic/claude-sonnet-5", # $3/$15 — near-Opus quality, tau2 + Terminal-Bench calibrated + "xai/grok-4.5", # $2.50/$9 — 503-resistant, independent infra (was grok-4-0709, now hidden) + "google/gemini-2.5-pro", # $1.25/$10 + "anthropic/claude-sonnet-4.6", # $3/$15 + "openai/gpt-5.6-terra", # $2/$12 — GPT-5.6 balanced tier (Sol excluded: #202) + "openai/gpt-5.5", # $5/$30 — prior OpenAI flagship + "openai/gpt-5.4", # $2.50/$15 — previous flagship, benchmarked + "zai/glm-5.3", # $1.40/$4.40, 1M ctx, always-on thinking — verified live 2026-08-19 + "moonshot/kimi-k3", # $3/$15, 1M ctx — Moonshot flagship (K2.7 successor) + "deepseek/deepseek-v4-pro", # $0.435/$0.87 — strongest open-weight reasoner + "deepseek/deepseek-chat", # $0.14/$0.28 — cheap last resort + "google/gemini-2.5-flash", # $0.30/$2.50 + ], + }, + "REASONING": { + # Was xai/grok-4-1-fast-reasoning ($0.20/$0.50, hidden 2026-08). DeepSeek + # Reasoner is the cheapest listed reasoner at the same 1M context. + "primary": "deepseek/deepseek-reasoner", # $0.14/$0.28, 1M ctx + "fallback": [ + "deepseek/deepseek-v4-pro", # $0.435/$0.87 — calibrated reasoning band 0.95 + "xai/grok-4.3", # $1.50/$4, 1M ctx — xAI reasoning model, vision + "qwen/qwen3.7-plus", # $0.32/$1.28, 1M ctx — reasoning; needs a generous max_tokens (thinking is billed) + "google/gemini-3.5-flash", # $1.50/$9 — MGSM 5/5 + "openai/o4-mini", # $1.10/$4.40 + "openai/o3", # $2/$8 + ], + }, + }, + # Eco tier configs - absolute cheapest (blockrun/eco) + "eco_tiers": { + "SIMPLE": { + "primary": "nvidia/nemotron-3.5-lightning", # FREE — NVIDIA free tier flagship, 1M ctx + "fallback": [ + "nvidia/nemotron-3-nano-30b", # FREE — fastest free model (~121 tok/s) + # The free head keeps rotting with NVIDIA's hosting (deepseek-v4-flash + # 410 2026-08-12, seed-oss-36b 410 2026-08-03, gpt-oss-120b/20b 400 + # 2026-08-21, and on 2026-08-30 FOUR of the five visible free models at + # once — step-3.7-flash, nemotron-nano-9b-v2 and nemotron-nano-12b-v2-vl + # all 410, mistral-nemotron hung). Each retirement retargets the two + # free rungs to the current free tier; the paid rungs below never move. + # The head follows blockrun's own redirect of the model it replaces, so + # the router and the gateway never name different models. + "google/gemini-2.5-flash-lite", # $0.10/$0.40 — cheapest paid rung + "zai/glm-5.3-flash", # $0.15/$0.50, 1M ctx, vision + tools + "openai/gpt-5.6-luna", # $0.20/$1.20, 1M ctx + "openai/gpt-5.4-nano", # $0.20/$1.25 + "google/gemini-3.1-flash-lite", # $0.25/$1.50 + ], + }, + "MEDIUM": { + "primary": "zai/glm-5.3-flash", # $0.15/$0.50, 1M ctx, vision + tools verified live — cheapest full-capability model + "fallback": [ + "deepseek/deepseek-chat", # $0.14/$0.28 + "google/gemini-3.1-flash-lite", # $0.25/$1.50 + "openai/gpt-5.6-luna", # $0.20/$1.20 + "openai/gpt-5.4-nano", # $0.20/$1.25 + "google/gemini-2.5-flash-lite", # $0.10/$0.40 + "google/gemini-2.5-flash", # $0.30/$2.50 + ], + }, + "COMPLEX": { + "primary": "zai/glm-5.3-flash", # $0.15/$0.50, 1M ctx + "fallback": [ + "deepseek/deepseek-chat", # $0.14/$0.28, 1M ctx + "minimax/minimax-m3", # $0.30/$1.20, 1M ctx + "deepseek/deepseek-v4-pro", # $0.435/$0.87 + "google/gemini-3.1-flash-lite", # $0.25/$1.50 + "google/gemini-2.5-flash", # $0.30/$2.50 + ], + }, + "REASONING": { + "primary": "deepseek/deepseek-reasoner", # $0.14/$0.28, 1M ctx — cheapest listed reasoner + "fallback": [ + "deepseek/deepseek-v4-pro", # $0.435/$0.87 + "qwen/qwen3.7-plus", # $0.32/$1.28 — reasoning + "minimax/minimax-m3", # $0.30/$1.20 — reasoning + coding + "zai/glm-5.3-flash", # $0.15/$0.50 — reasoning tokens alongside content + ], + }, + }, + # Premium tier configs - best quality (blockrun/premium) + # codex=complex coding, flash=simple coding, sonnet=reasoning/instructions, fable/opus=architecture/PM/audits + "premium_tiers": { + "SIMPLE": { + # Was moonshot/kimi-k2.7 (hidden 2026-08). + "primary": "google/gemini-3.5-flash", # $1.50/$9, 1M ctx, vision + tools — calibrated + "fallback": [ + "google/gemini-3.6-flash", # $1.50/$7.50 — newest Flash + "anthropic/claude-haiku-4.5", # $1/$5 + "zai/glm-5.3", # $1.40/$4.40, 1M ctx + "google/gemini-2.5-flash", # $0.30/$2.50 + "google/gemini-3.5-flash-lite", # $0.30/$2.50 + "deepseek/deepseek-chat", # $0.14/$0.28 + ], + }, + "MEDIUM": { + "primary": "openai/gpt-5.3-codex", # $1.75/$14 - 400K context, 128K output — code_edit/debug lead (portfolio.ts) + "fallback": [ + "anthropic/claude-sonnet-5", # $3/$15 — code_agent band 0.98 + "moonshot/kimi-k3", # $3/$15, 1M ctx — Moonshot flagship + "zai/glm-5.3", # $1.40/$4.40 — long-horizon coding + "google/gemini-3.6-flash", # $1.50/$7.50 + "google/gemini-3.5-flash", # $1.50/$9 + "google/gemini-2.5-pro", # $1.25/$10 + "xai/grok-4.5", # $2.50/$9 + "anthropic/claude-sonnet-4.6", # $3/$15 + "openai/gpt-5.6-terra", # $2/$12 + ], + }, + "COMPLEX": { + # fable-5 was promoted here 2026-06-11, force-reverted 2026-06-13 when Anthropic + # withdrew the offer, and restored 2026-07-14 now that BlockRun has relisted it. + "primary": "anthropic/claude-fable-5", # Best quality for complex tasks — Mythos-class flagship above Opus ($10/$50, 1M ctx, always-on thinking) + # Fallback chain de-Gemini'd 2026-04-22: when Anthropic 503s, Gemini is + # also prone to "high demand" 503s (correlated failure — everyone falls + # back to Google at the same time). Prefer in-family → xAI → Moonshot → + # OpenAI flagship → Z.AI → DeepSeek → NVIDIA free instead. + "fallback": [ + "anthropic/claude-opus-5", # in-family hot swap first (half the price, 1M ctx + adaptive thinking) + "anthropic/claude-opus-4.8", # in-family hot swap (identical cost to 5) + "anthropic/claude-opus-4.7", # in-family hot swap (identical cost to 4.8) + "anthropic/claude-sonnet-5", # Sonnet-tier drop-down, near-Opus quality + "anthropic/claude-sonnet-4.6", + "xai/grok-4.5", # xAI flagship — 503-resistant, direct-xAI SKU + "moonshot/kimi-k3", # Moonshot flagship, independent infra + "openai/gpt-5.6-terra", # GPT-5.6 balanced tier — stable (Sol excluded: #202) + "openai/gpt-5.5", # Prior OpenAI flagship — 1M+ ctx, native agent + computer use + "openai/gpt-5.4", # Previous flagship (slow but stable, benchmarked at 6,213ms) + "openai/gpt-5.3-codex", + "zai/glm-5.3", # Z.AI flagship, 1M ctx + "deepseek/deepseek-v4-pro", # strongest open-weight reasoner + "deepseek/deepseek-chat", # Cheap, reliable + "nvidia/nemotron-3.5-lightning", # NVIDIA free ultimate backstop + ], + }, + "REASONING": { + # Sonnet 5 promoted over Sonnet 4.6 (same price; reasoning band 0.98 for both, + # plus Sonnet 5's tau2/BrowseComp trajectory evidence). + "primary": "anthropic/claude-sonnet-5", # $3/$15, 1M ctx, adaptive thinking + "fallback": [ + "anthropic/claude-sonnet-4.6", # in-family hot swap — same cost + "anthropic/claude-opus-5", # Newest flagship Opus w/ adaptive thinking + "anthropic/claude-opus-4.8", # Prior flagship Opus — identical cost to 5 + "anthropic/claude-opus-4.7", # Flagship Opus w/ adaptive thinking + "xai/grok-4.5", # reasoning band 0.94 + "deepseek/deepseek-v4-pro", # reasoning band 0.95 + "xai/grok-4.3", # $1.50/$4 — xAI reasoning model + "openai/o4-mini", # $1.10/$4.40 + "openai/o3", # $2/$8 + ], + }, + }, + # Agentic tier configs - models that excel at multi-step autonomous tasks + "agentic_tiers": { + "SIMPLE": { + "primary": "openai/gpt-4o-mini", # $0.15/$0.60 - best tool compliance at lowest cost + "fallback": [ + "openai/gpt-5.6-luna", # $0.20/$1.20 — lightweight agentic tier of GPT-5.6 + "zai/glm-5.3-flash", # $0.15/$0.50 — tool calls verified live 2026-08-27 + "anthropic/claude-haiku-4.5", # $1/$5 + "google/gemini-2.5-flash", # $0.30/$2.50 + ], + }, + "MEDIUM": { + # Was moonshot/kimi-k2.7 (hidden 2026-08). GPT-5 Mini carries the + # Terminal-Bench and tau2 trajectory evidence in portfolio.ts. + "primary": "openai/gpt-5-mini", # $0.25/$2 — 4/7 Terminal-Bench, 5/6 tau2 airline + "fallback": [ + "google/gemini-3.5-flash", # $1.50/$9 — tool_agent band 0.88 + "zai/glm-5.3-flash", # $0.15/$0.50 — tools verified + "openai/gpt-5.6-terra", # $2/$12 + "openai/gpt-4o-mini", # $0.15/$0.60 — reliable tool calling + "anthropic/claude-haiku-4.5", # $1/$5 + "deepseek/deepseek-chat", # $0.14/$0.28 + "moonshot/kimi-k3", # $3/$15 — tool_agent band 0.85 + ], + }, + "COMPLEX": { + # Sonnet 5 promoted over Sonnet 4.6: tau2 airline + retail reward 1.0, + # Terminal-Bench safety band lead (portfolio.ts). + "primary": "anthropic/claude-sonnet-5", # $3/$15 — best agentic quality per trajectory evidence + # Fallback chain de-Gemini'd 2026-04-22: Gemini's "high demand" 503s + # correlate with Anthropic outages (everyone falls back together). + # Prefer 503-resistant providers first. + "fallback": [ + "anthropic/claude-sonnet-4.6", # in-family hot swap — same cost + "anthropic/claude-opus-5", # Newest flagship Opus — in-family hot swap + "anthropic/claude-opus-4.8", # Prior flagship Opus — identical cost to 5 + "anthropic/claude-opus-4.7", # Flagship Opus — in-family hot swap + "xai/grok-4.5", # xAI flagship — strong tool use, independent infra + "moonshot/kimi-k3", # Moonshot flagship — independent infra + "openai/gpt-5.6-terra", # GPT-5.6 balanced tier — stable (Sol excluded: #202) + "openai/gpt-5.5", # Prior flagship — native agent + computer use (exactly the agentic-tier use case) + "openai/gpt-5.4", # Previous flagship — reliable + "openai/gpt-5.3-codex", # code_agent lead + "zai/glm-5.3", # long-horizon coding + "deepseek/deepseek-v4-pro", # retail high-risk 3/3 + "deepseek/deepseek-chat", # cheap, reliable + "nvidia/nemotron-3.5-lightning", # NVIDIA free ultimate backstop + ], + }, + "REASONING": { + "primary": "anthropic/claude-sonnet-5", # $3/$15 — strong tool use + adaptive thinking + "fallback": [ + "anthropic/claude-sonnet-4.6", # in-family hot swap — same cost + "anthropic/claude-opus-5", # Newest flagship Opus w/ adaptive thinking + "anthropic/claude-opus-4.8", # Prior flagship Opus — identical cost to 5 + "anthropic/claude-opus-4.7", # Flagship Opus w/ adaptive thinking + "xai/grok-4.5", # reasoning band 0.94 + "deepseek/deepseek-v4-pro", # reasoning band 0.95 + "deepseek/deepseek-reasoner", # $0.14/$0.28 + ], + }, + }, + # Time-windowed promotions — auto-applied when active, ignored when expired. + # The GLM-5.1 launch promo (2026-04-01 → 2026-05-01) was the last entry and + # has expired; the list is kept empty so the mechanism stays wired. + "promotions": [], + "overrides": { + "max_tokens_force_complex": 100_000, + "structured_output_min_tier": "MEDIUM", + "ambiguous_default_tier": "MEDIUM", + # agenticMode left undefined → auto-detect via tools/agenticScore. + # Set to `true` to force agentic tiers; `false` to disable them entirely. + }, +} diff --git a/blockrun_llm/router_core/model_capabilities.py b/blockrun_llm/router_core/model_capabilities.py new file mode 100644 index 0000000..9228ce0 --- /dev/null +++ b/blockrun_llm/router_core/model_capabilities.py @@ -0,0 +1,475 @@ +""" +Model capabilities used for hard routing constraints. + +Python port of ``@blockrun/router-core`` ``model-capabilities.ts``. + +Hosts may inject fresher values through ``RouterOptions["model_capabilities"]``. +Keeping a small built-in snapshot makes the core safe and useful when a +product catalog is temporarily unavailable, without importing product code. + +GENERATED upstream by ``scripts/sync-model-capabilities.mjs`` from the public +catalog (GET https://blockrun.ai/api/v1/models) on 2026-08-31; ``supports_tools`` +comes from a live function-calling probe. Re-sync from ``model-capabilities.ts`` +rather than editing by hand — a hand edit is lost on the next sync. +""" + +from __future__ import annotations + +from collections.abc import Mapping +from types import MappingProxyType + +from .types import ModelCapabilities + +DEFAULT_MODEL_CAPABILITIES: Mapping[str, ModelCapabilities] = MappingProxyType( + { + "anthropic/claude-fable-5": { + "context_window": 1_000_000, + "max_output_tokens": 128_000, + "supports_tools": True, + "supports_vision": True, + }, + # override: The public catalog's `categories` omit "vision" for this Anthropic model even + # though the gateway accepts image input for it (the prior hand-maintained snapshot had + # it, and Anthropic's model card lists it). Without this the vision filter would silently + # drop it — reported against the catalog; remove once the categories carry vision. + "anthropic/claude-haiku-4.5": { + "context_window": 200_000, + "max_output_tokens": 64_000, + "supports_tools": True, + "supports_vision": True, + }, + "anthropic/claude-opus-4.5": { + "context_window": 200_000, + "max_output_tokens": 64_000, + "supports_tools": True, + "supports_vision": True, + }, + "anthropic/claude-opus-4.7": { + "context_window": 1_000_000, + "max_output_tokens": 128_000, + "supports_tools": True, + "supports_vision": True, + }, + "anthropic/claude-opus-4.8": { + "context_window": 1_000_000, + "max_output_tokens": 128_000, + "supports_tools": True, + "supports_vision": True, + }, + "anthropic/claude-opus-5": { + "context_window": 1_000_000, + "max_output_tokens": 128_000, + "supports_tools": True, + "supports_vision": True, + }, + "anthropic/claude-sonnet-4.5": { + "context_window": 200_000, + "max_output_tokens": 64_000, + "supports_tools": True, + "supports_vision": True, + }, + # override: The public catalog's `categories` omit "vision" for this Anthropic model even + # though the gateway accepts image input for it (the prior hand-maintained snapshot had + # it, and Anthropic's model card lists it). Without this the vision filter would silently + # drop it — reported against the catalog; remove once the categories carry vision. + "anthropic/claude-sonnet-4.6": { + "context_window": 1_000_000, + "max_output_tokens": 128_000, + "supports_tools": True, + "supports_vision": True, + }, + "anthropic/claude-sonnet-5": { + "context_window": 1_000_000, + "max_output_tokens": 128_000, + "supports_tools": True, + "supports_vision": True, + }, + # supportsTools: not probed — fails closed + "cohere/north-mini-code": { + "context_window": 256_000, + "max_output_tokens": 16_384, + "supports_tools": False, + "supports_vision": False, + }, + "deepseek/deepseek-chat": { + "context_window": 1_048_576, + "max_output_tokens": 65_536, + "supports_tools": True, + "supports_vision": False, + }, + "deepseek/deepseek-reasoner": { + "context_window": 1_048_576, + "max_output_tokens": 65_536, + "supports_tools": True, + "supports_vision": False, + }, + "deepseek/deepseek-v4-pro": { + "context_window": 1_048_576, + "max_output_tokens": 65_536, + "supports_tools": True, + "supports_vision": False, + }, + "google/gemini-2.5-flash": { + "context_window": 1_048_576, + "max_output_tokens": 65_536, + "supports_tools": True, + "supports_vision": True, + }, + "google/gemini-2.5-flash-lite": { + "context_window": 1_048_576, + "max_output_tokens": 65_536, + "supports_tools": True, + "supports_vision": False, + }, + "google/gemini-2.5-pro": { + "context_window": 1_048_576, + "max_output_tokens": 65_536, + "supports_tools": True, + "supports_vision": True, + }, + "google/gemini-3-flash-preview": { + "context_window": 1_048_576, + "max_output_tokens": 65_536, + "supports_tools": True, + "supports_vision": True, + }, + "google/gemini-3.1-flash-lite": { + "context_window": 1_048_576, + "max_output_tokens": 65_536, + "supports_tools": True, + "supports_vision": False, + }, + "google/gemini-3.1-pro": { + "context_window": 1_048_576, + "max_output_tokens": 65_536, + "supports_tools": True, + "supports_vision": True, + }, + "google/gemini-3.5-flash": { + "context_window": 1_048_576, + "max_output_tokens": 65_536, + "supports_tools": True, + "supports_vision": True, + }, + "google/gemini-3.5-flash-lite": { + "context_window": 1_048_576, + "max_output_tokens": 65_536, + "supports_tools": True, + "supports_vision": False, + }, + "google/gemini-3.6-flash": { + "context_window": 1_048_576, + "max_output_tokens": 65_536, + "supports_tools": True, + "supports_vision": True, + }, + "minimax/minimax-m2.7": { + "context_window": 204_800, + "max_output_tokens": 16_384, + "supports_tools": True, + "supports_vision": False, + }, + "minimax/minimax-m3": { + "context_window": 1_048_576, + "max_output_tokens": 65_536, + "supports_tools": True, + "supports_vision": False, + }, + "moonshot/kimi-k3": { + "context_window": 1_048_576, + "max_output_tokens": 65_536, + "supports_tools": True, + "supports_vision": True, + }, + # supportsTools: not probed — fails closed + "nvidia/llama-3.2-11b-vision": { + "context_window": 128_000, + "max_output_tokens": 16_384, + "supports_tools": False, + "supports_vision": True, + }, + # supportsTools: not probed — fails closed + "nvidia/nemotron-3-nano-30b": { + "context_window": 131_072, + "max_output_tokens": 16_384, + "supports_tools": False, + "supports_vision": False, + }, + "nvidia/nemotron-3-nano-omni-30b-a3b-reasoning": { + "context_window": 256_000, + "max_output_tokens": 16_384, + "supports_tools": False, + "supports_vision": True, + }, + # supportsTools: not probed — fails closed + "nvidia/nemotron-3-ultra-550b": { + "context_window": 1_000_000, + "max_output_tokens": 16_384, + "supports_tools": False, + "supports_vision": False, + }, + # supportsTools: not probed — fails closed + "nvidia/nemotron-3.5-lightning": { + "context_window": 1_000_000, + "max_output_tokens": 16_384, + "supports_tools": False, + "supports_vision": False, + }, + "openai/chat-latest": { + "context_window": 128_000, + "max_output_tokens": 128_000, + "supports_tools": True, + "supports_vision": True, + }, + "openai/gpt-4.1": { + "context_window": 128_000, + "max_output_tokens": 32_768, + "supports_tools": True, + "supports_vision": True, + }, + "openai/gpt-4.1-mini": { + "context_window": 128_000, + "max_output_tokens": 32_768, + "supports_tools": True, + "supports_vision": False, + }, + "openai/gpt-4.1-nano": { + "context_window": 128_000, + "max_output_tokens": 32_768, + "supports_tools": True, + "supports_vision": False, + }, + "openai/gpt-4o": { + "context_window": 128_000, + "max_output_tokens": 16_384, + "supports_tools": True, + "supports_vision": True, + }, + "openai/gpt-4o-mini": { + "context_window": 128_000, + "max_output_tokens": 16_384, + "supports_tools": True, + "supports_vision": False, + }, + "openai/gpt-5-mini": { + "context_window": 200_000, + "max_output_tokens": 128_000, + "supports_tools": True, + "supports_vision": False, + }, + "openai/gpt-5.2": { + "context_window": 400_000, + "max_output_tokens": 128_000, + "supports_tools": True, + "supports_vision": True, + }, + # supportsTools: not probed — fails closed + "openai/gpt-5.2-pro": { + "context_window": 400_000, + "max_output_tokens": 128_000, + "supports_tools": False, + "supports_vision": True, + }, + # supportsTools: gateway unavailable at probe time — fails closed; override: 2026-08-29 + # probe: every request (6 plain + 3 tool attempts) returned a gateway 500, so the probe + # measured an incident, not the model. Codex's function calling is established by the + # 2026-07 Terminal-Bench / tau2 calibration trajectories in portfolio.ts. Hosts observing + # the 500s should drop it with RouterOptions.unavailableModels rather than this snapshot + # claiming the model cannot call tools. + "openai/gpt-5.3-codex": { + "context_window": 400_000, + "max_output_tokens": 128_000, + "supports_tools": True, + "supports_vision": False, + }, + "openai/gpt-5.4": { + "context_window": 1_050_000, + "max_output_tokens": 128_000, + "supports_tools": True, + "supports_vision": True, + }, + "openai/gpt-5.4-mini": { + "context_window": 400_000, + "max_output_tokens": 128_000, + "supports_tools": True, + "supports_vision": True, + }, + "openai/gpt-5.4-nano": { + "context_window": 1_050_000, + "max_output_tokens": 128_000, + "supports_tools": True, + "supports_vision": False, + }, + # supportsTools: not probed — fails closed + "openai/gpt-5.4-pro": { + "context_window": 1_050_000, + "max_output_tokens": 128_000, + "supports_tools": False, + "supports_vision": True, + }, + "openai/gpt-5.5": { + "context_window": 1_050_000, + "max_output_tokens": 128_000, + "supports_tools": True, + "supports_vision": True, + }, + # supportsTools: not probed — fails closed + "openai/gpt-5.5-pro": { + "context_window": 1_050_000, + "max_output_tokens": 128_000, + "supports_tools": False, + "supports_vision": True, + }, + "openai/gpt-5.6-luna": { + "context_window": 1_050_000, + "max_output_tokens": 128_000, + "supports_tools": True, + "supports_vision": True, + }, + "openai/gpt-5.6-luna-pro": { + "context_window": 1_050_000, + "max_output_tokens": 128_000, + "supports_tools": False, + "supports_vision": True, + }, + "openai/gpt-5.6-sol": { + "context_window": 1_050_000, + "max_output_tokens": 128_000, + "supports_tools": True, + "supports_vision": True, + }, + "openai/gpt-5.6-sol-pro": { + "context_window": 1_050_000, + "max_output_tokens": 128_000, + "supports_tools": True, + "supports_vision": True, + }, + "openai/gpt-5.6-terra": { + "context_window": 1_050_000, + "max_output_tokens": 128_000, + "supports_tools": True, + "supports_vision": True, + }, + "openai/gpt-5.6-terra-pro": { + "context_window": 1_050_000, + "max_output_tokens": 128_000, + "supports_tools": True, + "supports_vision": True, + }, + "openai/o1": { + "context_window": 200_000, + "max_output_tokens": 100_000, + "supports_tools": True, + "supports_vision": False, + }, + "openai/o3": { + "context_window": 200_000, + "max_output_tokens": 100_000, + "supports_tools": True, + "supports_vision": False, + }, + "openai/o3-mini": { + "context_window": 128_000, + "max_output_tokens": 100_000, + "supports_tools": True, + "supports_vision": False, + }, + "openai/o4-mini": { + "context_window": 128_000, + "max_output_tokens": 100_000, + "supports_tools": True, + "supports_vision": False, + }, + # supportsTools: not probed — fails closed + "poolside/laguna-xs-2.1": { + "context_window": 131_072, + "max_output_tokens": 16_384, + "supports_tools": False, + "supports_vision": False, + }, + "qwen/qwen3.7-flash": { + "context_window": 1_000_000, + "max_output_tokens": 65_536, + "supports_tools": True, + "supports_vision": False, + }, + "qwen/qwen3.7-max": { + "context_window": 1_000_000, + "max_output_tokens": 65_536, + "supports_tools": True, + "supports_vision": False, + }, + "qwen/qwen3.7-plus": { + "context_window": 1_000_000, + "max_output_tokens": 131_072, + "supports_tools": True, + "supports_vision": False, + }, + "tencent/hy3": { + "context_window": 262_144, + "max_output_tokens": 128_000, + "supports_tools": True, + "supports_vision": False, + }, + "xai/grok-4.3": { + "context_window": 1_000_000, + "max_output_tokens": 16_384, + "supports_tools": True, + "supports_vision": True, + }, + "xai/grok-4.5": { + "context_window": 500_000, + "max_output_tokens": 16_384, + "supports_tools": True, + "supports_vision": True, + }, + "xai/grok-build-0.1": { + "context_window": 256_000, + "max_output_tokens": 16_384, + "supports_tools": True, + "supports_vision": False, + }, + "xiaomi/mimo-v2.5-pro": { + "context_window": 1_048_576, + "max_output_tokens": 131_072, + "supports_tools": True, + "supports_vision": False, + }, + "zai/glm-5": { + "context_window": 200_000, + "max_output_tokens": 128_000, + "supports_tools": True, + "supports_vision": False, + }, + "zai/glm-5-turbo": { + "context_window": 200_000, + "max_output_tokens": 128_000, + "supports_tools": True, + "supports_vision": False, + }, + "zai/glm-5.1": { + "context_window": 200_000, + "max_output_tokens": 128_000, + "supports_tools": True, + "supports_vision": False, + }, + "zai/glm-5.2": { + "context_window": 1_000_000, + "max_output_tokens": 131_072, + "supports_tools": True, + "supports_vision": False, + }, + "zai/glm-5.3": { + "context_window": 1_000_000, + "max_output_tokens": 131_072, + "supports_tools": True, + "supports_vision": False, + }, + "zai/glm-5.3-flash": { + "context_window": 1_000_000, + "max_output_tokens": 131_072, + "supports_tools": True, + "supports_vision": True, + }, + } +) diff --git a/blockrun_llm/router_core/model_profiles.generated.json b/blockrun_llm/router_core/model_profiles.generated.json new file mode 100644 index 0000000..59fe603 --- /dev/null +++ b/blockrun_llm/router_core/model_profiles.generated.json @@ -0,0 +1,530 @@ +{ + "anthropic/claude-fable-5": { + "measuredAt": "2026-08-29T16:51:33Z", + "latencyMs": 9298.5, + "p95LatencyMs": 9873.4, + "outputTokensPerSecond": 55.17, + "errorRate": 0, + "samples": 3 + }, + "anthropic/claude-haiku-4.5": { + "measuredAt": "2026-08-29T16:51:33Z", + "latencyMs": 3157.4, + "p95LatencyMs": 3170.7, + "outputTokensPerSecond": 162.16, + "errorRate": 0, + "samples": 3 + }, + "anthropic/claude-opus-4.5": { + "measuredAt": "2026-08-29T16:51:33Z", + "latencyMs": 6497.7, + "p95LatencyMs": 6953.7, + "outputTokensPerSecond": 78.99, + "errorRate": 0, + "samples": 3 + }, + "anthropic/claude-opus-4.7": { + "measuredAt": "2026-08-29T16:51:33Z", + "latencyMs": 5316.5, + "p95LatencyMs": 6121.5, + "outputTokensPerSecond": 97.34, + "errorRate": 0, + "samples": 3 + }, + "anthropic/claude-opus-4.8": { + "measuredAt": "2026-08-29T16:51:33Z", + "latencyMs": 6216.1, + "p95LatencyMs": 6847.7, + "outputTokensPerSecond": 82.81, + "errorRate": 0, + "samples": 3 + }, + "anthropic/claude-opus-5": { + "measuredAt": "2026-08-29T16:51:33Z", + "latencyMs": 7309, + "p95LatencyMs": 7745.2, + "outputTokensPerSecond": 70.17, + "errorRate": 0, + "samples": 3 + }, + "anthropic/claude-sonnet-4.5": { + "measuredAt": "2026-08-29T16:51:33Z", + "latencyMs": 6330.4, + "p95LatencyMs": 6631.6, + "outputTokensPerSecond": 81.03, + "errorRate": 0, + "samples": 3 + }, + "anthropic/claude-sonnet-4.6": { + "measuredAt": "2026-08-29T16:51:33Z", + "latencyMs": 6508, + "p95LatencyMs": 6698.3, + "outputTokensPerSecond": 78.6, + "errorRate": 0, + "samples": 3 + }, + "anthropic/claude-sonnet-5": { + "measuredAt": "2026-08-29T16:51:33Z", + "latencyMs": 6165.4, + "p95LatencyMs": 6582.9, + "outputTokensPerSecond": 83.62, + "errorRate": 0, + "samples": 3 + }, + "deepseek/deepseek-chat": { + "measuredAt": "2026-08-29T16:51:33Z", + "latencyMs": 4351.4, + "p95LatencyMs": 4543.7, + "outputTokensPerSecond": 117.78, + "errorRate": 0, + "samples": 3 + }, + "deepseek/deepseek-reasoner": { + "measuredAt": "2026-08-29T16:51:33Z", + "latencyMs": 5201.2, + "p95LatencyMs": 6079.6, + "outputTokensPerSecond": 99.77, + "errorRate": 0, + "samples": 3 + }, + "deepseek/deepseek-v4-pro": { + "measuredAt": "2026-08-29T16:51:33Z", + "latencyMs": 8781.2, + "p95LatencyMs": 9881.1, + "outputTokensPerSecond": 58.98, + "errorRate": 0, + "samples": 3 + }, + "google/gemini-2.5-flash": { + "measuredAt": "2026-08-29T16:51:33Z", + "latencyMs": 5416.4, + "p95LatencyMs": 6442.8, + "outputTokensPerSecond": 213.07, + "errorRate": 0, + "samples": 3 + }, + "google/gemini-2.5-flash-lite": { + "measuredAt": "2026-08-29T16:51:33Z", + "latencyMs": 5002.6, + "p95LatencyMs": 5780.3, + "outputTokensPerSecond": 408.43, + "errorRate": 0, + "samples": 3 + }, + "google/gemini-2.5-pro": { + "measuredAt": "2026-08-29T16:51:33Z", + "latencyMs": 28169.5, + "p95LatencyMs": 29491.4, + "outputTokensPerSecond": 147.3, + "errorRate": 0, + "samples": 3 + }, + "google/gemini-3-flash-preview": { + "measuredAt": "2026-08-29T16:51:33Z", + "latencyMs": 4717.1, + "p95LatencyMs": 5037.1, + "outputTokensPerSecond": 198.71, + "errorRate": 0, + "samples": 3 + }, + "google/gemini-3.1-flash-lite": { + "measuredAt": "2026-08-29T16:51:33Z", + "latencyMs": 2855.8, + "p95LatencyMs": 3172.7, + "outputTokensPerSecond": 286.91, + "errorRate": 0, + "samples": 3 + }, + "google/gemini-3.1-pro": { + "measuredAt": "2026-08-29T16:59:54Z", + "latencyMs": 24194.1, + "p95LatencyMs": 27269.6, + "outputTokensPerSecond": 109.47, + "errorRate": 0, + "samples": 3 + }, + "google/gemini-3.5-flash": { + "measuredAt": "2026-08-29T16:51:33Z", + "latencyMs": 5320.6, + "p95LatencyMs": 5429.8, + "outputTokensPerSecond": 226.21, + "errorRate": 0, + "samples": 3 + }, + "google/gemini-3.5-flash-lite": { + "measuredAt": "2026-08-29T16:51:33Z", + "latencyMs": 3515.8, + "p95LatencyMs": 4363.4, + "outputTokensPerSecond": 248.9, + "errorRate": 0, + "samples": 3 + }, + "google/gemini-3.6-flash": { + "measuredAt": "2026-08-29T16:51:33Z", + "latencyMs": 13020, + "p95LatencyMs": 15383.1, + "outputTokensPerSecond": 187.87, + "errorRate": 0, + "samples": 3 + }, + "minimax/minimax-m2.7": { + "measuredAt": "2026-08-29T16:51:33Z", + "latencyMs": 8761.1, + "p95LatencyMs": 10199.3, + "outputTokensPerSecond": 59.18, + "errorRate": 0, + "samples": 3 + }, + "minimax/minimax-m3": { + "measuredAt": "2026-08-29T16:51:33Z", + "latencyMs": 11101.9, + "p95LatencyMs": 26087.1, + "outputTokensPerSecond": 101.12, + "errorRate": 0, + "samples": 3 + }, + "moonshot/kimi-k3": { + "measuredAt": "2026-08-29T16:51:33Z", + "latencyMs": 24498.9, + "p95LatencyMs": 40365.3, + "outputTokensPerSecond": 25.11, + "errorRate": 0, + "samples": 3 + }, + "nvidia/mistral-nemotron": { + "measuredAt": "2026-08-29T16:51:33Z", + "latencyMs": 7349.6, + "p95LatencyMs": 9932.3, + "outputTokensPerSecond": 79.48, + "errorRate": 0.3333, + "samples": 3 + }, + "nvidia/nemotron-3-nano-omni-30b-a3b-reasoning": { + "measuredAt": "2026-08-29T16:59:54Z", + "latencyMs": 9324.6, + "p95LatencyMs": 12992, + "outputTokensPerSecond": 64.96, + "errorRate": 0.3333, + "samples": 3 + }, + "nvidia/nemotron-nano-12b-v2-vl": { + "measuredAt": "2026-08-29T16:51:33Z", + "latencyMs": 5846.9, + "p95LatencyMs": 5846.9, + "outputTokensPerSecond": 87.57, + "errorRate": 0.6667, + "samples": 3 + }, + "nvidia/nemotron-nano-9b-v2": { + "measuredAt": "2026-08-29T16:51:33Z", + "latencyMs": 5282.5, + "p95LatencyMs": 5282.5, + "outputTokensPerSecond": 96.92, + "errorRate": 0.6667, + "samples": 3 + }, + "nvidia/step-3.7-flash": { + "measuredAt": "2026-08-29T16:51:33Z", + "latencyMs": 4617.4, + "p95LatencyMs": 5237.4, + "outputTokensPerSecond": 112.92, + "errorRate": 0.3333, + "samples": 3 + }, + "openai/chat-latest": { + "measuredAt": "2026-08-29T16:51:33Z", + "latencyMs": 3690.9, + "p95LatencyMs": 4344, + "outputTokensPerSecond": 111.85, + "errorRate": 0, + "samples": 3 + }, + "openai/gpt-4.1": { + "measuredAt": "2026-08-29T16:51:33Z", + "latencyMs": 3527.9, + "p95LatencyMs": 3831.7, + "outputTokensPerSecond": 141.27, + "errorRate": 0, + "samples": 3 + }, + "openai/gpt-4.1-mini": { + "measuredAt": "2026-08-29T16:51:33Z", + "latencyMs": 4268.2, + "p95LatencyMs": 5101.5, + "outputTokensPerSecond": 103.42, + "errorRate": 0, + "samples": 3 + }, + "openai/gpt-4.1-nano": { + "measuredAt": "2026-08-29T16:51:33Z", + "latencyMs": 3088.3, + "p95LatencyMs": 3369.2, + "outputTokensPerSecond": 150.31, + "errorRate": 0, + "samples": 3 + }, + "openai/gpt-4o": { + "measuredAt": "2026-08-29T16:51:33Z", + "latencyMs": 2995.2, + "p95LatencyMs": 3174.2, + "outputTokensPerSecond": 171.32, + "errorRate": 0, + "samples": 3 + }, + "openai/gpt-4o-mini": { + "measuredAt": "2026-08-29T16:51:33Z", + "latencyMs": 4751.5, + "p95LatencyMs": 4930.4, + "outputTokensPerSecond": 107.84, + "errorRate": 0, + "samples": 3 + }, + "openai/gpt-5-mini": { + "measuredAt": "2026-08-29T16:51:33Z", + "latencyMs": 4558.1, + "p95LatencyMs": 5081.9, + "outputTokensPerSecond": 113.25, + "errorRate": 0, + "samples": 3 + }, + "openai/gpt-5.2": { + "measuredAt": "2026-08-29T16:51:33Z", + "latencyMs": 5436.6, + "p95LatencyMs": 5928.8, + "outputTokensPerSecond": 95.47, + "errorRate": 0, + "samples": 3 + }, + "openai/gpt-5.3-codex": { + "measuredAt": "2026-08-29T16:59:54Z", + "latencyMs": 15290.4, + "p95LatencyMs": 15290.4, + "outputTokensPerSecond": 33.49, + "errorRate": 0.6667, + "samples": 3 + }, + "openai/gpt-5.4": { + "measuredAt": "2026-08-29T16:51:33Z", + "latencyMs": 5596, + "p95LatencyMs": 5919.4, + "outputTokensPerSecond": 91.67, + "errorRate": 0, + "samples": 3 + }, + "openai/gpt-5.4-mini": { + "measuredAt": "2026-08-29T16:51:33Z", + "latencyMs": 3377.8, + "p95LatencyMs": 3646.8, + "outputTokensPerSecond": 138.08, + "errorRate": 0, + "samples": 3 + }, + "openai/gpt-5.4-nano": { + "measuredAt": "2026-08-29T16:51:33Z", + "latencyMs": 4040.4, + "p95LatencyMs": 4205.9, + "outputTokensPerSecond": 118.52, + "errorRate": 0, + "samples": 3 + }, + "openai/gpt-5.5": { + "measuredAt": "2026-08-29T16:51:33Z", + "latencyMs": 6367.8, + "p95LatencyMs": 7330.7, + "outputTokensPerSecond": 81.29, + "errorRate": 0, + "samples": 3 + }, + "openai/gpt-5.6-luna": { + "measuredAt": "2026-08-29T16:51:33Z", + "latencyMs": 6064.5, + "p95LatencyMs": 7347.3, + "outputTokensPerSecond": 87.93, + "errorRate": 0, + "samples": 3 + }, + "openai/gpt-5.6-luna-pro": { + "measuredAt": "2026-08-29T16:59:54Z", + "latencyMs": 13914.9, + "p95LatencyMs": 13914.9, + "outputTokensPerSecond": 36.79, + "errorRate": 0.6667, + "samples": 3 + }, + "openai/gpt-5.6-sol": { + "measuredAt": "2026-08-29T16:51:33Z", + "latencyMs": 7720.2, + "p95LatencyMs": 9108.2, + "outputTokensPerSecond": 67.47, + "errorRate": 0, + "samples": 3 + }, + "openai/gpt-5.6-sol-pro": { + "measuredAt": "2026-08-29T16:51:33Z", + "latencyMs": 11442.7, + "p95LatencyMs": 13363.1, + "outputTokensPerSecond": 148.75, + "errorRate": 0, + "samples": 3 + }, + "openai/gpt-5.6-terra": { + "measuredAt": "2026-08-29T16:51:33Z", + "latencyMs": 4941, + "p95LatencyMs": 5095.3, + "outputTokensPerSecond": 103.69, + "errorRate": 0, + "samples": 3 + }, + "openai/gpt-5.6-terra-pro": { + "measuredAt": "2026-08-29T16:51:33Z", + "latencyMs": 3574.1, + "p95LatencyMs": 4126.3, + "outputTokensPerSecond": 133.59, + "errorRate": 0, + "samples": 3 + }, + "openai/o1": { + "measuredAt": "2026-08-29T16:51:33Z", + "latencyMs": 4324.9, + "p95LatencyMs": 5838.1, + "outputTokensPerSecond": 125.86, + "errorRate": 0, + "samples": 3 + }, + "openai/o3": { + "measuredAt": "2026-08-29T16:51:33Z", + "latencyMs": 5463.4, + "p95LatencyMs": 5613.1, + "outputTokensPerSecond": 93.8, + "errorRate": 0, + "samples": 3 + }, + "openai/o3-mini": { + "measuredAt": "2026-08-29T16:51:33Z", + "latencyMs": 2912.7, + "p95LatencyMs": 3092.1, + "outputTokensPerSecond": 176.49, + "errorRate": 0, + "samples": 3 + }, + "openai/o4-mini": { + "measuredAt": "2026-08-29T16:51:33Z", + "latencyMs": 4958.7, + "p95LatencyMs": 5313, + "outputTokensPerSecond": 103.81, + "errorRate": 0, + "samples": 3 + }, + "qwen/qwen3.7-flash": { + "measuredAt": "2026-08-29T16:51:33Z", + "latencyMs": 3385.5, + "p95LatencyMs": 4042.7, + "outputTokensPerSecond": 153.94, + "errorRate": 0, + "samples": 3 + }, + "qwen/qwen3.7-max": { + "measuredAt": "2026-08-29T16:51:33Z", + "latencyMs": 9387.1, + "p95LatencyMs": 10490.2, + "outputTokensPerSecond": 54.92, + "errorRate": 0, + "samples": 3 + }, + "qwen/qwen3.7-plus": { + "measuredAt": "2026-08-29T16:51:33Z", + "latencyMs": 9766.6, + "p95LatencyMs": 9798.2, + "outputTokensPerSecond": 52.42, + "errorRate": 0, + "samples": 3 + }, + "tencent/hy3": { + "measuredAt": "2026-08-29T16:51:33Z", + "latencyMs": 6062.3, + "p95LatencyMs": 7070.2, + "outputTokensPerSecond": 87.3, + "errorRate": 0, + "samples": 3 + }, + "xai/grok-4.3": { + "measuredAt": "2026-08-29T16:51:33Z", + "latencyMs": 9467.7, + "p95LatencyMs": 10087.9, + "outputTokensPerSecond": 48.36, + "errorRate": 0, + "samples": 3 + }, + "xai/grok-4.5": { + "measuredAt": "2026-08-29T16:51:33Z", + "latencyMs": 13564.8, + "p95LatencyMs": 17351.9, + "outputTokensPerSecond": 60.71, + "errorRate": 0, + "samples": 3 + }, + "xai/grok-build-0.1": { + "measuredAt": "2026-08-29T16:51:33Z", + "latencyMs": 16394.8, + "p95LatencyMs": 18035.4, + "outputTokensPerSecond": 96.86, + "errorRate": 0, + "samples": 3 + }, + "xiaomi/mimo-v2.5-pro": { + "measuredAt": "2026-08-29T16:51:33Z", + "latencyMs": 12070.7, + "p95LatencyMs": 12386.8, + "outputTokensPerSecond": 42.44, + "errorRate": 0, + "samples": 3 + }, + "zai/glm-5": { + "measuredAt": "2026-08-29T16:51:33Z", + "latencyMs": 6839.7, + "p95LatencyMs": 7261.4, + "outputTokensPerSecond": 75.16, + "errorRate": 0, + "samples": 3 + }, + "zai/glm-5-turbo": { + "measuredAt": "2026-08-29T16:51:33Z", + "latencyMs": 55348.5, + "p95LatencyMs": 114086.6, + "outputTokensPerSecond": 14.64, + "errorRate": 0, + "samples": 3 + }, + "zai/glm-5.1": { + "measuredAt": "2026-08-29T16:51:33Z", + "latencyMs": 15658.4, + "p95LatencyMs": 17307.1, + "outputTokensPerSecond": 32.9, + "errorRate": 0, + "samples": 3 + }, + "zai/glm-5.2": { + "measuredAt": "2026-08-29T16:51:33Z", + "latencyMs": 10308.5, + "p95LatencyMs": 15127.6, + "outputTokensPerSecond": 54.87, + "errorRate": 0, + "samples": 3 + }, + "zai/glm-5.3": { + "measuredAt": "2026-08-29T16:51:33Z", + "latencyMs": 7272.4, + "p95LatencyMs": 7998.1, + "outputTokensPerSecond": 71.09, + "errorRate": 0, + "samples": 3 + }, + "zai/glm-5.3-flash": { + "measuredAt": "2026-08-29T16:51:33Z", + "latencyMs": 10545.3, + "p95LatencyMs": 11672.4, + "outputTokensPerSecond": 49.01, + "errorRate": 0, + "samples": 3 + } +} diff --git a/blockrun_llm/router_core/model_profiles.py b/blockrun_llm/router_core/model_profiles.py new file mode 100644 index 0000000..4d2252c --- /dev/null +++ b/blockrun_llm/router_core/model_profiles.py @@ -0,0 +1,112 @@ +""" +Model-performance priors consumed by the portfolio router. + +Python port of ``@blockrun/router-core`` ``model-profiles.ts``. + +The entries below are a small, auditable seed extracted from the 2026-03-16 +BlockRun performance run. They are deliberately weak priors: live data injected +by the host should replace them through configuration before a release. +Historical numbers must never be presented as a current provider SLA or as +task-quality measurements. +""" + +from __future__ import annotations + +import json +from collections.abc import Mapping +from pathlib import Path +from types import MappingProxyType +from typing import Any + +from .types import ModelPerformanceProfile + +_GENERATED_PATH = Path(__file__).with_name("model_profiles.generated.json") + +#: camelCase (upstream JSON) -> snake_case (this port). +_FIELD_ALIASES = { + "measuredAt": "measured_at", + "latencyMs": "latency_ms", + "p95LatencyMs": "p95_latency_ms", + "outputTokensPerSecond": "output_tokens_per_second", + "intelligenceIndex": "intelligence_index", + "errorRate": "error_rate", + "samples": "samples", +} + + +def _normalize(raw: Mapping[str, Any]) -> ModelPerformanceProfile: + """Accept either the upstream camelCase JSON or already-ported keys.""" + profile: dict[str, Any] = {} + for key, value in raw.items(): + profile[_FIELD_ALIASES.get(key, key)] = value + return profile # type: ignore[return-value] + + +def _load_generated() -> Mapping[str, ModelPerformanceProfile]: + try: + with _GENERATED_PATH.open(encoding="utf-8") as handle: + payload: dict[str, dict[str, Any]] = json.load(handle) + except (OSError, ValueError): + # A missing or corrupt asset must not take routing down: these are + # weak priors, and the router already handles an absent observation. + return MappingProxyType({}) + return MappingProxyType({model: _normalize(raw) for model, raw in payload.items()}) + + +#: Generated from benchmark files that satisfy the uncached-inference +#: invariant. These are weak performance priors (speed/reliability), never +#: task-quality labels. +LIVE_MODEL_PROFILES: Mapping[str, ModelPerformanceProfile] = _load_generated() + +HISTORICAL_MODEL_PROFILES: Mapping[str, ModelPerformanceProfile] = MappingProxyType( + { + "anthropic/claude-haiku-4.5": { + "measured_at": "2026-03-16T13:50:48Z", + "latency_ms": 2305, + "output_tokens_per_second": 140.6, + }, + "anthropic/claude-sonnet-4.6": { + "measured_at": "2026-03-16T13:50:48Z", + "latency_ms": 2110, + "output_tokens_per_second": 121.3, + }, + "deepseek/deepseek-chat": { + "measured_at": "2026-03-16T13:50:48Z", + "latency_ms": 1431, + "output_tokens_per_second": 179.2, + "intelligence_index": 32, + }, + "google/gemini-2.5-flash": { + "measured_at": "2026-03-16T13:50:48Z", + "latency_ms": 1238, + "output_tokens_per_second": 207.6, + "intelligence_index": 20, + }, + "google/gemini-2.5-flash-lite": { + "measured_at": "2026-03-16T13:50:48Z", + "latency_ms": 1353, + "output_tokens_per_second": 192.5, + "intelligence_index": 20, + }, + "google/gemini-2.5-pro": { + "measured_at": "2026-03-16T13:50:48Z", + "latency_ms": 1294, + "output_tokens_per_second": 197.8, + }, + "google/gemini-3.1-pro": { + "measured_at": "2026-03-16T13:50:48Z", + "latency_ms": 1609, + "output_tokens_per_second": 167.2, + }, + "openai/gpt-4o-mini": { + "measured_at": "2026-03-16T13:50:48Z", + "latency_ms": 2764, + "output_tokens_per_second": 92.8, + }, + "openai/gpt-5.3-codex": { + "measured_at": "2026-03-16T13:50:48Z", + "latency_ms": 7935, + "output_tokens_per_second": 32.3, + }, + } +) diff --git a/blockrun_llm/router_core/portfolio.py b/blockrun_llm/router_core/portfolio.py new file mode 100644 index 0000000..77fd07f --- /dev/null +++ b/blockrun_llm/router_core/portfolio.py @@ -0,0 +1,1390 @@ +""" +V3 portfolio router. + +Python port of ``@blockrun/router-core`` ``portfolio.ts``. + +This is deliberately local and deterministic: feature extraction, eligibility +checks and scoring read only request data plus the in-process model registry. +It is therefore safe for the hot path and provides a stable baseline for the +RouterBench evaluation before health telemetry / an optional judge are added. +""" + +from __future__ import annotations + +import math +from dataclasses import dataclass +from datetime import datetime + +from ._js import as_utc, js_bool, js_regex, parse_date +from .model_capabilities import DEFAULT_MODEL_CAPABILITIES +from .model_profiles import HISTORICAL_MODEL_PROFILES, LIVE_MODEL_PROFILES +from .selector import get_fallback_chain, select_model +from .strategy import RulesStrategy, sample_prompt, scan_limit_for +from .tool_intent import infer_tool_requirement +from .types import ( + CandidateScore, + ModelPerformanceProfile, + PortfolioBandWeights, + PortfolioConfig, + RouterOptions, + RoutingDecision, + TaskType, + Tier, + TierConfig, +) + +DEFAULT_PORTFOLIO_WEIGHTS: PortfolioConfig = { + "auto": { + "quality": 0.47, + "capability": 0.2, + "cost": 0.18, + "speed": 0.07, + "reliability": 0.03, + "legacy": 0.05, + }, + "eco": { + "quality": 0.36, + "capability": 0.2, + "cost": 0.28, + "speed": 0.1, + "reliability": 0.04, + "legacy": 0.02, + }, + "premium": { + "quality": 0.58, + "capability": 0.2, + "cost": 0.08, + "speed": 0.06, + "reliability": 0.06, + "legacy": 0.02, + }, + "high_stakes_boost": {"quality": 0.08, "reliability": 0.05}, + "latency_sensitive_speed_boost": 0.08, + "affinity_floor_gap": {"auto": 0.1, "eco": 0.22, "premium": 0.05}, +} + + +@dataclass(frozen=True) +class TaskFeatures: + task_type: TaskType + estimated_input_tokens: int + has_code: bool + needs_tools: bool + tools_available: bool + needs_vision: bool + needs_structured_output: bool + latency_sensitive: bool + high_stakes: bool + language: str # "zh" | "other" + likely_parallel_tool_calls: bool + complex_multi_tool_plan: bool + agent_domain: str # "airline" | "retail" | "web_research" | "other" + deep_web_research: bool + #: "standard" | "high" | "complex_high" | "policy_exception_simple" | "policy_exception" + agent_risk: str + terminal_tool_signal: bool + terminal_safety_sensitive: bool + implicit_terminal_code: bool + + +# ─── Compiled request features (ported 1:1 from the TypeScript regexes) ─── + +_EXPLICIT_REPEAT = js_regex( + r"\b(?:in parallel|simultaneously|concurrently|for each|each of|every one|both" + r"|(?:two|three|multiple|several)\s+(?:cities|locations|items|tasks|orders|users|files))\b" + r"|并行|同时|分别|每个|各自|(?:两个|三个|多个)(?:城市|地点|项目|任务|订单|用户|文件)" + r"|cada uno|para cada|simult[aá]neamente", + ignorecase=True, +) +_SENTENCE_SPLIT = js_regex(r"[.!?。!?]+") +_ADDITIONALLY = js_regex(r"\b(?:also|additionally|furthermore)\b|另外|此外|그리고", ignorecase=True) +_AND_ALSO = js_regex(r"\band\s+(?:also|for the)\b", ignorecase=True) +_PAIRED_QUANTITY = js_regex( + r"\b\d+(?:\.\d+)?\s+(?:and|or)\s+\d+(?:\.\d+)?\s*(?:gb|mb|tb|kg|g|ml|oz|cups?|cores?|cpus?)\b", + ignorecase=True, +) +_TOOL_NAME_SPLIT = js_regex(r"[^a-z0-9\u3400-\u9fff]+") +_LINE_SPLIT = js_regex(r"\r?\n") +_QUANTITY_MENTION = js_regex( + r"\b(?:\d+(?:\.\d+)?|one|two|three|four|five|six|seven|eight|nine|ten)\s*" + r"(?:oz|ounce|ounces|g|gram|grams|kg|ml|cups?|pieces?|tablespoons?)\b", + ignorecase=True, +) +_REPEATED_LOOKUP = js_regex( + r"\b(?:weather|climate|clima|tiempo|temperature|snow|news|report)\b" + r"|天气|气象|温度|降雪|新闻|报告", + ignorecase=True, +) +_MULTI_LOCATION_CONNECTOR = js_regex(r"\b(?:and also|both|y|e)\b|还有|以及|和|、", ignorecase=True) +_COMMA = js_regex(r"[,,]") +_ASCII_COMMA = js_regex(r",") +_DISTINCT_ORDER_PARTS = js_regex( + r"\b(?:food|meal)\b[\s\S]*\bdrink\b|\bdrink\b[\s\S]*\b(?:food|meal)\b", ignorecase=True +) +_KOREAN_CLAUSES = js_regex(r"하고|그리고") + +_OPERATION_TOKENS = frozenset( + { + "add", + "delete", + "remove", + "cancel", + "return", + "exchange", + "modify", + "book", + "transfer", + "send", + "upload", + "download", + "create", + "close", + } +) + +_EXPLICIT_CODE_SIGNAL = js_regex( + r"```|\b(?:typescript|javascript|python|rust|java|sql|stack trace|traceback|exception)\b" + r"|\.(?:ts|tsx|js|py|go|rs)\b", + ignorecase=True, +) +_CODE_CONSTRUCT_SIGNAL = js_regex( + r"\b(?:implement|refactor|debug|write|edit|modify|create|define|review|fix)\b[\s\S]{0,48}" + r"\b(?:api|function|class|method)\b" + r"|\b(?:api|function|class|method)\b[\s\S]{0,48}" + r"\b(?:code|implementation|typescript|javascript|python|rust|java)\b", + ignorecase=True, +) +_NATIVE_CODE_SIGNAL = js_regex( + r"\b(?:programmed|written|implemented?|code)\s+(?:in|using)\s+(?:c\+\+|c|rust|go)\b", + ignorecase=True, +) +_AIRLINE_TOOL = js_regex(r"(?:flight|reservation|airport|baggage|passenger)") +_RETAIL_TOOL = js_regex(r"(?:order|product|item|return|exchange|address)") +_WEB_RESEARCH_TOOL = js_regex(r"^(?:web_?search|web_?fetch)$") +_CLUE_CONNECTORS = js_regex( + r"\b(?:after|before|while|where|whose|which|in \d{4}|as of|over \d+|another|also|furthermore)\b" + r"|(?:之后|之前|其中|截至|超过|另一个|此外)", + ignorecase=True, +) +_ENTITY_RESOLUTION = js_regex( + r"\b(?:identify|who (?:is|was)|what (?:is|was) the name" + r"|which (?:person|player|company|country|city)|find the (?:person|player|name|entity))\b" + r"|(?:找出|识别|是谁|哪位|名称是什么)", + ignorecase=True, +) +_EXACT_ANSWER = js_regex( + r"\b(?:exact answer|single best-supported answer|following clues|multiple public sources)\b" + r"|(?:精确答案|根据.*线索|多个公开来源)", + ignorecase=True, +) +_GLOBAL_OPTIMIZATION = js_regex( + r"\b(?:cheapest|lowest[- ]price|least expensive|most expensive|highest(?:[- ]priced)?" + r"|largest|smallest|maximum|minimum|best available|closest|not (?:cost|exceed))\b" + r"|最便宜|最低价|最贵|最高价|最大|最小", + ignorecase=True, +) +_GLOBAL_SCOPE = js_regex( + r"\b(?:everything|all (?:(?:my|your|their|the) )?(?:future |upcoming )?" + r"(?:items|orders|passengers|flights|reservations|bookings)" + r"|every (?:item|order|passenger|flight|reservation|booking))\b" + r"|全部|所有|每个", + ignorecase=True, +) +_CROSS_RECORD = js_regex( + r"\b(?:another|other|different|previous)\s+(?:order|reservation|booking|account|address)\b" + r"|另一(?:个)?(?:订单|预订|账户|地址)|其他(?:订单|预订|账户|地址)", + ignorecase=True, +) +_RESERVATION_ID = js_regex(r"\b[A-Z0-9]{6}\b") +_CROSS_RESERVATION_BATCH = js_regex( + r"\b(?:two|three|multiple|several)(?:\s+of\s+(?:my|our|the))?\s+(?:upcoming\s+)?" + r"(?:reservations?|bookings?)\b" + r"|\b(?:a\s+)?(?:second|third)\s+(?:reservation|booking)\b", + ignorecase=True, +) +_CONDITIONAL_GLOBAL_TERMS = js_regex( + r"\b(?:if|that (?:contain|have)|longer than|shorter than|under|over|at (?:most|least)" + r"|wherever possible)\b" + r"|如果|超过|少于|不超过|尽可能", + ignorecase=True, +) +_CONDITIONAL_GLOBAL_ACTIONS = js_regex( + r"\b(?:cancel|change|upgrade|move|book)\b[\s\S]*\b(?:cancel|change|upgrade|move|book)\b" + r"|取消[\s\S]*(?:升级|更改)|升级[\s\S]*(?:取消|更改)", + ignorecase=True, +) +_RETURN_INTENT = js_regex( + r"\b(?:return|refund|send back|get (?:my |the )?money back)\b|退货|退款|退回", ignorecase=True +) +_CARD_INTENT = js_regex( + r"\b(?:amex|american express|visa|mastercard|credit card|debit card|different card" + r"|another card|other card)\b" + r"|信用卡|借记卡|其他卡|另一张卡", + ignorecase=True, +) +_SINGLE_SELECTED_RETURN = js_regex( + r"\b(?:return|refund|send back)\b[^.!?。!?]{0,96}" + r"\b(?:the )?(?:pricier|cheaper|more expensive|less expensive|costlier|one)\b", + ignorecase=True, +) +_NEGOTIATED_WORKFLOW = js_regex(r"\b(?:return|exchange)\b|退货|退回|换货|交换", ignorecase=True) +_NUMBERED_STEP = js_regex(r"(?:^|\s)\d+(?:\.\d+)*[.)]\s+") +_LATENCY_SENSITIVE = js_regex( + r"\b(?:urgent|asap|fast|quick|low latency|real[- ]time)\b|尽快|马上|快速|低延迟", + ignorecase=True, +) +_HIGH_STAKES = js_regex( + r"\b(?:production|security|payment|legal|medical|financial|audit)\b" + r"|生产|安全|支付|法律|医疗|财务|审计", + ignorecase=True, +) +_TERMINAL_TOOL = js_regex(r"^(?:terminalexec|terminalinspect|terminalsendkeys)$") +_SIMPLE_TERMINAL_ARTIFACT = js_regex( + r"\b(?:create|write|convert|generate|build|implement|run|fix|repair|debug|make)\b" + r"[\s\S]{0,120}\b(?:file|script|csv|parquet|json|txt|server|endpoint)\b", + ignorecase=True, +) +_TERMINAL_COMPLEX_REPAIR = js_regex( + r"\b(?:multiple|several)\s+(?:scripts?|files?|components?)\b" + r"|\b(?:pipeline|dependencies)\b[\s\S]{0,100}\b(?:fail|issue|fix|repair|run|execute)\b" + r"|\b(?:identify|find|fix|repair)\s+(?:and\s+)?(?:fix\s+)?all\s+(?:the\s+)?issues\b", + ignorecase=True, +) +_TERMINAL_RUNTIME = js_regex( + r"\b(?:gcc|clang|rustc|javac|go\s+build|node|python)\b", ignorecase=True +) +_POLYGLOT = js_regex(r"\bpolyglot\b", ignorecase=True) +_BOTH_TOOLCHAINS = js_regex( + r"\b(?:both|each)\b[\s\S]{0,120}\b(?:compilers?|runtimes?|toolchains?)\b", ignorecase=True +) +_COMPILE_VERB = js_regex(r"\b(?:compile|build|run|execute)\b", ignorecase=True) +_FRAMEWORK_ARTIFACT = js_regex( + r"\b(?:pytorch|tensorflow|jax|onnx|state[_ -]?dict|checkpoint|safetensors?)\b" + r"|\.(?:pth|pt|onnx)\b", + ignorecase=True, +) +_NATIVE_TARGET = js_regex( + r"\b(?:pure|native|programmed|written|implemented?)\s+(?:in|using)\s+(?:c\+\+|c|rust|go)\b" + r"|\b(?:c\+\+|c|rust|go)\s+(?:program|binary|executable|cli|tool|implementation)\b", + ignorecase=True, +) +_INFERENCE_VERB = js_regex( + r"\b(?:inference|model|weights?|tensor|export|convert|load)\b", ignorecase=True +) +_COMPLEX_TERMINAL_OPERATION = js_regex( + r"\b(?:git|ssh|nginx|https|certificate|authentication|credential|deploy|production|encrypt" + r"|gpg|shred|securely delete|decommission|benchmark|evaluate|embedding|chess|image" + r"|search the web|schema|statistical|statistics|aggregate|join|multiple inputs?)\b", + ignorecase=True, +) +_TERMINAL_CREDENTIAL = js_regex( + r"\b(?:ssh|nginx|certificate|authentication|credentials?|passwords?|api keys?|deploy" + r"|production|encrypt|gpg|shred|securely delete|decommission)\b", + ignorecase=True, +) +_TERMINAL_TOKEN_CREDENTIAL = js_regex( + r"\b(?:access|auth|authentication|bearer|secret|api)\s+tokens?\b" + r"|\btokens?\s+(?:secret|credential|authentication)\b", + ignorecase=True, +) +_HAN = js_regex(r"[\u3400-\u9fff]") +_MULTIPLE_CHOICE = js_regex(r"(?:^|\n)\s*[A-D][.)]\s+", ignorecase=True, multiline=True) +_NUMERIC = js_regex(r"-?\d+(?:[.,]\d+)?") +_MATH_MARKERS = js_regex( + r"[+×÷=%$€£¥]|\b(?:total|each|per|times|half|twice|percent|how many|how much|calculate)\b", + ignorecase=True, +) +_TRAILING_QUESTION = js_regex(r"[??]\s*\Z") +_DEBUG_TASK = js_regex( + r"\b(?:bug|debug|error|failure|failing|regression|crash|修复|报错|错误|调试)\b", ignorecase=True +) +_CODE_EDIT_TASK = js_regex( + r"\b(?:refactor|implement|patch|edit|rewrite|重构|实现|修改)\b", ignorecase=True +) +_EXTRACTION_TASK = js_regex(r"\b(?:extract|json|schema|csv|字段|提取)\b", ignorecase=True) +_REASONING_TASK = js_regex( + r"\b(?:prove|derive|theorem|formal|mathematical|reasoning|证明|推导|定理|数学)\b", + ignorecase=True, +) + + +def _likely_needs_parallel_tool_calls( + prompt: str, + needs_tools: bool, + tool_count: int | None, + tool_names: list[str] | None, +) -> bool: + """Detect turns that probably need several tool calls. + + A deliberately conservative request-side feature: it uses only the prompt + and the visible tool count, never benchmark categories or expected answers. + """ + if not needs_tools or tool_count is None or tool_count < 1: + return False + text = prompt.strip() + if _EXPLICIT_REPEAT.search(text): + return True + + sentence_clauses = [ + part.strip() for part in _SENTENCE_SPLIT.split(text) if len(part.strip()) >= 8 + ] + if (_ADDITIONALLY.search(text) and len(sentence_clauses) >= 2) or _AND_ALSO.search(text): + return True + + if _PAIRED_QUANTITY.search(text): + return True + + # Distinctive tokens from two visible tool names are a strong local signal + # for a multi-operation turn (for example add_task + delete_task). + lowered = text.lower() + matched_operation_tokens = { + token + for name in (tool_names or []) + for token in _TOOL_NAME_SPLIT.split(name.lower()) + if token in _OPERATION_TOKENS and token in lowered + } + # A single workflow naturally mentions domain nouns like order/item plus one + # action. Upgrade only when two different visible operation verbs are + # requested (for example cancel + book or add + delete). + if len(matched_operation_tokens) >= 2: + return True + + # Repeated food/logging entries are commonly expressed as several lines, + # each with its own quantity rather than an explicit "for each" phrase. + non_empty_lines = [line.strip() for line in _LINE_SPLIT.split(text) if line.strip()] + quantity_mentions = _QUANTITY_MENTION.findall(text) + if len(non_empty_lines) >= 2 and len(quantity_mentions) >= 2: + return True + + # Weather prompts provide a useful language-independent high-confidence + # pattern: a single lookup tool plus multiple locations joined in one turn. + repeated_lookup = bool(_REPEATED_LOOKUP.search(text)) + multi_location_connector = bool(_MULTI_LOCATION_CONNECTOR.search(text)) + comma_separated_locations = len(_COMMA.findall(text)) >= 2 + if repeated_lookup and (multi_location_connector or comma_separated_locations): + return True + + distinct_order_parts = bool(_DISTINCT_ORDER_PARTS.search(text)) + korean_parallel_clauses = len(_ASCII_COMMA.findall(text)) >= 3 and bool( + _KOREAN_CLAUSES.search(text) + ) + return distinct_order_parts or korean_parallel_clauses + + +def classify_task(prompt: str, system_prompt: str | None, options: RouterOptions) -> TaskFeatures: + """Extract the request-side features the portfolio scorer ranks against.""" + full_text = f"{system_prompt or ''} {prompt}" + estimated_input_tokens = math.ceil(len(full_text) / 4) + # Feature regexes need request shape and intent, not the entire document. + # Sample both ends so a long pasted artifact keeps the task instruction at + # either boundary, while the full length still drives capacity decisions. + scan_limit = scan_limit_for(options) + scanned_prompt = sample_prompt(prompt, scan_limit) + scanned_system_prompt = sample_prompt(system_prompt or "", scan_limit) + scanned_full_text = f"{scanned_system_prompt} {scanned_prompt}" + text = scanned_prompt.lower() + + explicit_code_signal = bool(_EXPLICIT_CODE_SIGNAL.search(scanned_prompt)) + # `class` is common in non-code Agent domains (for example airline cabin + # class). Treat code constructs as code only when the prompt also contains + # an implementation/editing cue, instead of letting a single ambiguous noun + # redirect an entire tool session to the code-agent portfolio. + code_construct_signal = bool(_CODE_CONSTRUCT_SIGNAL.search(scanned_prompt)) + native_code_signal = bool(_NATIVE_CODE_SIGNAL.search(scanned_prompt)) + has_code = explicit_code_signal or code_construct_signal or native_code_signal + + tools_available = options.get("has_tools", False) + requires_tools = options.get("requires_tools") + needs_tools = ( + requires_tools + if requires_tools is not None + else bool(tools_available and infer_tool_requirement(scanned_prompt, scanned_system_prompt)) + ) + tool_names = list(options.get("tool_names") or []) + likely_parallel_tool_calls = _likely_needs_parallel_tool_calls( + scanned_prompt, needs_tools, options.get("tool_count"), tool_names + ) + normalized_tool_names = [name.lower() for name in tool_names] + airline_tool_signal = any(_AIRLINE_TOOL.search(name) for name in normalized_tool_names) + retail_tool_signal = any(_RETAIL_TOOL.search(name) for name in normalized_tool_names) + web_research_tool_signal = any( + _WEB_RESEARCH_TOOL.search(name) for name in normalized_tool_names + ) + if airline_tool_signal and not retail_tool_signal: + agent_domain = "airline" + elif retail_tool_signal and not airline_tool_signal: + agent_domain = "retail" + elif web_research_tool_signal: + agent_domain = "web_research" + else: + agent_domain = "other" + + # Distinguish a cheap lookup from a BrowseComp-like investigation. These + # prompts require joining several clues, resolving an entity, and ending in + # one exact answer; complete agent trajectories show that treating them as + # ordinary search causes long, costly loops. This is request/tool-surface + # evidence only and does not depend on a benchmark id or hidden answer. + clue_connectors = _CLUE_CONNECTORS.findall(scanned_full_text) + entity_resolution_signal = bool(_ENTITY_RESOLUTION.search(scanned_full_text)) + exact_answer_signal = bool(_EXACT_ANSWER.search(scanned_full_text)) + deep_web_research = agent_domain == "web_research" and ( + exact_answer_signal + or (entity_resolution_signal and (len(clue_connectors) >= 3 or len(prompt) >= 320)) + ) + + global_optimization_signal = bool(_GLOBAL_OPTIMIZATION.search(scanned_prompt)) + global_scope_signal = bool(_GLOBAL_SCOPE.search(scanned_prompt)) + global_choice_signal = global_optimization_signal or global_scope_signal + cross_record_signal = bool(_CROSS_RECORD.search(scanned_prompt)) + reservation_ids = _RESERVATION_ID.findall(scanned_prompt) + cross_reservation_batch_signal = agent_domain == "airline" and ( + bool(_CROSS_RESERVATION_BATCH.search(scanned_prompt)) or len(set(reservation_ids)) >= 2 + ) + conditional_global_workflow_signal = ( + agent_domain == "airline" + and global_scope_signal + and bool(_CONDITIONAL_GLOBAL_TERMS.search(scanned_prompt)) + and bool(_CONDITIONAL_GLOBAL_ACTIONS.search(scanned_prompt)) + ) + # A refund explicitly targeted at a named/non-original card can conflict + # with account state and require escalation rather than a substitute action. + # This narrow feature is visible on the first turn and avoids sending every + # ordinary return workflow to the expensive policy specialist. + policy_exception_signal = ( + agent_domain == "retail" + and bool(_RETURN_INTENT.search(scanned_prompt)) + and bool(_CARD_INTENT.search(scanned_prompt)) + ) + # A comparative selector can mention two products while requesting only one + # write (for example "send back the pricier one"). Three-repeat tau2 + # calibration found no quality gain from the policy specialist on these + # single-write cases, so keep them in a distinct, lower-cost risk band. + single_selected_policy_exception = policy_exception_signal and bool( + _SINGLE_SELECTED_RETURN.search(scanned_prompt) + ) + # Returns and exchanges often pivot after confirmation (return -> rethink -> + # exchange -> choose a variant). That future state is not visible to a + # task-start router, so treat the observable workflow verb as the risk cue. + # Simpler cancellation and one-field order edits stay on the standard path. + negotiated_workflow_signal = agent_domain == "retail" and bool( + _NEGOTIATED_WORKFLOW.search(scanned_prompt) + ) + numbered_steps = len(_NUMBERED_STEP.findall(scanned_prompt)) + complex_multi_tool_plan = likely_parallel_tool_calls and ( + (options.get("tool_count") or 0) >= 6 or numbered_steps >= 3 or len(prompt) > 1_200 + ) + + if needs_tools and single_selected_policy_exception: + agent_risk = "policy_exception_simple" + elif needs_tools and policy_exception_signal: + agent_risk = "policy_exception" + # Airline prompts that require a global optimum (for example the cheapest + # itinerary across several candidates) are materially harder than applying + # one change to every passenger in a known reservation. Full-session + # evidence supports Sonnet for the former, while upgrading the latter merely + # because it says "all passengers" caused a large cost increase without a + # quality gain. + elif ( + needs_tools + and agent_domain == "airline" + and (global_optimization_signal or conditional_global_workflow_signal) + ): + agent_risk = "complex_high" + elif needs_tools and ( + likely_parallel_tool_calls + or global_choice_signal + or cross_record_signal + or cross_reservation_batch_signal + or negotiated_workflow_signal + ): + agent_risk = "high" + else: + agent_risk = "standard" + + needs_vision = options.get("has_vision", False) + needs_structured_output = options.get("requires_structured_output", False) + latency_sensitive = bool(_LATENCY_SENSITIVE.search(scanned_full_text)) + high_stakes = bool(_HIGH_STAKES.search(scanned_full_text)) + + # Terminal tasks often describe the desired artifact rather than naming a + # programming language. Treat only small, deterministic local build/file + # work as implicit code. Operational deployment, credentials, destructive + # work, evaluation, vision, and broad search stay on the stronger generic + # tool-agent path. This is a request-side feature, not a benchmark ID list. + terminal_tool_signal = any(_TERMINAL_TOOL.search(name) for name in normalized_tool_names) + simple_terminal_artifact = bool(_SIMPLE_TERMINAL_ARTIFACT.search(scanned_prompt)) + # Multi-file repair is qualitatively different from fixing one known local + # script. The agent must preserve state across inspections, infer ordering + # and dependencies, edit several artifacts, and close the loop with tests. + terminal_complex_repair = terminal_tool_signal and bool( + _TERMINAL_COMPLEX_REPAIR.search(scanned_prompt) + ) + # One artifact that must be accepted by multiple compilers/runtimes is not a + # routine file-writing task. It requires reasoning across incompatible + # grammars and validating every execution path. + mentioned_terminal_runtimes = { + " ".join(name.lower().split()) for name in _TERMINAL_RUNTIME.findall(scanned_prompt) + } + terminal_cross_runtime_artifact = terminal_tool_signal and ( + bool(_POLYGLOT.search(scanned_prompt)) + or bool(_BOTH_TOOLCHAINS.search(scanned_prompt)) + or (len(mentioned_terminal_runtimes) >= 2 and bool(_COMPILE_VERB.search(scanned_prompt))) + ) + # Framework-to-native ports combine binary checkpoint inspection, weight + # export, tensor-layout reasoning, image/data decoding, and a separately + # compiled runtime. + terminal_framework_to_native_artifact = ( + terminal_tool_signal + and bool(_FRAMEWORK_ARTIFACT.search(scanned_prompt)) + and bool(_NATIVE_TARGET.search(scanned_prompt)) + and bool(_INFERENCE_VERB.search(scanned_prompt)) + ) + if ( + needs_tools + and ( + terminal_complex_repair + or terminal_cross_runtime_artifact + or terminal_framework_to_native_artifact + ) + and agent_risk in ("standard", "high") + ): + agent_risk = "complex_high" + + complex_terminal_operation = bool(_COMPLEX_TERMINAL_OPERATION.search(scanned_prompt)) + # A bare "token" is not a credential signal: blockchain, tokenizer, and LLM + # tasks use that word routinely (for example "token transfers"). Only treat + # it as sensitive when the prompt gives it an authentication/secret + # qualifier. API keys remain an unambiguous high-risk signal on their own. + terminal_credential_signal = bool(_TERMINAL_CREDENTIAL.search(scanned_prompt)) or bool( + _TERMINAL_TOKEN_CREDENTIAL.search(scanned_prompt) + ) + terminal_safety_sensitive = terminal_tool_signal and (high_stakes or terminal_credential_signal) + implicit_terminal_code = bool( + needs_tools + and terminal_tool_signal + and agent_risk == "standard" + and not high_stakes + and not complex_terminal_operation + and numbered_steps < 3 + and len(prompt) <= 1_000 + and simple_terminal_artifact + ) + language = "zh" if _HAN.search(scanned_full_text) else "other" + multiple_choice_signals = len(_MULTIPLE_CHOICE.findall(scanned_prompt)) + numeric_signals = len(_NUMERIC.findall(scanned_prompt)) + compact_math_problem = ( + not has_code + and len(prompt) < 2_500 + and numeric_signals >= 2 + and ( + bool(_MATH_MARKERS.search(scanned_prompt)) + or bool(_TRAILING_QUESTION.search(scanned_prompt.strip())) + or numeric_signals >= 3 + ) + ) + + task_type: TaskType = "chat" + if needs_vision: + task_type = "vision" + elif estimated_input_tokens > 80_000: + task_type = "long_context" + elif needs_tools and (has_code or implicit_terminal_code): + task_type = "code_agent" + elif needs_tools and likely_parallel_tool_calls and not complex_multi_tool_plan: + task_type = "tool_agent_parallel" + elif needs_tools: + task_type = "tool_agent" + elif multiple_choice_signals >= 3: + task_type = "reasoning_mcq" + elif compact_math_problem: + task_type = "reasoning_math" + elif _DEBUG_TASK.search(text): + task_type = "debug" + elif has_code or _CODE_EDIT_TASK.search(text): + task_type = "code_edit" + elif needs_structured_output or _EXTRACTION_TASK.search(text): + task_type = "extraction" + elif _REASONING_TASK.search(text): + task_type = "reasoning" + + return TaskFeatures( + task_type=task_type, + estimated_input_tokens=estimated_input_tokens, + has_code=has_code, + needs_tools=bool(needs_tools), + tools_available=bool(tools_available), + needs_vision=bool(needs_vision), + needs_structured_output=bool(needs_structured_output), + latency_sensitive=latency_sensitive, + high_stakes=high_stakes, + language=language, + likely_parallel_tool_calls=likely_parallel_tool_calls, + complex_multi_tool_plan=bool(complex_multi_tool_plan), + agent_domain=agent_domain, + deep_web_research=bool(deep_web_research), + agent_risk=agent_risk, + terminal_tool_signal=terminal_tool_signal, + terminal_safety_sensitive=terminal_safety_sensitive, + implicit_terminal_code=implicit_terminal_code, + ) + + +_AFFINITY_BASE = 0.68 + + +def affinity( + model_id: str, + task: TaskType, + language: str = "other", + agent_domain: str = "other", + deep_web_research: bool = False, + agent_risk: str = "standard", + terminal_tool_signal: bool = False, + terminal_safety_sensitive: bool = False, +) -> float: + """Task affinity for a model, on the same evidence bands as upstream. + + Model family names are intentionally similar (for example + ``gemini-2.5-flash`` vs ``gemini-2.5-flash-lite``). A substring match would + let a smaller sibling inherit a capability claim measured only for the + flagship, so these assignments are model-exact; a sibling can be added only + with its own evidence. + """ + model_id_lower = model_id.lower() + model_name = model_id_lower[model_id_lower.find("/") + 1 :] + + def match(values: list[str], score: float) -> float: + return score if model_name in values else 0.0 + + base = _AFFINITY_BASE + + if task == "code_agent": + if terminal_tool_signal and agent_risk == "complex_high": + # Strong native tool loop until the Responses function-output fix is + # deployed on both gateways; keep Codex available below the floor. + return max( + base, + match(["claude-sonnet-5"], 1), + match(["gpt-5.3-codex"], 0.87), + match(["gpt-5-mini"], 0.78), + match(["gemini-3.5-flash"], 0.76), + ) + # Seven valid full agent + official Terminal-Bench trajectories + # (2026-07-28) gave GPT-5 Mini 4/7 resolved tasks versus 1/7 for the + # prior dynamic code-agent choice. Its token-normalized total cost was + # higher in this small calibration, so keep Codex and Sonnet's quality + # priors above it. DeepSeek V4 Pro is kept below the primary band after + # two consecutive mid-trajectory provider timeouts. + return max( + base, + match(["gpt-5.3-codex"], 1), + match(["claude-sonnet-5"], 0.98), + match(["gpt-5-mini"], 0.96), + match(["gemini-3.5-flash"], 0.92), + match(["kimi-k3"], 0.9), + match(["deepseek-v4-pro", "glm-5.2"], 0.88), + ) + + if task == "tool_agent": + if terminal_tool_signal and agent_risk == "complex_high": + # Keep the Responses-API Codex path outside auto's affinity floor + # until the gateway fix that preserves function_call_output is live + # on both chains. Sonnet has a verified native multi-turn tool loop. + return max( + base, + match(["claude-sonnet-5"], 1), + match(["gpt-5.3-codex"], 0.87), + match(["gpt-5-mini"], 0.78), + match(["gemini-3.5-flash"], 0.76), + ) + if terminal_tool_signal and not terminal_safety_sensitive: + # Seven official Terminal-Bench calibration trajectories favoured + # GPT-5 Mini over the prior dynamic choice. Admit Codex/Sonnet as + # close fallbacks, but let actual request cost break the tie. + return max( + base, + match(["gpt-5-mini"], 1), + match(["gpt-5.3-codex"], 0.98), + match(["claude-sonnet-5"], 0.9), + match(["gemini-3.5-flash"], 0.89), + ) + if terminal_tool_signal and terminal_safety_sensitive: + # Two complete agent observations on the public Terminal-Bench + # new-encrypt-command task ended in Codex repeating the same + # TerminalExec input until the loop guard fired. + return max( + base, + match(["claude-sonnet-5"], 1), + match(["claude-opus-4.8"], 0.9), + match(["gpt-5.3-codex"], 0.84), + ) + if agent_domain == "web_research": + # Complete-session BrowseComp calibration supersedes the earlier + # single-case Opus promotion: strict deduplicated evidence has + # Sonnet 5 at 2/9 versus Opus 5 at 0/3, while Opus also costs more + # and has a much longer tail. + if deep_web_research: + return max( + base, + match(["claude-sonnet-5"], 1), + match(["gpt-5-mini"], 0.88), + match(["gemini-3.5-flash"], 0.84), + match(["claude-opus-5"], 0.8), + match(["claude-opus-4.8"], 0.78), + ) + return max( + base, + match(["claude-sonnet-5"], 1), + match(["gpt-5-mini"], 0.88), + match(["gemini-3.5-flash"], 0.86), + match(["claude-opus-5"], 0.84), + match(["claude-opus-4.8"], 0.82), + ) + # Full-trajectory tau2 calibration (2026-07-28, official gpt-4.1 + # simulator): Sonnet 5 completed both an airline policy task and a + # retail multi-write task with reward 1.0. Gemini 3.5 Flash emitted + # function calls as plain text after the first structured calls. + if agent_domain == "retail": + # Full-session calibration: GPT-5 Mini completed two local/single + # retail workflows at a fraction of Sonnet's token cost. It remains + # ineligible for promotion when the prompt asks for multiple + # actions, cross-record discovery, or a global optimum. DeepSeek V4 + # Pro completed all three high-risk retail calibration trajectories. + if agent_risk == "standard": + return max( + base, + match(["gpt-5-mini"], 1), + match(["claude-sonnet-5"], 0.88), + match(["gemini-3.5-flash"], 0.82), + match(["gpt-5.3-codex"], 0.81), + match(["kimi-k3"], 0.78), + match(["deepseek-v4-pro"], 0.76), + ) + if agent_risk == "policy_exception": + return max( + base, + match(["gpt-4.1"], 1), + match(["claude-sonnet-5"], 0.9), + match(["deepseek-v4-pro"], 0.82), + match(["gpt-5-mini"], 0.8), + match(["gpt-4o-mini"], 0.76), + ) + if agent_risk == "policy_exception_simple": + return max( + base, + match(["gpt-5-mini"], 1), + match(["gpt-4.1"], 0.86), + match(["deepseek-v4-pro"], 0.82), + match(["gpt-4o-mini"], 0.8), + ) + return max( + base, + match(["deepseek-v4-pro"], 1), + match(["claude-sonnet-5"], 0.88), + match(["gemini-3.5-flash"], 0.82), + match(["gpt-5.3-codex"], 0.81), + match(["kimi-k3"], 0.78), + match(["gpt-5-mini"], 0.76), + ) + # Standard airline workflows stay on GPT-5 Mini: six full-session + # development trajectories gave it the same 5/6 success as Sonnet at + # roughly one order of magnitude lower normalized token cost. Promote + # only global optimization / conditional-global work. + if agent_domain == "airline": + if agent_risk == "complex_high": + return max( + base, + match(["claude-sonnet-5"], 1), + match(["gpt-5-mini"], 0.78), + match(["gemini-3.5-flash"], 0.76), + match(["deepseek-v4-pro"], 0.74), + ) + return max( + base, + match(["gpt-5-mini"], 1), + match(["claude-sonnet-5"], 0.9), + match(["gemini-3.5-flash"], 0.8), + match(["deepseek-v4-pro"], 0.76), + ) + return max( + base, + match(["claude-sonnet-5"], 1), + match(["gemini-3.5-flash"], 0.88), + match(["gpt-5.3-codex"], 0.87), + match(["gpt-5-mini"], 0.84), + match(["kimi-k3"], 0.85), + match(["deepseek-v4-pro"], 0.82), + ) + + if task == "tool_agent_parallel": + if terminal_tool_signal: + # Multi-file Terminal work is not equivalent to a one-turn parallel + # function-call benchmark. Sonnet is the strongest trajectory-tested + # cost-controlled default; Opus remains a close safety fallback. + if terminal_safety_sensitive: + return max( + base, + match(["claude-sonnet-5"], 1), + match(["claude-opus-4.8"], 0.9), + match(["gpt-5.3-codex"], 0.86), + ) + return max( + base, + match(["gpt-5-mini"], 1), + match(["gpt-5.3-codex"], 0.98), + match(["claude-sonnet-5"], 0.92), + match(["gemini-3.5-flash"], 0.88), + ) + if agent_domain == "web_research": + if deep_web_research: + return max( + base, + match(["claude-sonnet-5"], 1), + match(["gpt-5-mini"], 0.88), + match(["gemini-3.5-flash"], 0.84), + match(["claude-opus-5"], 0.8), + match(["claude-opus-4.8"], 0.78), + ) + return max( + base, + match(["claude-sonnet-5"], 1), + match(["gpt-5-mini"], 0.88), + match(["gemini-3.5-flash"], 0.86), + match(["claude-opus-5"], 0.84), + match(["claude-opus-4.8"], 0.82), + ) + if agent_domain == "retail": + if agent_risk == "policy_exception": + return max( + base, + match(["gpt-4.1"], 1), + match(["claude-sonnet-5"], 0.9), + match(["deepseek-v4-pro"], 0.82), + match(["gpt-5-mini"], 0.8), + match(["gpt-4o-mini"], 0.76), + ) + if agent_risk == "policy_exception_simple": + return max( + base, + match(["gpt-5-mini"], 1), + match(["gpt-4.1"], 0.86), + match(["deepseek-v4-pro"], 0.82), + match(["gpt-4o-mini"], 0.8), + ) + return max( + base, + match(["deepseek-v4-pro"], 1), + match(["claude-sonnet-5"], 0.88), + match(["claude-opus-4.8"], 0.84), + match(["gpt-5-mini"], 0.78), + match(["gemini-3.5-flash"], 0.76), + ) + if agent_domain == "airline": + if agent_risk == "complex_high": + return max( + base, + match(["claude-sonnet-5"], 1), + match(["gpt-5-mini"], 0.78), + match(["claude-opus-4.8"], 0.76), + match(["gemini-3.5-flash"], 0.74), + ) + return max( + base, + match(["gpt-5-mini"], 1), + match(["claude-sonnet-5"], 0.9), + match(["gemini-3.5-flash"], 0.8), + ) + # RouterBench calibration, 2026-07-26: Opus 4.8 produced complete + # multi-call payloads on 2/3 multilingual BFCL parallel cases. Gemini + # 3.5 Flash, Sonnet 5, DeepSeek V4 Pro, and Grok 4.5 were 0/3. This + # narrow prior only applies after the conservative prompt feature above. + return max( + base, + match(["claude-opus-4.8"], 1), + match(["claude-sonnet-5"], 0.84), + match(["grok-4.5"], 0.82), + match(["gemini-3.5-flash"], 0.8), + match(["deepseek-v4-pro"], 0.78), + ) + + if task in ("code_edit", "debug"): + return max( + base, + match(["gpt-5.3-codex"], 1), + match(["claude-sonnet-4.6"], 0.94), + match(["glm-5.2"], 0.9), + match(["deepseek-v4-pro"], 0.86), + ) + + if task == "reasoning": + return max( + base, + match(["claude-sonnet-5", "claude-sonnet-4.6"], 0.98), + match(["deepseek-v4-pro"], 0.95), + match(["grok-4.5"], 0.94), + match(["gemini-3.1-pro", "gemini-3.5-flash"], 0.92), + ) + + if task == "reasoning_mcq": + # RouterBench calibration (2026-07-28, six stratified GPQA Diamond + # tasks, identical agent adapter and 512-token budget): Gemini 3 Flash + # Preview scored 5/6, Gemini 3.5 Flash 4/6, and Gemini 3.1 Pro 3/6 while + # costing ~170x more than Flash. Version recency alone is not a quality + # signal, and unused host tools must not change this model choice. + return max( + base, + match(["gemini-3-flash-preview"], 1), + match(["gemini-3.5-flash"], 0.91), + match(["grok-4.5"], 0.9), + match(["claude-sonnet-5"], 0.88), + match(["deepseek-v4-pro"], 0.84), + ) + + if task == "reasoning_math": + # Same calibration, five multilingual MGSM tasks: Gemini 3.5 Flash was + # 5/5 with the lowest cost and latency; four current flagships were 4/5 + # and Kimi K2.7 was 3/5. + return max( + base, + match(["gemini-3.5-flash"], 1), + match(["grok-4.5"], 0.93), + match(["claude-sonnet-5", "deepseek-v4-pro", "kimi-k3"], 0.9), + ) + + if task == "vision": + return max( + base, + match(["gemini-3.1-pro"], 0.96), + match(["qwen3.7-max", "claude-sonnet-4.6", "kimi-k3", "grok-4.3"], 0.9), + ) + + if task == "long_context": + # Long-context eligibility is necessary but not sufficient: a provider + # can advertise a 1M window yet return an empty completion near that + # boundary. Keep the proven long-context flagship in the lead and put + # less-established alternatives in a separate affinity band so price + # alone cannot displace it. + return max( + base, + match(["gemini-3.1-pro"], 1), + match(["qwen3.7-max", "glm-5.2"], 0.89), + match(["gemini-3.5-flash"], 0.88), + match(["deepseek-v4-pro"], 0.85), + ) + + if task == "extraction": + # A structured extraction must preserve both the output contract and the + # source-language fields. For Mandarin input, keep the language-native + # Kimi candidate in a distinct affinity band. This is deliberately a + # candidate-pool decision (rather than a brittle post-hoc override): it + # still falls back normally if that model is unavailable or ineligible. + # Kimi K3 costs ~5x its retired sibling K2.7, so the band must be wider + # than the auto affinity_floor_gap (0.10) or price alone re-selects a + # non-native model for Mandarin input; 0.12 keeps K3 alone in the + # primary band for zh and leaves every other language untouched. + kimi_extraction_affinity = 1.0 if language == "zh" else 0.9 + other_extraction_affinity = 0.88 if language == "zh" else 0.9 + return max( + base, + match( + ["gemini-3.5-flash", "gemini-2.5-flash", "gpt-4o-mini"], + other_extraction_affinity, + ), + match(["claude-sonnet-5", "claude-sonnet-4.6"], other_extraction_affinity), + match(["kimi-k3"], kimi_extraction_affinity), + ) + + return max(base, match(["gemini-3.5-flash", "gemini-2.5-flash", "kimi-k3"], 0.86)) + + +def evidence_candidates(task: TaskType) -> list[str]: + """Models with task-level calibration evidence, added to the tier chain.""" + if task == "code_agent": + return [ + "openai/gpt-5.3-codex", + "anthropic/claude-sonnet-5", + "openai/gpt-5-mini", + "google/gemini-3.5-flash", + "moonshot/kimi-k3", + "deepseek/deepseek-v4-pro", + ] + if task == "tool_agent": + return [ + "anthropic/claude-sonnet-5", + "anthropic/claude-opus-5", + "openai/gpt-5-mini", + "openai/gpt-4.1", + "openai/gpt-4o-mini", + "google/gemini-3.5-flash", + "openai/gpt-5.3-codex", + "moonshot/kimi-k3", + "deepseek/deepseek-v4-pro", + ] + if task == "tool_agent_parallel": + return [ + "anthropic/claude-opus-5", + "anthropic/claude-opus-4.8", + "anthropic/claude-sonnet-5", + "openai/gpt-5-mini", + "openai/gpt-4.1", + "openai/gpt-4o-mini", + "xai/grok-4.5", + "google/gemini-3.5-flash", + "deepseek/deepseek-v4-pro", + ] + if task == "long_context": + return [ + "google/gemini-3.1-pro", + "deepseek/deepseek-v4-pro", + "qwen/qwen3.7-max", + "zai/glm-5.2", + "google/gemini-3.5-flash", + ] + if task == "reasoning_mcq": + return [ + "google/gemini-3-flash-preview", + "google/gemini-3.5-flash", + "xai/grok-4.5", + "anthropic/claude-sonnet-5", + "deepseek/deepseek-v4-pro", + ] + if task == "extraction": + # Kimi K3 is no longer on the auto MEDIUM chain (K2.7 was); the + # language-native extraction band in affinity() needs it in the pool. + return ["moonshot/kimi-k3", "google/gemini-3.5-flash", "anthropic/claude-sonnet-5"] + if task == "reasoning_math": + return [ + "google/gemini-3.5-flash", + "xai/grok-4.5", + "anthropic/claude-sonnet-5", + "deepseek/deepseek-v4-pro", + "moonshot/kimi-k3", + ] + return [] + + +def is_eligible( + model_id: str, + features: TaskFeatures, + max_output_tokens: int, + options: RouterOptions, +) -> bool: + """Hard capability filter: capacity, tools, vision, structured output.""" + host_capabilities = options.get("model_capabilities") or {} + model = host_capabilities.get(model_id) or DEFAULT_MODEL_CAPABILITIES.get(model_id) + # Preserve compatibility for temporarily catalog-less fallback IDs. They are + # kept behind known-model candidates but are not silently dropped. + if not model: + return True + if features.needs_tools and not model["supports_tools"]: + return False + if features.needs_vision and not model["supports_vision"]: + return False + if features.needs_structured_output and not model["supports_tools"]: + return False + if model["max_output_tokens"] < max_output_tokens: + return False + return model["context_window"] >= (features.estimated_input_tokens + max_output_tokens) * 1.1 + + +def estimated_cost( + model_id: str, options: RouterOptions, input_tokens: int, output_tokens: int +) -> float: + price = options["model_pricing"].get(model_id) + if not price: + return math.inf + flat = price.get("flat_price") + if flat: + return float(flat) + return ( + input_tokens * price.get("input_price", 0) + output_tokens * price.get("output_price", 0) + ) / 1_000_000 + + +@dataclass(frozen=True) +class _ProfileScore: + quality: float | None + speed: float + tail_speed: float + reliability: float + freshness: float + + +def profile_score(model_id: str, options: RouterOptions, now: datetime) -> _ProfileScore | None: + """Weak speed/reliability priors, decayed by age and sample count.""" + host_performance = options.get("model_performance") or {} + profile: ModelPerformanceProfile | None = ( + host_performance.get(model_id) + or LIVE_MODEL_PROFILES.get(model_id) + or HISTORICAL_MODEL_PROFILES.get(model_id) + ) + if not profile: + return None + measured_at = parse_date(profile.get("measured_at", "")) + if measured_at is None: + return None + age_days = max(0.0, (now - measured_at).total_seconds() / 86_400) + # A 30-day half-life makes old data a tie-breaker only. Small probe runs are + # also weak evidence: three quick samples should not overturn a curated tier + # ordering merely because of a transient provider tail. Callers that inject + # an observation without a sample count retain the legacy full-confidence + # behaviour for compatibility. + samples = profile.get("samples") + sample_confidence = 1.0 if samples is None else min(1.0, max(0.0, samples) / 10) + freshness = math.pow(0.5, age_days / 30) * sample_confidence + intelligence_index = profile.get("intelligence_index") + quality = None if intelligence_index is None else min(1.0, intelligence_index / 50) + latency_ms = profile.get("latency_ms", 0) + speed = min( + 1.0, + (2_000 / max(500, latency_ms) + profile.get("output_tokens_per_second", 0) / 250) / 2, + ) + tail_speed = min(1.0, 3_000 / max(750, profile.get("p95_latency_ms", latency_ms))) + reliability = max(0.0, 1 - profile.get("error_rate", 0)) + return _ProfileScore( + quality=quality, + speed=speed, + tail_speed=tail_speed, + reliability=reliability, + freshness=freshness, + ) + + +_WEB_RESEARCH_FALLBACK_ORDER = [ + "anthropic/claude-sonnet-5", + "openai/gpt-5-mini", + "google/gemini-3.5-flash", + "anthropic/claude-opus-5", + "anthropic/claude-opus-4.8", + "openai/gpt-5.3-codex", +] + + +class PortfolioStrategy: + """Candidate router used for Auto. + + Rules still set the capability tier; V3 ranks within it. + """ + + name = "portfolio" + + def route( + self, + prompt: str, + system_prompt: str | None, + max_output_tokens: int, + options: RouterOptions, + ) -> RoutingDecision: + features = classify_task(prompt, system_prompt, options) + rules_options: RouterOptions = dict(options) # type: ignore[assignment] + rules_options["requires_tools"] = features.needs_tools + base = RulesStrategy().route(prompt, system_prompt, max_output_tokens, rules_options) + tier_configs = base.get("tier_configs") + if not tier_configs: + return base + + target_tier: Tier = ( + "REASONING" + if features.task_type in ("reasoning_mcq", "reasoning_math") + and base["tier"] in ("SIMPLE", "MEDIUM") + else base["tier"] + ) + tier_config = tier_configs.get(target_tier) + configured_candidates = get_fallback_chain(target_tier, tier_configs) if tier_config else [] + # Evidence candidates join the configured chain, but the host's + # unavailable set applies to both: the configured side arrives filtered + # through RulesStrategy, and a dead evidence model must not re-enter here. + unavailable = set(options.get("unavailable_models") or ()) + chain = [ + model + for model in dict.fromkeys( + [*configured_candidates, *evidence_candidates(features.task_type)] + ) + if isinstance(model, str) and model and model not in unavailable + ] + eligible = [ + model for model in chain if is_eligible(model, features, max_output_tokens, options) + ] + eligible_candidates = eligible if eligible else chain + if not eligible_candidates: + return base + + routing_profile = options.get("routing_profile") + portfolio = options["config"].get("portfolio") or DEFAULT_PORTFOLIO_WEIGHTS + profile_weights: PortfolioBandWeights + if routing_profile == "eco": + profile_weights = portfolio["eco"] + base_floor_gap = portfolio["affinity_floor_gap"]["eco"] + elif routing_profile == "premium": + profile_weights = portfolio["premium"] + base_floor_gap = portfolio["affinity_floor_gap"]["premium"] + else: + profile_weights = portfolio["auto"] + base_floor_gap = portfolio["affinity_floor_gap"]["auto"] + + affinities = { + model: affinity( + model, + features.task_type, + features.language, + features.agent_domain, + features.deep_web_research, + features.agent_risk, + features.terminal_tool_signal, + features.terminal_safety_sensitive, + ) + for model in eligible_candidates + } + best_affinity = max(affinities.values()) + specific_affinity = [ + model for model in eligible_candidates if affinities[model] > _AFFINITY_BASE + ] + # A tier's fallback list is primarily an availability/recovery chain, not + # a set of equally validated substitutes. Re-ranking every fallback lets + # a cheap generic model displace the curated primary merely because it + # has a favourable short performance probe. Only promote models with + # explicit task affinity; otherwise retain the first eligible tier model. + affinity_pool = specific_affinity if specific_affinity else [eligible_candidates[0]] + # Generic Terminal work has much wider trajectory variance than a + # BFCL-like one-turn parallel call. Keep the strong-model safety band, + # but admit the next capable tier so Auto's cost/reliability score can + # reject an Opus primary that is materially more expensive without + # measured benefit. + affinity_floor_gap = ( + max(base_floor_gap, 0.15 if features.terminal_safety_sensitive else 0.12) + if features.terminal_tool_signal + else base_floor_gap + ) + candidates = [ + model + for model in affinity_pool + if affinities[model] >= best_affinity - affinity_floor_gap + ] + costs = [ + estimated_cost(model, options, features.estimated_input_tokens, max_output_tokens) + for model in candidates + ] + finite_costs = [cost for cost in costs if math.isfinite(cost)] + min_cost = min(finite_costs) if finite_costs else 0.0 + max_cost = max(finite_costs) if finite_costs else 1.0 + + now = as_utc(options.get("now")) + ranked_entries: list[CandidateScore] = [] + for index, model in enumerate(candidates): + cost = estimated_cost( + model, options, features.estimated_input_tokens, max_output_tokens + ) + cost_score = ( + 1 - (cost - min_cost) / (max_cost - min_cost) + if math.isfinite(cost) and max_cost > min_cost + else 0.5 + ) + capability_score = ( + 1.0 if is_eligible(model, features, max_output_tokens, options) else 0.0 + ) + profile = profile_score(model, options, now) + # Fresh observations can refine affinity. Historical observations + # fade quickly and never replace task-level RouterBench evidence. + model_affinity = affinities[model] + if profile is None or profile.quality is None: + observed_quality = model_affinity + else: + observed_quality = ( + model_affinity * (1 - profile.freshness) + profile.quality * profile.freshness + ) + observed_speed = profile.speed * profile.freshness if profile else 0.5 + observed_tail_speed = profile.tail_speed * profile.freshness if profile else 0.5 + observed_reliability = ( + profile.reliability * profile.freshness + (1 - profile.freshness) + if profile + else 1.0 + ) + # Preserve a small amount of the hand-curated fallback order while + # V3's task affinity and real request constraints do the main work. + legacy_score = 1 - index / max(1, len(candidates) - 1) + quality_weight = profile_weights["quality"] + ( + portfolio["high_stakes_boost"]["quality"] if features.high_stakes else 0 + ) + speed_score = observed_tail_speed if features.latency_sensitive else observed_speed + speed_weight = profile_weights["speed"] + ( + portfolio["latency_sensitive_speed_boost"] if features.latency_sensitive else 0 + ) + reliability_weight = profile_weights["reliability"] + ( + portfolio["high_stakes_boost"]["reliability"] if features.high_stakes else 0 + ) + score = ( + observed_quality * quality_weight + + capability_score * profile_weights["capability"] + + cost_score * profile_weights["cost"] + + speed_score * speed_weight + + observed_reliability * reliability_weight + + legacy_score * profile_weights["legacy"] + ) + ranked_entries.append( + { + "model": model, + "score": score, + "quality": observed_quality, + "cost": cost_score, + "speed": speed_score, + "reliability": observed_reliability, + } + ) + ranked_entries.sort(key=lambda entry: entry["score"], reverse=True) + scored_models = [entry["model"] for entry in ranked_entries] + # The affinity floor controls which models may compete for the primary; + # it must not erase availability fallbacks. Append all remaining eligible + # models in their curated chain order after the scored primary pool. + if features.agent_domain == "web_research": + ranked = [ + *scored_models, + *[ + model + for model in _WEB_RESEARCH_FALLBACK_ORDER + if model in eligible_candidates and model not in scored_models + ], + *[ + model + for model in eligible_candidates + if model not in scored_models and model not in _WEB_RESEARCH_FALLBACK_ORDER + ], + ] + else: + ranked = [ + *scored_models, + *[model for model in eligible_candidates if model not in scored_models], + ] + + model = ranked[0] if ranked else base["model"] + selected_tier_configs: dict[str, TierConfig] = { + **tier_configs, + target_tier: {"primary": model, "fallback": ranked[1:]}, + } + # select_model only reads the selected tier; retain the complete tier map + # for host fallback. + decision = select_model( + target_tier, + base["confidence"], + "portfolio", + f"{base['reasoning']} | v3 task={features.task_type}" + f" agentRisk={features.agent_risk}" + f" deepWebResearch={js_bool(features.deep_web_research)}" + f" terminalCode={js_bool(features.implicit_terminal_code)}" + f" terminalSafety={js_bool(features.terminal_safety_sensitive)}" + f" candidates={len(ranked)}", + selected_tier_configs, + options["model_pricing"], + features.estimated_input_tokens, + max_output_tokens, + routing_profile, + base.get("agentic_score"), + ) + decision["tier_configs"] = selected_tier_configs + profile_value = base.get("profile") + if profile_value is not None: + decision["profile"] = profile_value + decision["candidates"] = ranked + decision["candidate_scores"] = ranked_entries + decision["task_type"] = features.task_type + decision["router_version"] = "v3-portfolio" + return decision diff --git a/blockrun_llm/router_core/rules.py b/blockrun_llm/router_core/rules.py new file mode 100644 index 0000000..baebe18 --- /dev/null +++ b/blockrun_llm/router_core/rules.py @@ -0,0 +1,326 @@ +""" +Rule-Based Classifier (v2 — Weighted Scoring) + +Python port of ``@blockrun/router-core`` ``rules.ts``. + +Scores a request across 15 weighted dimensions and maps the aggregate score to +a tier using configurable boundaries. Confidence is calibrated via sigmoid — +low confidence triggers the fallback classifier. + +Handles 70-80% of requests in < 1ms with zero cost. +""" + +from __future__ import annotations + +import math + +from ._js import js_regex +from .types import DimensionScore, ScoringConfig, ScoringResult, Tier, TokenCountThresholds + +_MULTI_STEP_PATTERNS = [ + js_regex(r"first.*then", ignorecase=True), + js_regex(r"step \d", ignorecase=True), + js_regex(r"\d\.\s"), +] +_QUESTION_MARK = js_regex(r"\?") + + +# ─── Dimension Scorers ─── +# Each returns a score in [-1, 1] and an optional signal string. + + +def _score_token_count( + estimated_tokens: int, + thresholds: TokenCountThresholds, +) -> DimensionScore: + if estimated_tokens < thresholds["simple"]: + return { + "name": "tokenCount", + "score": -1.0, + "signal": f"short ({estimated_tokens} tokens)", + } + if estimated_tokens > thresholds["complex"]: + return {"name": "tokenCount", "score": 1.0, "signal": f"long ({estimated_tokens} tokens)"} + return {"name": "tokenCount", "score": 0, "signal": None} + + +def _score_keyword_match( + text: str, + keywords: list[str], + name: str, + signal_label: str, + thresholds: tuple[int, int], + scores: tuple[float, float, float], +) -> DimensionScore: + """``thresholds`` is ``(low, high)``; ``scores`` is ``(none, low, high)``.""" + low_threshold, high_threshold = thresholds + none_score, low_score, high_score = scores + matches = [keyword for keyword in keywords if keyword.lower() in text] + if len(matches) >= high_threshold: + return { + "name": name, + "score": high_score, + "signal": f"{signal_label} ({', '.join(matches[:3])})", + } + if len(matches) >= low_threshold: + return { + "name": name, + "score": low_score, + "signal": f"{signal_label} ({', '.join(matches[:3])})", + } + return {"name": name, "score": none_score, "signal": None} + + +def _score_multi_step(text: str) -> DimensionScore: + if any(pattern.search(text) for pattern in _MULTI_STEP_PATTERNS): + return {"name": "multiStepPatterns", "score": 0.5, "signal": "multi-step"} + return {"name": "multiStepPatterns", "score": 0, "signal": None} + + +def _score_question_complexity(prompt: str) -> DimensionScore: + count = len(_QUESTION_MARK.findall(prompt)) + if count > 3: + return {"name": "questionComplexity", "score": 0.5, "signal": f"{count} questions"} + return {"name": "questionComplexity", "score": 0, "signal": None} + + +def _score_agentic_task(text: str, keywords: list[str]) -> tuple[DimensionScore, float]: + """Score agentic task indicators. + + Returns ``(dimension, agentic_score)`` where the 0-1 agentic score is based + on keyword matches: 4+ matches = 1.0 (high agentic), 3 = 0.6 (moderate, + triggers auto-agentic mode), 1-2 = 0.2 (low). Thresholds were raised + because common keywords were pruned from the list. + """ + match_count = 0 + signals: list[str] = [] + + for keyword in keywords: + if keyword.lower() in text: + match_count += 1 + if len(signals) < 3: + signals.append(keyword) + + if match_count >= 4: + return ( + {"name": "agenticTask", "score": 1.0, "signal": f"agentic ({', '.join(signals)})"}, + 1.0, + ) + if match_count >= 3: + return ( + {"name": "agenticTask", "score": 0.6, "signal": f"agentic ({', '.join(signals)})"}, + 0.6, + ) + if match_count >= 1: + return ( + { + "name": "agenticTask", + "score": 0.2, + "signal": f"agentic-light ({', '.join(signals)})", + }, + 0.2, + ) + + return ({"name": "agenticTask", "score": 0, "signal": None}, 0.0) + + +# ─── Main Classifier ─── + + +def classify_by_rules( + prompt: str, + system_prompt: str | None, + estimated_tokens: int, + config: ScoringConfig, +) -> ScoringResult: + """Classify a request into a tier with calibrated confidence.""" + # Score against user prompt only — system prompts contain boilerplate + # keywords (tool definitions, skill descriptions, behavioral rules) that + # dominate scoring and make every request score identically. + user_text = prompt.lower() + + # Score the base dimensions against user text only; the agentic dimension is + # appended below, so the scored total is one more than this list. + dimensions: list[DimensionScore] = [ + # Token count uses total estimated tokens (system + user) — context size + # matters for model selection. + _score_token_count(estimated_tokens, config["token_count_thresholds"]), + _score_keyword_match( + user_text, config["code_keywords"], "codePresence", "code", (1, 2), (0, 0.5, 1.0) + ), + _score_keyword_match( + user_text, + config["reasoning_keywords"], + "reasoningMarkers", + "reasoning", + (1, 2), + (0, 0.7, 1.0), + ), + _score_keyword_match( + user_text, + config["technical_keywords"], + "technicalTerms", + "technical", + (2, 4), + (0, 0.5, 1.0), + ), + _score_keyword_match( + user_text, + config["creative_keywords"], + "creativeMarkers", + "creative", + (1, 2), + (0, 0.5, 0.7), + ), + _score_keyword_match( + user_text, + config["simple_keywords"], + "simpleIndicators", + "simple", + (1, 2), + (0, -1.0, -1.0), + ), + _score_multi_step(user_text), + _score_question_complexity(prompt), + # 6 new dimensions + _score_keyword_match( + user_text, + config["imperative_verbs"], + "imperativeVerbs", + "imperative", + (1, 2), + (0, 0.3, 0.5), + ), + _score_keyword_match( + user_text, + config["constraint_indicators"], + "constraintCount", + "constraints", + (1, 3), + (0, 0.3, 0.7), + ), + _score_keyword_match( + user_text, + config["output_format_keywords"], + "outputFormat", + "format", + (1, 2), + (0, 0.4, 0.7), + ), + _score_keyword_match( + user_text, + config["reference_keywords"], + "referenceComplexity", + "references", + (1, 2), + (0, 0.3, 0.5), + ), + _score_keyword_match( + user_text, + config["negation_keywords"], + "negationComplexity", + "negation", + (2, 3), + (0, 0.3, 0.5), + ), + _score_keyword_match( + user_text, + config["domain_specific_keywords"], + "domainSpecificity", + "domain-specific", + (1, 2), + (0, 0.5, 0.8), + ), + ] + + # Score agentic task indicators — user prompt only. The system prompt + # describes assistant behavior, not the user's intent: a coding assistant + # system prompt with "edit files" / "fix bugs" should NOT force every + # request into agentic mode. + agentic_dimension, agentic_score = _score_agentic_task( + user_text, config["agentic_task_keywords"] + ) + dimensions.append(agentic_dimension) + + signals = [dimension["signal"] for dimension in dimensions if dimension["signal"] is not None] + + weights = config["dimension_weights"] + weighted_score = sum( + dimension["score"] * weights.get(dimension["name"], 0) for dimension in dimensions + ) + + # Count reasoning markers for override — only the USER prompt, so a system + # prompt saying "step by step" cannot force REASONING for simple queries. + reasoning_matches = [ + keyword for keyword in config["reasoning_keywords"] if keyword.lower() in user_text + ] + + # Direct reasoning override: 2+ reasoning markers = high confidence REASONING + if len(reasoning_matches) >= 2: + confidence = _calibrate_confidence( + max(weighted_score, 0.3), # ensure positive for confidence calc + config["confidence_steepness"], + ) + return { + "score": weighted_score, + "tier": "REASONING", + "confidence": max(confidence, 0.85), + "signals": signals, + "agentic_score": agentic_score, + "dimensions": dimensions, + } + + # Map weighted score to tier using boundaries + boundaries = config["tier_boundaries"] + simple_medium = boundaries["simple_medium"] + medium_complex = boundaries["medium_complex"] + complex_reasoning = boundaries["complex_reasoning"] + tier: Tier + if weighted_score < simple_medium: + tier = "SIMPLE" + distance_from_boundary = simple_medium - weighted_score + elif weighted_score < medium_complex: + tier = "MEDIUM" + distance_from_boundary = min( + weighted_score - simple_medium, medium_complex - weighted_score + ) + elif weighted_score < complex_reasoning: + tier = "COMPLEX" + distance_from_boundary = min( + weighted_score - medium_complex, complex_reasoning - weighted_score + ) + else: + tier = "REASONING" + distance_from_boundary = weighted_score - complex_reasoning + + # Calibrate confidence via sigmoid of distance from nearest boundary + confidence = _calibrate_confidence(distance_from_boundary, config["confidence_steepness"]) + + # If confidence is below threshold → ambiguous + if confidence < config["confidence_threshold"]: + return { + "score": weighted_score, + "tier": None, + "confidence": confidence, + "signals": signals, + "agentic_score": agentic_score, + "dimensions": dimensions, + } + + return { + "score": weighted_score, + "tier": tier, + "confidence": confidence, + "signals": signals, + "agentic_score": agentic_score, + "dimensions": dimensions, + } + + +def _calibrate_confidence(distance: float, steepness: float) -> float: + """Sigmoid confidence calibration onto the [0.5, 1.0] range.""" + try: + return 1 / (1 + math.exp(-steepness * distance)) + except OverflowError: + # JS evaluates exp() to Infinity here and collapses to 0; Python raises. + return 0.0 diff --git a/blockrun_llm/router_core/selector.py b/blockrun_llm/router_core/selector.py new file mode 100644 index 0000000..7546465 --- /dev/null +++ b/blockrun_llm/router_core/selector.py @@ -0,0 +1,244 @@ +""" +Tier → Model Selection + +Python port of ``@blockrun/router-core`` ``selector.ts``. + +Maps a classification tier to the cheapest capable model and builds +RoutingDecision metadata with cost estimates and savings. +""" + +from __future__ import annotations + +from collections.abc import Callable, Iterable, Mapping + +from .types import Capacity, Method, ModelPricing, RoutingDecision, Tier, TierConfig + +# The savings baseline is a price anchor, not "the current flagship" — it is +# deliberately NOT bumped every time a new Opus ships. Opus 4.7, 4.8 and 5 all +# bill $5/$25, so moving it would change no reported number while breaking +# comparability with historical journal entries. Only move it if the Opus tier +# itself is repriced. +BASELINE_MODEL_ID = "anthropic/claude-opus-4.7" + +# Hardcoded fallback: Claude Opus 4.7 pricing (per 1M tokens), used when the +# baseline model is absent from the dynamic pricing map. +BASELINE_INPUT_PRICE = 5.0 +BASELINE_OUTPUT_PRICE = 25.0 + +# Server-side margin applied to all x402 payments (must match the blockrun +# server's MARGIN_PERCENT). +SERVER_MARGIN_PERCENT = 5 +# Minimum payment enforced by the CDP Facilitator (must match the blockrun +# server's MIN_PAYMENT_USD). +MIN_PAYMENT_USD = 0.001 + + +def _flat_price(pricing: ModelPricing | None) -> float | None: + """Active promo flat price, or ``None`` for per-token billing. + + The catalog reports ``flat_price: 0`` for per-token models where the + TypeScript host omits the field, so a falsy value means "not flat". + """ + if not pricing: + return None + flat = pricing.get("flat_price") + return float(flat) if flat else None + + +def _baseline_cost( + model_pricing: Mapping[str, ModelPricing], + estimated_input_tokens: int, + max_output_tokens: int, +) -> float: + """What the premium reference model would cost for the same request.""" + opus_pricing = model_pricing.get(BASELINE_MODEL_ID) + opus_input_price = (opus_pricing or {}).get("input_price", BASELINE_INPUT_PRICE) + opus_output_price = (opus_pricing or {}).get("output_price", BASELINE_OUTPUT_PRICE) + baseline_input = (estimated_input_tokens / 1_000_000) * opus_input_price + baseline_output = (max_output_tokens / 1_000_000) * opus_output_price + return baseline_input + baseline_output + + +def _savings(cost_estimate: float, baseline_cost: float, routing_profile: str | None) -> float: + # Premium profile doesn't calculate savings (it's about quality, not cost). + if routing_profile == "premium": + return 0.0 + if baseline_cost > 0: + return max(0.0, (baseline_cost - cost_estimate) / baseline_cost) + return 0.0 + + +def select_model( + tier: Tier, + confidence: float, + method: Method, + reasoning: str, + tier_configs: Mapping[str, TierConfig], + model_pricing: Mapping[str, ModelPricing], + estimated_input_tokens: int, + max_output_tokens: int, + routing_profile: str | None = None, + agentic_score: float | None = None, +) -> RoutingDecision: + """Select the primary model for a tier and build the RoutingDecision.""" + tier_config = tier_configs[tier] + model = tier_config["primary"] + pricing = model_pricing.get(model) + + flat = _flat_price(pricing) + if flat is not None: + cost_estimate = flat + else: + input_price = (pricing or {}).get("input_price", 0) + output_price = (pricing or {}).get("output_price", 0) + cost_estimate = (estimated_input_tokens / 1_000_000) * input_price + ( + max_output_tokens / 1_000_000 + ) * output_price + + baseline_cost = _baseline_cost(model_pricing, estimated_input_tokens, max_output_tokens) + + decision: RoutingDecision = { + "model": model, + "tier": tier, + "confidence": confidence, + "method": method, + "reasoning": reasoning, + "cost_estimate": cost_estimate, + "baseline_cost": baseline_cost, + "savings": _savings(cost_estimate, baseline_cost, routing_profile), + } + if agentic_score is not None: + decision["agentic_score"] = agentic_score + return decision + + +def get_fallback_chain(tier: Tier, tier_configs: Mapping[str, TierConfig]) -> list[str]: + """Get the ordered fallback chain for a tier: ``[primary, *fallbacks]``.""" + config = tier_configs[tier] + return [config["primary"], *config["fallback"]] + + +def calculate_model_cost( + model: str, + model_pricing: Mapping[str, ModelPricing], + estimated_input_tokens: int, + max_output_tokens: int, + routing_profile: str | None = None, +) -> dict[str, float]: + """Calculate cost for a specific model (used when a fallback model is used). + + Includes the server margin and the facilitator minimum so the estimate + matches the actual x402 charge. + """ + pricing = model_pricing.get(model) + + flat = _flat_price(pricing) + if flat is not None: + # Active promo: fixed cost per request + cost_estimate = max(flat * (1 + SERVER_MARGIN_PERCENT / 100), MIN_PAYMENT_USD) + else: + # Defensive: guard against undefined price fields (not just absent pricing) + input_price = (pricing or {}).get("input_price", 0) + output_price = (pricing or {}).get("output_price", 0) + input_cost = (estimated_input_tokens / 1_000_000) * input_price + output_cost = (max_output_tokens / 1_000_000) * output_price + cost_estimate = max( + (input_cost + output_cost) * (1 + SERVER_MARGIN_PERCENT / 100), MIN_PAYMENT_USD + ) + + baseline_cost = _baseline_cost(model_pricing, estimated_input_tokens, max_output_tokens) + return { + "cost_estimate": cost_estimate, + "baseline_cost": baseline_cost, + "savings": _savings(cost_estimate, baseline_cost, routing_profile), + } + + +def filter_by_tool_calling( + models: list[str], + has_tools: bool, + supports_tool_calling: Callable[[str], bool], +) -> list[str]: + """Keep only models that support tool calling when the request has tools. + + When every model lacks tool calling the full list is returned unchanged — + better to let the API error than to produce an empty chain. + """ + if not has_tools: + return models + filtered = [model for model in models if supports_tool_calling(model)] + return filtered if filtered else models + + +def filter_by_vision( + models: list[str], + has_vision: bool, + supports_vision: Callable[[str], bool], +) -> list[str]: + """Keep only vision-capable models when the request carries images. + + Same empty-chain safety net as :func:`filter_by_tool_calling`. + """ + if not has_vision: + return models + filtered = [model for model in models if supports_vision(model)] + return filtered if filtered else models + + +def filter_by_exclude_list(models: list[str], exclude_list: Iterable[str]) -> list[str]: + """Remove user-excluded models, with the same empty-chain safety net.""" + excluded = set(exclude_list) + if not excluded: + return models + filtered = [model for model in models if model not in excluded] + return filtered if filtered else models + + +def get_fallback_chain_filtered( + tier: Tier, + tier_configs: Mapping[str, TierConfig], + estimated_total_tokens: int, + get_context_window: Callable[[str], int | None], +) -> list[str]: + """Get the tier's fallback chain filtered by context length. + + Models with an unknown context window are kept (let the API reject them), + and an entirely filtered-out chain falls back to the full chain. + """ + full_chain = get_fallback_chain(tier, tier_configs) + + filtered = [] + for model_id in full_chain: + context_window = get_context_window(model_id) + # Unknown model - include it (let API reject if needed) + # Add 10% buffer for safety + if context_window is None or context_window >= estimated_total_tokens * 1.1: + filtered.append(model_id) + + return filtered if filtered else full_chain + + +def filter_candidates_by_capacity( + models: list[str], + estimated_input_tokens: int, + requested_output_tokens: int, + get_capabilities: Callable[[str], Capacity | None], +) -> list[str]: + """Filter an already-ranked candidate list by context and output capacity. + + Unlike :func:`get_fallback_chain_filtered` this supports the V3 portfolio + order and returns an empty list when nothing fits. + """ + filtered = [] + for model_id in models: + capabilities = get_capabilities(model_id) + if not capabilities: + filtered.append(model_id) + continue + if ( + capabilities["context_window"] + >= (estimated_input_tokens + requested_output_tokens) * 1.1 + and capabilities["max_output"] >= requested_output_tokens + ): + filtered.append(model_id) + return filtered diff --git a/blockrun_llm/router_core/strategy.py b/blockrun_llm/router_core/strategy.py new file mode 100644 index 0000000..4cfd156 --- /dev/null +++ b/blockrun_llm/router_core/strategy.py @@ -0,0 +1,308 @@ +""" +Router Strategy Registry + +Python port of ``@blockrun/router-core`` ``strategy.ts``. + +Pluggable strategy system for request routing. +Default: RulesStrategy — identical to the original inline route() logic, <1ms. +""" + +from __future__ import annotations + +import copy +import math +from collections.abc import Sequence +from datetime import datetime +from typing import Protocol + +from ._js import as_utc, js_regex, parse_date, to_fixed +from .rules import classify_by_rules +from .selector import select_model +from .types import ( + TIER_RANK, + Profile, + Promotion, + RouterOptions, + RoutingDecision, + Tier, + TierConfig, +) + +_STRUCTURED_OUTPUT = js_regex(r"json|structured|schema", ignorecase=True) + + +class RouterStrategy(Protocol): + """Interface implemented by every routing strategy.""" + + name: str + + def route( + self, + prompt: str, + system_prompt: str | None, + max_output_tokens: int, + options: RouterOptions, + ) -> RoutingDecision: ... + + +def sample_prompt(value: str, scan_limit: int) -> str: + """Sample both ends of a long prompt, keeping instructions at either edge.""" + if len(value) <= scan_limit: + return value + prefix_length = math.ceil(scan_limit / 2) + suffix_length = scan_limit - prefix_length + suffix = value[-suffix_length:] if suffix_length else value + return f"{value[:prefix_length]}\n{suffix}" + + +def scan_limit_for(options: RouterOptions) -> int: + return max(1, min(8_000, options["config"]["classifier"]["prompt_truncation_chars"])) + + +def apply_unavailable_models( + tier_configs: dict[str, TierConfig], + unavailable_models: Sequence[str] | None, +) -> dict[str, TierConfig]: + """Remove host-declared-dead models from every tier chain. + + Promotes the first surviving rung to primary. A tier whose chain is + entirely dead keeps its original config — the router has nothing live to + offer there, and inventing a model would hide the outage from the host + that reported it. + """ + if not unavailable_models: + return tier_configs + dead = set(unavailable_models) + result = tier_configs + for tier, config in tier_configs.items(): + alive = [model for model in [config["primary"], *config["fallback"]] if model not in dead] + if not alive or ( + alive[0] == config["primary"] and len(alive) == len(config["fallback"]) + 1 + ): + continue + if result is tier_configs: + result = dict(tier_configs) + result[tier] = {"primary": alive[0], "fallback": alive[1:]} + return result + + +def apply_promotions( + tier_configs: dict[str, TierConfig], + promotions: list[Promotion] | None, + profile: Profile, + now: datetime | None = None, +) -> dict[str, TierConfig]: + """Apply active time-windowed promotions to tier configs. + + Returns a new tier-config mapping with promotion overrides merged in. + Expired or not-yet-active promotions are ignored. + """ + if not promotions: + return tier_configs + + current = now if now is not None else as_utc(None) + result = tier_configs + for promo in promotions: + start = parse_date(promo.get("start_date", "")) + end = parse_date(promo.get("end_date", "")) + if start is None or end is None: + continue + if current < start or current >= end: + continue + + profiles = promo.get("profiles") + if profiles and profile not in profiles: + continue + + # Shallow-clone on first mutation + if result is tier_configs: + result = {tier: copy.copy(config) for tier, config in tier_configs.items()} + + for tier, override in promo.get("tier_overrides", {}).items(): + if tier not in result: + continue + primary = override.get("primary") + fallback = override.get("fallback") + if primary: + result[tier]["primary"] = primary + if fallback: + result[tier]["fallback"] = fallback + + return result + + +class RulesStrategy: + """Rules-based routing strategy. + + Attaches ``tier_configs`` and ``profile`` to the decision for downstream use. + """ + + name = "rules" + + def route( + self, + prompt: str, + system_prompt: str | None, + max_output_tokens: int, + options: RouterOptions, + ) -> RoutingDecision: + config = options["config"] + model_pricing = options["model_pricing"] + + # Estimate input tokens (~4 chars per token) + full_text = f"{system_prompt or ''} {prompt}" + estimated_tokens = math.ceil(len(full_text) / 4) + scan_limit = scan_limit_for(options) + scanned_prompt = sample_prompt(prompt, scan_limit) + scanned_system_prompt = sample_prompt(system_prompt, scan_limit) if system_prompt else None + + # --- Rule-based classification (runs first to get agentic_score) --- + rule_result = classify_by_rules( + scanned_prompt, scanned_system_prompt, estimated_tokens, config["scoring"] + ) + + # --- Select tier configs based on routing profile --- + routing_profile = options.get("routing_profile") + profile: Profile + if routing_profile == "eco": + # `eco_tiers: None` explicitly disables the special eco tier set + # while keeping eco routing semantics. Fall back to regular tiers + # instead of dropping into auto routing (which could select agentic + # tiers). + eco_tiers = config.get("eco_tiers") + tier_configs = eco_tiers if eco_tiers else config["tiers"] + profile_suffix = " | eco" if eco_tiers else " | eco (default tiers)" + profile = "eco" + elif routing_profile == "premium": + # `premium_tiers: None` disables the premium-specific tier set but + # the request is still a premium-profile request, so use regular + # tiers while preserving premium metadata/cost semantics. + premium_tiers = config.get("premium_tiers") + tier_configs = premium_tiers if premium_tiers else config["tiers"] + profile_suffix = " | premium" if premium_tiers else " | premium (default tiers)" + profile = "premium" + else: + # Auto profile (or unset): intelligent routing with agentic detection. + # + # `agentic_mode` semantics: + # - True -> force agentic tiers (ignore heuristics) + # - False -> disable agentic tiers entirely (even if tools present) + # - unset -> auto-detect via heuristics (tools present OR high + # agentic score) + agentic_score = rule_result.get("agentic_score", 0) or 0 + is_auto_agentic = agentic_score >= 0.5 + agentic_mode_setting = config["overrides"].get("agentic_mode") + requires_tools = options.get("requires_tools") + has_tools_in_request = ( + requires_tools if requires_tools is not None else options.get("has_tools", False) + ) + agentic_tiers = config.get("agentic_tiers") + if agentic_mode_setting is False: + # Explicitly disabled — never use agentic tiers + use_agentic_tiers = False + elif agentic_mode_setting is True: + # Explicitly enabled — use agentic tiers if available + use_agentic_tiers = agentic_tiers is not None + else: + use_agentic_tiers = bool( + (has_tools_in_request or is_auto_agentic) and agentic_tiers is not None + ) + if use_agentic_tiers and agentic_tiers is not None: + tier_configs = agentic_tiers + profile_suffix = f" | agentic{' (tools)' if has_tools_in_request else ''}" + profile = "agentic" + else: + tier_configs = config["tiers"] + profile_suffix = "" + profile = "auto" + + # Apply time-windowed promotions + now = as_utc(options.get("now")) + tier_configs = apply_promotions(tier_configs, config.get("promotions"), profile, now) + + # Hard-remove models the host has observed dead at the gateway. After + # promotions, so a promo cannot resurrect a rung the host just killed. + tier_configs = apply_unavailable_models(tier_configs, options.get("unavailable_models")) + + agentic_score_value = rule_result.get("agentic_score") + + # --- Override: large context → force COMPLEX --- + force_complex_at = config["overrides"]["max_tokens_force_complex"] + if estimated_tokens > force_complex_at: + decision = select_model( + "COMPLEX", + 0.95, + "rules", + f"Input exceeds {force_complex_at} tokens{profile_suffix}", + tier_configs, + model_pricing, + estimated_tokens, + max_output_tokens, + routing_profile, + agentic_score_value, + ) + decision["tier_configs"] = tier_configs + decision["profile"] = profile + return decision + + # Structured output detection + has_structured_output = options.get("requires_structured_output") is True or ( + bool(_STRUCTURED_OUTPUT.search(scanned_system_prompt)) + if scanned_system_prompt + else False + ) + + tier: Tier + signals = ", ".join(rule_result.get("signals", [])) + reasoning = f"score={to_fixed(rule_result['score'], 2)} | {signals}" + + if rule_result.get("tier") is not None: + tier = rule_result["tier"] # type: ignore[assignment] + confidence = rule_result["confidence"] + else: + # Ambiguous — default to configurable tier (no external API call) + tier = config["overrides"]["ambiguous_default_tier"] + confidence = 0.5 + reasoning += f" | ambiguous -> default: {tier}" + + # Apply structured output minimum tier + if has_structured_output: + min_tier = config["overrides"]["structured_output_min_tier"] + if TIER_RANK[tier] < TIER_RANK[min_tier]: + reasoning += f" | upgraded to {min_tier} (structured output)" + tier = min_tier + + # Add routing profile suffix to reasoning + reasoning += profile_suffix + + decision = select_model( + tier, + confidence, + "rules", + reasoning, + tier_configs, + model_pricing, + estimated_tokens, + max_output_tokens, + routing_profile, + agentic_score_value, + ) + decision["tier_configs"] = tier_configs + decision["profile"] = profile + return decision + + +# --- Strategy Registry --- + +_registry: dict[str, RouterStrategy] = {"rules": RulesStrategy()} + + +def get_strategy(name: str) -> RouterStrategy: + strategy = _registry.get(name) + if strategy is None: + raise ValueError(f"Unknown routing strategy: {name}") + return strategy + + +def register_strategy(strategy: RouterStrategy) -> None: + _registry[strategy.name] = strategy diff --git a/blockrun_llm/router_core/tool_intent.py b/blockrun_llm/router_core/tool_intent.py new file mode 100644 index 0000000..5be2fb3 --- /dev/null +++ b/blockrun_llm/router_core/tool_intent.py @@ -0,0 +1,72 @@ +""" +Whether the request actually requires an external action/tool, as distinct +from merely being sent by a host that exposes tools on every turn. + +Python port of ``@blockrun/router-core`` ``tool-intent.ts``. + +The detector intentionally looks for action+target pairs. A generic factual or +multiple-choice question must stay false even when the host attaches a large +tool schema; otherwise every tool-enabled host turn is over-routed as an agent +task and models may browse or mutate state unnecessarily. +""" + +from __future__ import annotations + +from typing import Any + +from ._js import js_regex + +# System prompts commonly describe every tool a host exposes. They are not +# evidence that the user asked to perform an action on this turn. Explicit host +# requirements should use tool_choice / requires_tools instead. +_EXPLICIT_TOOL = js_regex( + r"\b(?:use|call|invoke)\s+(?:the\s+)?[\w.-]+\s+(?:tool|function|api)\b|\btool[_ -]?call\b" + r"|使用.{0,20}(?:工具|函数|接口)|调用.{0,20}(?:工具|函数|接口)", + ignorecase=True, +) +_CODE_ENVIRONMENT = js_regex( + r"\b(?:run|execute)\s+(?:the\s+)?(?:tests?|command|script|build|linter)" + r"|\b(?:edit|modify|patch|create|write|save|delete|rename|move|inspect|read)\b.{0,60}" + r"\b(?:file|repository|repo|codebase|directory|folder)\b" + r"|\b(?:terminal|shell|bash|zsh|pytest|npm test|pnpm test|git\s+(?:status|diff|commit)|docker)\b" + r"|(?:运行|执行).{0,20}(?:测试|命令|脚本|构建)" + r"|(?:修改|编辑|修复|创建|读取|检查|保存).{0,30}(?:文件|仓库|代码库|目录)", + ignorecase=True, +) +_WEB_ACTION = js_regex( + r"\b(?:browse|search|look up|fetch|open)\b.{0,80}" + r"\b(?:web|website|url|online|documentation|docs|news|weather|price)\b" + r"|(?:浏览|搜索|查询|打开).{0,30}(?:网页|网站|链接|文档|新闻|天气|价格)", + ignorecase=True, +) +_STATEFUL_ACTION = js_regex( + r"\b(?:refund|cancel|book|reserve|purchase|buy|return|exchange|transfer|update|change)\b.{0,80}" + r"\b(?:order|booking|reservation|account|address|payment|subscription|ticket|flight|item)\b" + r"|(?:退款|取消|预订|购买|退货|换货|转账|更新|修改).{0,30}" + r"(?:订单|预订|账户|地址|付款|订阅|票|航班|商品)", + ignorecase=True, +) + + +def infer_tool_requirement( + prompt: str, + system_prompt: str | None = None, + tool_choice: Any = None, +) -> bool: + """Return ``True`` when this turn actually asks for a tool action.""" + # OpenAI-compatible clients can state this requirement directly. Treat that + # protocol signal as authoritative instead of trying to infer it from prose. + if tool_choice == "none": + return False + if tool_choice == "required": + return True + if isinstance(tool_choice, dict) and tool_choice.get("type") == "function": + return True + + text = prompt + return bool( + _EXPLICIT_TOOL.search(text) + or _CODE_ENVIRONMENT.search(text) + or _WEB_ACTION.search(text) + or _STATEFUL_ACTION.search(text) + ) diff --git a/blockrun_llm/router_core/types.py b/blockrun_llm/router_core/types.py new file mode 100644 index 0000000..6ece9e7 --- /dev/null +++ b/blockrun_llm/router_core/types.py @@ -0,0 +1,316 @@ +""" +Router Core types — Python port of ``@blockrun/router-core`` ``types.ts``. + +Four classification tiers — REASONING is distinct from COMPLEX because +reasoning tasks need different models (o3, gemini-pro) than general complex +tasks (gpt-4o, sonnet-4). + +Scoring uses weighted float dimensions with sigmoid confidence calibration. + +Field names are snake_case (the upstream TypeScript uses camelCase); the +mapping is 1:1 and mechanical, e.g. ``costEstimate`` -> ``cost_estimate``. +""" + +from __future__ import annotations + +from collections.abc import Mapping, Sequence +from typing import Literal, TypedDict + +Tier = Literal["SIMPLE", "MEDIUM", "COMPLEX", "REASONING"] + +TaskType = Literal[ + "chat", + "extraction", + "code_edit", + "code_agent", + "tool_agent", + "tool_agent_parallel", + "debug", + "reasoning", + "reasoning_mcq", + "reasoning_math", + "long_context", + "vision", +] + +Profile = Literal["auto", "eco", "premium", "agentic"] + +RoutingProfile = Literal["eco", "auto", "premium"] + +Method = Literal["rules", "llm", "portfolio"] + +#: Ordering used by the structured-output minimum-tier override. +TIER_RANK: dict[str, int] = {"SIMPLE": 0, "MEDIUM": 1, "COMPLEX": 2, "REASONING": 3} + +TIERS: tuple[str, ...] = ("SIMPLE", "MEDIUM", "COMPLEX", "REASONING") + + +class ModelPricing(TypedDict, total=False): + """Catalog prices per 1M tokens. + + ``flat_price`` overrides token pricing when present and non-zero (the + BlockRun catalog reports ``0`` rather than omitting the field, so falsy + means "per-token billing" here, matching the TypeScript ``undefined``). + """ + + input_price: float + output_price: float + flat_price: float + + +class ModelCapabilities(TypedDict): + context_window: int + max_output_tokens: int + supports_tools: bool + supports_vision: bool + + +class Capacity(TypedDict): + """Narrow capability view used by :func:`filter_candidates_by_capacity`.""" + + context_window: int + max_output: int + + +class ModelPerformanceProfile(TypedDict, total=False): + measured_at: str + #: Gateway end-to-end latency for the benchmark workload. + latency_ms: float + #: Tail latency is more relevant than mean latency for urgent requests. + p95_latency_ms: float + output_tokens_per_second: float + #: External intelligence index when one was available; not task success. + intelligence_index: float + #: Failure fraction observed in the same benchmark run. + error_rate: float + #: Number of sampled calls behind the observation. + samples: int + + +class TierConfig(TypedDict): + primary: str + fallback: list[str] + + +class DimensionScore(TypedDict): + name: str + score: float + signal: str | None + + +class ScoringResult(TypedDict, total=False): + #: weighted float (roughly [-0.3, 0.4]) + score: float + #: ``None`` = ambiguous, needs fallback classifier + tier: Tier | None + #: sigmoid-calibrated [0, 1] + confidence: float + signals: list[str] + #: 0-1 agentic task score for auto-switching to agentic tiers + agentic_score: float + #: per-dimension breakdown for /debug + dimensions: list[DimensionScore] + + +class CandidateScore(TypedDict): + model: str + score: float + quality: float + cost: float + speed: float + reliability: float + + +class _RoutingDecisionRequired(TypedDict): + model: str + tier: Tier + confidence: float + method: Method + reasoning: str + cost_estimate: float + baseline_cost: float + savings: float # 0-1 percentage + + +class RoutingDecision(_RoutingDecisionRequired, total=False): + #: 0-1 agentic task score (present when tier routing used) + agentic_score: float + #: Which tier configs were used (auto/eco/premium/agentic) + tier_configs: dict[str, TierConfig] + #: Which routing profile was applied + profile: Profile + #: Ordered, capability-eligible candidates. The first entry is ``model``. + candidates: list[str] + #: Explainable request classification used by the portfolio router. + task_type: TaskType + #: Router implementation that made the selection. + router_version: Literal["v2-rules", "v3-portfolio"] + #: Explainable local portfolio score breakdown, ordered with ``candidates``. + candidate_scores: list[CandidateScore] + + +class TokenCountThresholds(TypedDict): + simple: int + complex: int + + +class TierBoundaries(TypedDict): + simple_medium: float + medium_complex: float + complex_reasoning: float + + +class ScoringConfig(TypedDict): + token_count_thresholds: TokenCountThresholds + code_keywords: list[str] + reasoning_keywords: list[str] + simple_keywords: list[str] + technical_keywords: list[str] + creative_keywords: list[str] + imperative_verbs: list[str] + constraint_indicators: list[str] + output_format_keywords: list[str] + reference_keywords: list[str] + negation_keywords: list[str] + domain_specific_keywords: list[str] + agentic_task_keywords: list[str] + dimension_weights: dict[str, float] + tier_boundaries: TierBoundaries + confidence_steepness: float + confidence_threshold: float + + +class ClassifierConfig(TypedDict): + llm_model: str + llm_max_tokens: int + llm_temperature: float + prompt_truncation_chars: int + cache_ttl_ms: int + + +class OverridesConfig(TypedDict, total=False): + max_tokens_force_complex: int + structured_output_min_tier: Tier + ambiguous_default_tier: Tier + #: ``True`` forces agentic tiers, ``False`` disables them, absent = auto-detect. + agentic_mode: bool | None + + +class PortfolioBandWeights(TypedDict): + quality: float + capability: float + cost: float + speed: float + reliability: float + legacy: float + + +class HighStakesBoost(TypedDict): + quality: float + reliability: float + + +class AffinityFloorGap(TypedDict): + auto: float + eco: float + premium: float + + +class PortfolioConfig(TypedDict): + auto: PortfolioBandWeights + eco: PortfolioBandWeights + premium: PortfolioBandWeights + high_stakes_boost: HighStakesBoost + latency_sensitive_speed_boost: float + #: A candidate materially below the best task affinity cannot win on cost alone. + affinity_floor_gap: AffinityFloorGap + + +class PromotionTierOverride(TypedDict, total=False): + primary: str + fallback: list[str] + + +class Promotion(TypedDict, total=False): + """Time-windowed promotion that temporarily overrides tier routing. + + Active promotions are auto-applied; expired ones are ignored at runtime. + """ + + #: Human-readable label (e.g. "GLM-5 Launch Promo") + name: str + #: ISO date string, promotion starts (inclusive). e.g. "2026-04-01" + start_date: str + #: ISO date string, promotion ends (exclusive). e.g. "2026-04-15" + end_date: str + #: Partial tier overrides merged into the active tier configs. + tier_overrides: dict[str, PromotionTierOverride] + #: Which profiles this applies to. Default: all profiles. + profiles: list[Profile] + + +class ShadowConfig(TypedDict, total=False): + strategy: Literal["rules", "portfolio"] + sample_rate: float + + +class _RoutingConfigRequired(TypedDict): + version: str + classifier: ClassifierConfig + scoring: ScoringConfig + tiers: dict[str, TierConfig] + overrides: OverridesConfig + + +class RoutingConfig(_RoutingConfigRequired, total=False): + #: Enables a one-line rollback to the established V2 rules selector. + strategy: Literal["rules", "portfolio"] + #: Locally recompute a comparison strategy without changing the served model. + shadow: ShadowConfig + #: Calibratable local portfolio scoring weights; relative, not probabilities. + portfolio: PortfolioConfig + #: Tier configs for agentic mode. ``None`` disables agentic tier selection. + agentic_tiers: dict[str, TierConfig] | None + #: Tier configs for eco profile. ``None`` falls back to ``tiers``. + eco_tiers: dict[str, TierConfig] | None + #: Tier configs for premium profile. ``None`` falls back to ``tiers``. + premium_tiers: dict[str, TierConfig] | None + #: Time-windowed promotions that temporarily override tier routing. + promotions: list[Promotion] + + +class _RouterOptionsRequired(TypedDict): + config: RoutingConfig + model_pricing: Mapping[str, ModelPricing] + + +class RouterOptions(_RouterOptionsRequired, total=False): + """Per-request routing inputs.""" + + #: Host-provided capability snapshot; overrides the core's built-in one. + model_capabilities: Mapping[str, ModelCapabilities] + routing_profile: RoutingProfile | None + has_tools: bool + #: Number of tool definitions visible to the model on this turn. + tool_count: int + #: Local tool identifiers, used only for request/tool intent matching. + tool_names: Sequence[str] + #: Tools are attached by the host and this turn needs to use them. + requires_tools: bool | None + has_vision: bool + #: ``response_format`` / JSON schema requires reliable structured output. + requires_structured_output: bool + #: Model ids the host has observed to be unavailable at the gateway (a + #: 400/404/410 on a direct call, a provider EOL). Hard-removed from every + #: chain before selection and never restored by an eligibility fail-open — + #: the operational kill-switch for a dead chain rung, usable the moment the + #: host observes the failure instead of waiting on a core release and two + #: consumer repins. Distinct from user-preference exclusion + #: (``filter_by_exclude_list``), which deliberately fail-opens rather than + #: empty a chain. + unavailable_models: Sequence[str] + #: Override current time for promotion window checks (for testing). Naive + #: values are read as UTC. ``datetime.datetime``. + now: object + #: Fresh gateway performance observations, injected off the hot path. + model_performance: Mapping[str, ModelPerformanceProfile] diff --git a/blockrun_llm/rpc.py b/blockrun_llm/rpc.py new file mode 100644 index 0000000..66d970e --- /dev/null +++ b/blockrun_llm/rpc.py @@ -0,0 +1,446 @@ +""" +BlockRun RPC Client - Multi-chain JSON-RPC (Tatum gateway) via x402 micropayments. + +One endpoint, 40+ chains: Ethereum, Base, Solana, Polygon, BSC, Arbitrum, +Optimism, Avalanche, Bitcoin, Sui, and more. Standard JSON-RPC 2.0 +passthrough — no API key, pay-per-call in USDC. + +SECURITY NOTE - Private Key Handling: +===================================== +Your private key NEVER leaves your machine. Here's what happens: + +1. Key stays local - only used to sign an EIP-712 typed data message +2. Only the SIGNATURE is sent in the PAYMENT-SIGNATURE header +3. BlockRun verifies the signature on-chain via Coinbase CDP facilitator + +Usage: + from blockrun_llm import RpcClient + + client = RpcClient() # Uses BLOCKRUN_WALLET_KEY from env + + # EVM chains speak eth_* JSON-RPC + block = client.call("ethereum", "eth_blockNumber") + print(block.result) # e.g. "0x1499f7c" + + balance = client.call( + "base", "eth_getBalance", + ["0x4200000000000000000000000000000000000006", "latest"], + ) + + # Non-EVM chains speak their native JSON-RPC + slot = client.call("solana", "getSlot") + + # Batch: one payment, per-element pricing ($0.002 x N) + responses = client.batch("polygon", [ + {"method": "eth_blockNumber"}, + {"method": "eth_gasPrice"}, + ]) + +Pricing: + Flat $0.002 per JSON-RPC call; a batch charges per element. + +Networks: + 40 curated chains (see SUPPORTED_NETWORKS) plus common aliases + (eth, arb, op, matic, bnb, avax, sol, btc, xrp, dot, ...). Unknown but + well-formed slugs fall through to a generic `{slug}-mainnet` gateway + attempt, so new Tatum chains work without an SDK update. +""" + +from __future__ import annotations + +import os +from typing import Any + +import httpx +from dotenv import load_dotenv +from eth_account import Account + +from .apikey import ( + api_key_base_url, + auth_headers, + missing_credential_error, + payment_mode, + raise_for_api_key_402, + resolve_api_key, +) +from .tx_log import paid_request_error_prefix +from .types import APIError, PaymentError, RpcResponse, retry_after_of +from .validation import ( + sanitize_error_response, + validate_api_url, + validate_private_key, +) +from .x402 import create_payment_payload, extract_payment_details, parse_payment_required + +load_dotenv() + +# Curated chains accepted by /v1/rpc/{network}. Mirrors backend +# TATUM_RPC_CHAINS (src/lib/tatum.ts, verified live 2026-06-07). +# EVM chains use eth_* JSON-RPC; non-EVM (Solana / UTXO / NEAR / Sui / +# XRP Ledger / Polkadot) speak their own JSON-RPC dialect. +SUPPORTED_NETWORKS = [ + # EVM + "ethereum", + "base", + "arbitrum", + "arbitrum-nova", + "optimism", + "polygon", + "bsc", + "avalanche", + "fantom", + "cronos", + "celo", + "gnosis", + "zksync", + "berachain", + "unichain", + "monad", + "chiliz", + "moonbeam", + "aurora", + "flare", + "oasis", + "kaia", + "sonic", + "xdc", + "abstract", + "hyperevm", + "plume", + "ronin", + "rootstock", + # Non-EVM (JSON-RPC-compatible) + "solana", + "bitcoin", + "litecoin", + "dogecoin", + "bitcoin-cash", + "near", + "sui", + "ripple", + "polkadot", + "kusama", + "zcash", +] + +# Common short names the gateway also accepts (resolved server-side). +NETWORK_ALIASES = { + "eth": "ethereum", + "arb": "arbitrum", + "arbitrum-one": "arbitrum", + "arb-one": "arbitrum", + "arb-nova": "arbitrum-nova", + "op": "optimism", + "matic": "polygon", + "pol": "polygon", + "bnb": "bsc", + "binance": "bsc", + "binance-smart-chain": "bsc", + "avax": "avalanche", + "ftm": "fantom", + "bera": "berachain", + "klaytn": "kaia", + "chz": "chiliz", + "hyperliquid": "hyperevm", + "rsk": "rootstock", + "sol": "solana", + "btc": "bitcoin", + "ltc": "litecoin", + "doge": "dogecoin", + "bch": "bitcoin-cash", + "xrp": "ripple", + "xrpl": "ripple", + "dot": "polkadot", + "zec": "zcash", +} + +# Flat price per JSON-RPC call (batch = N x this). Informational only — +# the actual quote always comes from the 402 challenge. +RPC_PRICE_USD = 0.002 + + +class RpcClient: + """ + BlockRun Multi-chain RPC Client. + + Standard JSON-RPC 2.0 access to 40+ chains through BlockRun's Tatum + gateway with automatic x402 micropayments on Base chain. + + Flat $0.002 per call; a JSON-RPC batch charges per element. + """ + + DEFAULT_API_URL = "https://blockrun.ai/api" + DEFAULT_TIMEOUT = 60.0 # upstream gateway timeout is 20s + + def __init__( + self, + private_key: str | None = None, + api_url: str | None = None, + timeout: float = DEFAULT_TIMEOUT, + ): + """ + Initialize the BlockRun RPC client. + + Args: + private_key: EVM wallet private key (or set BLOCKRUN_WALLET_KEY env var) + api_url: API endpoint URL (default: https://blockrun.ai/api) + timeout: Request timeout in seconds (default: 60) + """ + from .wallet import load_wallet + + # Account rail first, and before any wallet variable is read: an API + # key is a complete credential on its own, so demanding a private key + # alongside it would make every API-key user invent a wallet they + # never use. See apikey.py for the precedence rule. + api_key = resolve_api_key(private_key) + key = ( + None + if api_key + else ( + private_key + or os.environ.get("BLOCKRUN_WALLET_KEY") + or os.environ.get("BASE_CHAIN_WALLET_KEY") + or load_wallet() + ) + ) + if not api_key and not key: + raise missing_credential_error() + + if key: + validate_private_key(key) + self.api_key = api_key + # No wallet on the account rail: nothing is signed locally. + self.account = Account.from_key(key) if key else None + + # BLOCKRUN_API_URL names an x402 gateway; an API-key client must not + # follow it and hand the key to a host set up for another rail. + api_url_raw = ( + api_key_base_url(api_url) + if api_key + else (api_url or os.environ.get("BLOCKRUN_API_URL") or self.DEFAULT_API_URL) + ) + validate_api_url(api_url_raw) + self.api_url = api_url_raw.rstrip("/") + + self.timeout = timeout + self._client = httpx.Client(headers=auth_headers(api_key), timeout=timeout) + + def call( + self, + network: str, + method: str, + params: list[Any] | None = None, + *, + id: str | int = 1, + ) -> RpcResponse: + """ + Make a single JSON-RPC 2.0 call. Flat $0.002. + + Args: + network: Chain name (e.g. "ethereum", "base", "solana") or a + common alias ("eth", "sol", "matic", ...). See + SUPPORTED_NETWORKS / NETWORK_ALIASES. + method: Chain RPC method, e.g. "eth_blockNumber", "eth_call", + "eth_getBalance" (EVM) or "getSlot", "getAccountInfo" + (Solana). + params: Method-specific params array (optional). + id: JSON-RPC request id (default: 1). + + Returns: + RpcResponse with `result` (or JSON-RPC `error`), plus + `network`, `cache_hit` and `tx_hash` metadata. + + Raises: + PaymentError: If wallet has insufficient balance + APIError: If the API returns an error + + Example: + block = client.call("ethereum", "eth_blockNumber") + print(int(block.result, 16)) + """ + body: dict[str, Any] = {"jsonrpc": "2.0", "id": id, "method": method} + if params is not None: + body["params"] = params + + data, headers = self._request_with_payment(network, body) + return self._to_response(data, headers) + + def batch( + self, + network: str, + requests: list[dict[str, Any]], + ) -> list[RpcResponse]: + """ + Make a JSON-RPC 2.0 batch call. Priced per element ($0.002 x N). + + Args: + network: Chain name or alias (see call()). + requests: List of dicts each with a "method" key and optional + "params" / "id". "jsonrpc" and missing ids are + filled in automatically. + + Returns: + List of RpcResponse, in upstream order. + + Example: + out = client.batch("base", [ + {"method": "eth_blockNumber"}, + {"method": "eth_gasPrice"}, + ]) + """ + if not requests: + raise ValueError("batch requires at least one request") + body = [] + for i, req in enumerate(requests): + if "method" not in req: + raise ValueError(f"batch request {i} is missing 'method'") + entry = {"jsonrpc": "2.0", "id": i + 1, **req} + body.append(entry) + + data, headers = self._request_with_payment(network, body) + if not isinstance(data, list): + # Upstream collapsed the batch (shouldn't happen) — wrap it. + data = [data] + return [self._to_response(item, headers) for item in data] + + @staticmethod + def _to_response(data: Any, headers: httpx.Headers) -> RpcResponse: + if not isinstance(data, dict): + data = {"result": data} + return RpcResponse( + **data, + network=headers.get("x-network"), + cache_hit=headers.get("x-cache", "").upper() == "HIT", + tx_hash=headers.get("x-payment-receipt"), + ) + + def _request_with_payment( + self, network: str, body: dict[str, Any] | list[dict[str, Any]] + ) -> tuple: + """POST the JSON-RPC body with automatic x402 payment handling.""" + endpoint = f"/v1/rpc/{network}" + url = f"{self.api_url}{endpoint}" + + response = self._client.post( + url, + json=body, + headers={"Content-Type": "application/json"}, + ) + + if response.status_code == 402: + # Account rail: a 402 is the account being out of credit, not a + # challenge to sign. Nothing here can sign, so say so plainly. + raise_for_api_key_402(response, self.api_key) + return self._handle_payment_and_retry(url, endpoint, body, response) + + if response.status_code != 200: + try: + error_body = response.json() + except Exception: + error_body = {"error": "Request failed"} + raise APIError( + f"API error: {response.status_code}", + response.status_code, + sanitize_error_response(error_body), + retry_after=retry_after_of(response), + ) + + return response.json(), response.headers + + def _handle_payment_and_retry( + self, + url: str, + endpoint: str, + body: dict[str, Any] | list[dict[str, Any]], + response: httpx.Response, + ) -> tuple: + """Handle 402 response: parse requirements, sign payment, retry.""" + payment_header = response.headers.get("payment-required") + if not payment_header: + try: + resp_body = response.json() + if "x402" in resp_body: + payment_header = resp_body + except Exception: + pass + + if not payment_header: + raise PaymentError("402 response but no payment requirements found") + + if isinstance(payment_header, str): + payment_required = parse_payment_required(payment_header) + else: + payment_required = payment_header + + details = extract_payment_details(payment_required) + resource = details.get("resource") or {} + extensions = payment_required.get("extensions", {}) + + payment_payload = create_payment_payload( + account=self.account, + recipient=details["recipient"], + amount=details["amount"], + network=details.get("network", "eip155:8453"), + resource_url=resource.get("url", f"{self.api_url}{endpoint}"), + resource_description=resource.get("description", "BlockRun Multi-chain RPC"), + max_timeout_seconds=details.get("maxTimeoutSeconds", 300), + extra=details.get("extra"), + asset=details.get("asset"), + extensions=extensions, + ) + + retry_response = self._client.post( + url, + json=body, + headers={ + "Content-Type": "application/json", + "PAYMENT-SIGNATURE": payment_payload, + }, + ) + + if retry_response.status_code == 402: + # Account rail: a 402 is the account being out of credit, not a + # challenge to sign. Nothing here can sign, so say so plainly. + raise_for_api_key_402(retry_response, self.api_key) + raise PaymentError("Payment was rejected. Check your wallet balance.") + + if retry_response.status_code != 200: + try: + error_body = retry_response.json() + except Exception: + error_body = {"error": "Request failed"} + raise APIError( + f"{paid_request_error_prefix(retry_response.headers)}: {retry_response.status_code}", + retry_response.status_code, + sanitize_error_response(error_body), + retry_after=retry_after_of(retry_response), + ) + + return retry_response.json(), retry_response.headers + + @property + def payment_mode(self) -> str: + """Which rail this client pays on: ``"apikey"`` or ``"wallet"``. + + Worth checking once at startup when both a key and a wallet are + configured in the environment: it is the difference between + spending credit and spending USDC.""" + return payment_mode(self) + + def get_wallet_address(self) -> str: + # No address on the account rail: payment comes from prepaid + # credit, so there is nothing to return but the empty string. + if self.api_key: + return "" + """Get the wallet address being used for payments.""" + return self.account.address + + def close(self): + """Close the HTTP client.""" + self._client.close() + + def __enter__(self): + return self + + def __exit__(self, exc_type, exc_val, exc_tb): + self.close() diff --git a/blockrun_llm/search.py b/blockrun_llm/search.py new file mode 100644 index 0000000..38df411 --- /dev/null +++ b/blockrun_llm/search.py @@ -0,0 +1,261 @@ +""" +BlockRun Search Client - Standalone Grok Live Search via x402 micropayments. + +Backend endpoint: POST /api/v1/search +Pricing: $0.025/source + margin (default 10 sources ≈ $0.26) + +Usage: + from blockrun_llm import SearchClient + + client = SearchClient() + result = client.search("Latest news on x402 adoption", sources=["x", "web"]) + print(result.summary) + for citation in (result.citations or []): + print(citation) + +SECURITY NOTE - Private Key Handling: +===================================== +Your private key NEVER leaves your machine. Only EIP-712 signatures are sent +in the PAYMENT-SIGNATURE header. +""" + +from __future__ import annotations + +import os +from typing import Any, Literal + +import httpx +from dotenv import load_dotenv +from eth_account import Account +from typing_extensions import Self + +from .apikey import ( + api_key_base_url, + auth_headers, + missing_credential_error, + payment_mode, + raise_for_api_key_402, + resolve_api_key, +) +from .tx_log import paid_request_error_prefix +from .types import APIError, PaymentError, SearchResult, retry_after_of +from .validation import ( + sanitize_error_response, + validate_api_url, + validate_private_key, +) +from .x402 import create_payment_payload, extract_payment_details, parse_payment_required + +load_dotenv() + +SearchSourceLiteral = Literal["x", "web", "news"] + + +class SearchClient: + """ + BlockRun Search Client. + + Calls the standalone `/v1/search` endpoint which routes through Grok Live + Search and returns a synthesized summary plus citations. Each source used + costs $0.025 (plus margin). + """ + + DEFAULT_API_URL = "https://blockrun.ai/api" + DEFAULT_TIMEOUT = 60.0 + DEFAULT_MAX_RESULTS = 10 + + def __init__( + self, + private_key: str | None = None, + api_url: str | None = None, + timeout: float = DEFAULT_TIMEOUT, + ): + from .wallet import load_wallet + + # Account rail first, and before any wallet variable is read: an API + # key is a complete credential on its own, so demanding a private key + # alongside it would make every API-key user invent a wallet they + # never use. See apikey.py for the precedence rule. + api_key = resolve_api_key(private_key) + key = ( + None + if api_key + else ( + private_key + or os.environ.get("BLOCKRUN_WALLET_KEY") + or os.environ.get("BASE_CHAIN_WALLET_KEY") + or load_wallet() + ) + ) + if not api_key and not key: + raise missing_credential_error() + + if key: + validate_private_key(key) + self.api_key = api_key + # No wallet on the account rail: nothing is signed locally. + self.account = Account.from_key(key) if key else None + + # BLOCKRUN_API_URL names an x402 gateway; an API-key client must not + # follow it and hand the key to a host set up for another rail. + api_url_raw = ( + api_key_base_url(api_url) + if api_key + else (api_url or os.environ.get("BLOCKRUN_API_URL") or self.DEFAULT_API_URL) + ) + validate_api_url(api_url_raw) + self.api_url = api_url_raw.rstrip("/") + + self.timeout = timeout + self._client = httpx.Client(headers=auth_headers(api_key), timeout=timeout) + + def search( + self, + query: str, + *, + sources: list[SearchSourceLiteral] | None = None, + max_results: int = DEFAULT_MAX_RESULTS, + from_date: str | None = None, + to_date: str | None = None, + ) -> SearchResult: + """ + Run a live search query. + + Args: + query: Search query (1-1000 chars). + sources: Subset of ["x", "web", "news"] (default: ["x", "web"]). + max_results: 1-50 (default 10). Price scales with this. + from_date, to_date: YYYY-MM-DD filters (optional). + + Returns: + SearchResult with summary, citations, and sources_used. + """ + if not query or len(query) > 1000: + raise ValueError("query must be 1-1000 characters") + if not 1 <= max_results <= 50: + raise ValueError("max_results must be between 1 and 50") + + body: dict[str, Any] = { + "query": query, + "max_results": max_results, + } + if sources is not None: + body["sources"] = sources + if from_date is not None: + body["from_date"] = from_date + if to_date is not None: + body["to_date"] = to_date + + data = self._request_with_payment("/v1/search", body) + return SearchResult(**data) + + def _request_with_payment(self, endpoint: str, body: dict[str, Any]) -> dict[str, Any]: + url = f"{self.api_url}{endpoint}" + response = self._client.post(url, json=body, headers={"Content-Type": "application/json"}) + if response.status_code == 402: + # Account rail: a 402 is the account being out of credit, not a + # challenge to sign. Nothing here can sign, so say so plainly. + raise_for_api_key_402(response, self.api_key) + return self._handle_payment_and_retry(url, body, response) + if response.status_code != 200: + try: + error_body = response.json() + except Exception: + error_body = {"error": "Request failed"} + raise APIError( + f"API error: {response.status_code}", + response.status_code, + sanitize_error_response(error_body), + retry_after=retry_after_of(response), + ) + return response.json() + + def _handle_payment_and_retry( + self, + url: str, + body: dict[str, Any], + response: httpx.Response, + ) -> dict[str, Any]: + payment_header: Any = response.headers.get("payment-required") + if not payment_header: + try: + resp_body = response.json() + if "x402" in resp_body or "accepts" in resp_body: + payment_header = resp_body + except Exception: + pass + if not payment_header: + raise PaymentError("402 response but no payment requirements found") + + if isinstance(payment_header, str): + payment_required = parse_payment_required(payment_header) + else: + payment_required = payment_header + + details = extract_payment_details(payment_required) + resource = details.get("resource") or {} + extensions = payment_required.get("extensions", {}) + + payment_payload = create_payment_payload( + account=self.account, + recipient=details["recipient"], + amount=details["amount"], + network=details.get("network", "eip155:8453"), + resource_url=resource.get("url", url), + resource_description=resource.get("description", "BlockRun Search"), + max_timeout_seconds=details.get("maxTimeoutSeconds", 300), + extra=details.get("extra"), + asset=details.get("asset"), + extensions=extensions, + ) + + retry = self._client.post( + url, + json=body, + headers={ + "Content-Type": "application/json", + "PAYMENT-SIGNATURE": payment_payload, + }, + ) + if retry.status_code == 402: + # Account rail: a 402 is the account being out of credit, not a + # challenge to sign. Nothing here can sign, so say so plainly. + raise_for_api_key_402(retry, self.api_key) + raise PaymentError("Payment was rejected. Check your wallet balance.") + if retry.status_code != 200: + try: + error_body = retry.json() + except Exception: + error_body = {"error": "Request failed"} + raise APIError( + f"{paid_request_error_prefix(retry.headers)}: {retry.status_code}", + retry.status_code, + sanitize_error_response(error_body), + retry_after=retry_after_of(retry), + ) + return retry.json() + + @property + def payment_mode(self) -> str: + """Which rail this client pays on: ``"apikey"`` or ``"wallet"``. + + Worth checking once at startup when both a key and a wallet are + configured in the environment: it is the difference between + spending credit and spending USDC.""" + return payment_mode(self) + + def get_wallet_address(self) -> str: + # No address on the account rail: payment comes from prepaid + # credit, so there is nothing to return but the empty string. + if self.api_key: + return "" + return self.account.address + + def close(self) -> None: + self._client.close() + + def __enter__(self) -> Self: + return self + + def __exit__(self, exc_type, exc_val, exc_tb) -> None: + self.close() diff --git a/blockrun_llm/solana_client.py b/blockrun_llm/solana_client.py new file mode 100644 index 0000000..0c0d39f --- /dev/null +++ b/blockrun_llm/solana_client.py @@ -0,0 +1,5601 @@ +""" +BlockRun Solana LLM Client. + +Usage: + from blockrun_llm import SolanaLLMClient + + # SOLANA_WALLET_KEY env var (bs58-encoded Solana secret key) + client = SolanaLLMClient() + + # Or pass key directly + client = SolanaLLMClient(private_key="your-bs58-key") + + # Same API as LLMClient + response = client.chat("openai/gpt-5.2", "gm Solana") + print(response) +""" + +from __future__ import annotations + +import asyncio +import json as _json +import os +import re +import sys +import threading +from collections.abc import Iterator +from typing import Any + +import httpx +from typing_extensions import Self + +from .apikey import ( + api_key_base_url, + auth_headers, + missing_credential_error, + payment_mode, + raise_for_api_key_402, + resolve_api_key, + resolve_poll_url, + wallet_only, +) + +# Shared with the Base client: signing is settlement on either chain, so the +# "already paid, do not retry on another model" tag has to mean the same thing +# in both fallback chains. client.py does not import this module, so there is +# no cycle. +from .client import _SETTLED_ATTR, _enforce_spend_limits, _mark_settled +from .price import Category, Market, Resolution, Session +from .realface import _GROUP_ID_RE +from .router_adapter import ( + SOLANA_MINIMUM_PAYMENT_USD, + build_model_pricing, + route_with_catalog, + routing_profile_for_model, + routing_text, +) +from .solana_wallet import get_solana_public_key +from .tx_log import ( + TransactionLogger, + _resolve_log_dir, + decode_settlement_header, + paid_request_error_prefix, + read_settlement_header, +) +from .types import ( + APIError, + ChatCompletionChunk, + ChatResponse, + ImageResponse, + MusicResponse, + PaymentError, + PortraitEnrollment, + PortraitList, + PriceHistoryResponse, + PricePoint, + RealFaceEnrollment, + RealFaceInit, + RealFaceList, + RealFaceStatus, + RetiredEndpointError, + RoutingDecision, + RoutingProfile, + RpcResponse, + SearchResult, + SmartChatCompletionResponse, + SmartChatResponse, + SpeechResponse, + SymbolListResponse, + VideoResponse, + chunk_meta, + chunk_usage_dict, + retry_after_of, + stream_choice_content, + stream_choice_finish_reason, +) +from .validation import ( + build_payment_rejected_error, + resolve_spend_limit, + sanitize_error_response, + validate_api_url, + validate_image_quality, + validate_max_tokens, + validate_video_input_type, +) + +try: + from x402 import x402ClientSync + from x402.http.utils import decode_payment_required_header, encode_payment_signature_header + from x402.mechanisms.svm import KeypairSigner + from x402.mechanisms.svm.exact.register import register_exact_svm_client + + _HAS_X402 = True +except ImportError: + _HAS_X402 = False + +SOLANA_API_URL = "https://sol.blockrun.ai/api" + + +def _create_signer(private_key: str) -> KeypairSigner: + """Create a KeypairSigner, handling both full keypair and seed-only formats.""" + try: + return KeypairSigner.from_base58(private_key) + except (ValueError, Exception): + # Fallback: might be a 32-byte seed (agentcash, etc.) + import base58 as b58 + from solders.keypair import Keypair + + decoded = b58.b58decode(private_key) + if len(decoded) == 32: + kp = Keypair.from_seed(decoded) + full_key = b58.b58encode(bytes(kp)).decode() + return KeypairSigner.from_base58(full_key) + raise + + +DEFAULT_MAX_TOKENS = 1024 + +# Per-use-case HTTP timeouts (seconds). The Base SDK splits these across +# multiple clients (LLMClient=120s, ImageClient=200s, MusicClient=210s, +# VideoClient=360s, ...); the Solana mega-class needs the same separation +# so a long chat or slow image generation does not silently die at 60s +# (the historical single-value default). +# +# Public callers can also override per-call via ``timeout=`` on +# ``chat_completion`` / ``image`` / ``image_edit`` / ``search``. +DEFAULT_CHAT_TIMEOUT = float( + os.environ.get("BLOCKRUN_CHAT_TIMEOUT", "600") +) # was 120; reasoning models need 200–300s+ +DEFAULT_IMAGE_TIMEOUT = 200.0 +DEFAULT_SEARCH_TIMEOUT = 300.0 +DEFAULT_FAST_TIMEOUT = 30.0 # pyth / x_user_info / quick lookups + +# Kept as a single fallback so old callers that pass a flat ``timeout=`` +# to ``SolanaLLMClient(...)`` continue to work — but the value now matches +# the chat client's budget so a 120s chat doesn't die under the old 60s. +DEFAULT_TIMEOUT = DEFAULT_CHAT_TIMEOUT + + +# --------------------------------------------------------------------------- +# Permanent payment errors — don't retry, don't fall back +# --------------------------------------------------------------------------- +# +# Mirrors the gateway-side classification at +# blockrun-sol/src/lib/x402-solana.ts PERMANENT_ERRORS. These reasons are +# deterministic on the SIGNED AUTHORIZATION level — re-signing without +# fixing the root cause produces the same failure within seconds. Surfacing +# the first failure immediately drops worst-case wall-clock from ~5min +# (3 generation attempts) to one attempt's worth. +_PERMANENT_PAYMENT_PATTERNS = ( + "insufficient", # insufficient_funds, "insufficient balance" + "invalid signature", # bad signing key / malformed payload + "invalid_payload", # gateway rejected payload shape + "expired", # payment_expired + "authorization is used", # replay-nonce hit + "transaction_simulation_failed", # CDP svm sim rejected (often blockhash window) + "blockhash not found", # blockhash already aged out — same class + "block height exceeded", # past slot lifetime — same class +) + + +def _is_permanent_payment_error(reason: str) -> bool: + """True iff the payment reason matches a permanent-failure pattern. + + Used by both the streaming fallback decision and the raw retry + classifier so the same policy applies to every Solana code path. + Case-insensitive substring match — patterns above are the BlockRun + gateway's own enums, and CDP returns the long form + (``invalid_exact_svm_payload_transaction_simulation_failed``) which + still contains the short form as a substring. + """ + if not reason: + return False + low = reason.lower() + return any(p in low for p in _PERMANENT_PAYMENT_PATTERNS) + + +# Payment failures a FRESH payment genuinely can't fix — re-running the whole +# request (new nonce, new 402 probe, new blockhash) won't help, so fail fast. +# Deliberately NARROWER than _PERMANENT_PAYMENT_PATTERNS: replay +# ("authorization is used"), amount mismatch, expired, blockhash/simulation +# errors ARE recoverable with a fresh signature and so are OMITTED here — they +# are exactly the concurrent-load failures the whole-request retry exists to fix. +_UNRECOVERABLE_PAYMENT_PATTERNS = ( + "insufficient", # wallet has no USDC + "invalid signature", # bad signing key + "invalid_payload", # structurally malformed payload + "denied", # payer denylisted +) + +# The gateway's `invalidMessage` (x402 VerifyResponse) names the simulation-level +# cause that `invalidReason` collapses into transaction_simulation_failed — which +# is deliberately absent above because it usually IS recoverable. These messages +# are the exception: the payer's USDC token account does not exist, so no fresh +# nonce/probe/blockhash will ever make the payment pass. Without them a wallet +# that can never pay burned all _MAX_PAYMENT_RETRIES + 1 attempts, every one of +# which cost the gateway its own verify retries. +# +# NOTE the asymmetry with the gateway's list (blockrun-sol x402-solana.ts): it +# ALSO fails fast on BlockhashNotFound, because retrying the SAME dead header is +# futile there. Here the opposite holds — re-signing with a FRESH blockhash is +# precisely what a pre-broadcast retry does, and it fixes it — so blockhash +# messages must stay OUT of this list. +_UNRECOVERABLE_INVALID_MESSAGES = ( + "invalidaccountdata", + "accountnotfound", + "couldnotfindaccount", +) + + +def _normalize_reason(reason: str) -> str: + """Lowercase and strip non-alphanumerics so one pattern matches every + spelling of a cause ("InvalidAccountData", "invalid account data", + "invalid_account_data"). Mirrors NORMALIZE in blockrun-sol x402-solana.ts.""" + return re.sub(r"[^a-z0-9]", "", reason.lower()) + + +def _is_unrecoverable_payment_error(reason: str) -> bool: + """True iff retrying with a brand-new payment cannot possibly succeed. + + Used by the whole-request payment retry to decide fail-fast vs retry. Unlike + :func:`_is_permanent_payment_error` (which classifies re-signing the SAME + authorization), a fresh nonce/probe/blockhash recovers replay, amount- + mismatch, expiry and blockhash-window failures, so only truly terminal + conditions (no funds, bad key, denylisted, no token account) short-circuit + the retry. + """ + if not reason: + return False + low = reason.lower() + if any(p in low for p in _UNRECOVERABLE_PAYMENT_PATTERNS): + return True + return any(p in _normalize_reason(reason) for p in _UNRECOVERABLE_INVALID_MESSAGES) + + +# --- Paid-leg re-sign policy ------------------------------------------------- +# +# The safe/unsafe line is the PAYMENT PHASE, not the specific cause. +# +# pre-broadcast — the gateway rejected the authorization before any transfer +# was submitted on-chain. Nothing settled, so re-running the +# whole request with a fresh nonce/amount/blockhash costs the +# payer nothing and is the documented cure. +# settlement — settle was attempted. If its acknowledgement was lost, +# re-signing pays twice for one request. Never re-signed. +# +# The gateway (blockrun-sol) emits two 402 body families and the policy has to +# read both: +# +# /v1/chat/completions {error, message, code, reason} +# the other paid routes {error, reason} — no `code`, no `message` +# +# `error` is the phase-bearing field in BOTH; validation.sanitize_error_response +# promotes it into `message` for the flat shape, which is why the title match +# below is against `message`. Titles are matched by PREFIX, never by substring: +# `_normalize_reason` strips separators, so a substring test can straddle word +# boundaries ("...verification failed before settlement; failed..." contains +# "settlementfailed"). blockrun-sol/src/lib/payment-rejection.ts documents that +# same over-match hazard and avoids it for the same reason. + +# Gateway `code` values that prove the rejection landed before any broadcast. +# PAYMENT_UNDERPAID — pre-verify amount-binding rejection +# PAYMENT_REPLAY — nonce claim rejected after verify, before inference +# PAYMENT_INVALID — facilitator verify rejection +_PRE_BROADCAST_CODES = frozenset({"paymentunderpaid", "paymentreplay", "paymentinvalid"}) + +# Normalized `error` titles for the same three, for the routes that send no code. +_PRE_BROADCAST_TITLES = ( + "paymentverificationfailed", + "paymentauthorizationalreadyused", + "paymentbelowquotedprice", +) + +_SETTLEMENT_CODE = "settlementfailed" +_SETTLEMENT_TITLE = "paymentsettlementfailed" + +# Verify-phase `reason` values a fresh signature can never satisfy. +_TERMINAL_VERIFY_REASONS = frozenset({"insufficientfunds"}) + + +def _is_safe_resign_error(exc: PaymentError) -> bool: + """Return whether a paid-leg 402 is safe to retry with a fresh signature. + + Retry iff the gateway proves the rejection was PRE-BROADCAST. Settlement + failures are terminal: settle has attempted an irreversible transfer, so a + lost acknowledgement must never authorize a second payment. + + Pre-broadcast rejections are exactly the concurrent single-wallet failures + the whole-request retry exists to fix, and each one's own gateway message + asks for the retry: + + * ``PAYMENT_UNDERPAID`` — "Re-fetch the 402 quote and sign the amount it + specifies." Emitted before verify runs. + * ``PAYMENT_REPLAY`` — "Sign a new payment for each request." The nonce + claim is taken after verify and before the result is served. + * ``PAYMENT_INVALID`` / ``Payment verification failed`` — every verify-phase + rejection, including ``expired_signature`` (stale blockhash), + ``verification_unavailable`` (the gateway's own docs: "Retry the request; + the signed payment was not rejected") and the ``verification_failed`` + catch-all that carries facilitator timeouts. Verify never broadcasts. + + ``insufficient_funds`` and the unrecoverable ``invalidMessage`` causes + (no USDC token account, bad signing key, denylisted payer) stay terminal — + no fresh signature makes them pass, and each wasted attempt costs the + gateway its own verify retries. + """ + body = exc.response if isinstance(exc.response, dict) else {} + code = _normalize_reason(str(body.get("code") or "")) + reason = _normalize_reason(str(body.get("reason") or "")) + message = _normalize_reason(str(body.get("message") or str(exc))) + + # Settlement phase is never re-signed. Checked first, and on all three + # fields, so no single missing field can turn a broadcast into a re-sign. + if code == _SETTLEMENT_CODE or reason == _SETTLEMENT_CODE: + return False + if message.startswith(_SETTLEMENT_TITLE): + return False + + # Fail fast on causes a fresh payment cannot cure (#23: payer has no USDC + # token account). These arrive as `reason` on newer routes and as folded + # `invalidMessage` text on older ones, so check both. + if reason in _TERMINAL_VERIFY_REASONS or _is_unrecoverable_payment_error(str(exc)): + return False + + # Positive pre-broadcast proof required — silence is terminal. + if code in _PRE_BROADCAST_CODES: + return True + return message.startswith(_PRE_BROADCAST_TITLES) + + +def _get_user_agent() -> str: + from . import __version__ + + return f"blockrun-python/{__version__}" + + +DEFAULT_SOLANA_RPC_URL = "https://sol.blockrun.ai/api/v1/solana/rpc" + + +def _resolve_rpc_config( + rpc_url: str | None, + rpc_headers: dict[str, str] | None, +) -> tuple[str, dict[str, str] | None]: + """Resolve the effective RPC URL + headers from explicit args, env vars, + or defaults — in that priority order. + + Since 0.24.0 the default is ``https://sol.blockrun.ai/api/v1/solana/rpc`` + — BlockRun's own multi-region Tatum-backed proxy. Free for anyone + using the BlockRun SDK; the cost is bundled into LLM inference fees + you already pay. Method-aware caching on the server side + (``getLatestBlockhash`` at 30s TTL) collapses bursty signing traffic + to a handful of upstream RPC calls, so partners no longer need to + register Helius / Tatum / QuickNode for typical loads. The public + ``api.mainnet-beta.solana.com`` is still reachable via explicit + config but is no longer the default — too aggressive a rate limit + for any real concurrency. + + Env vars (since 0.23.0): + * ``SOLANA_RPC_URL`` — full RPC URL. Override to point at your + own Helius / Tatum / QuickNode account, or to bypass the + BlockRun proxy entirely. + * ``SOLANA_RPC_API_KEY`` — convenience shortcut for the common + ``x-api-key: `` header style (Tatum, some QuickNode + setups). Not needed when using the BlockRun default (the + proxy handles its own upstream auth server-side). + * ``SOLANA_RPC_HEADERS`` — JSON dict for arbitrary headers + (``'{"x-api-key":"...", "x-rate-limit-tier":"pro"}'``). + + Helius style (key embedded in URL) needs ``SOLANA_RPC_URL`` only. + Tatum style (header auth) needs ``SOLANA_RPC_URL`` + one of + ``SOLANA_RPC_API_KEY`` / ``SOLANA_RPC_HEADERS``. + """ + import json as _json + + resolved_url = rpc_url or os.environ.get("SOLANA_RPC_URL") or DEFAULT_SOLANA_RPC_URL + + resolved_headers: dict[str, str] | None = None + if rpc_headers is not None: + resolved_headers = dict(rpc_headers) + else: + env_headers_json = os.environ.get("SOLANA_RPC_HEADERS") + env_api_key = os.environ.get("SOLANA_RPC_API_KEY") + if env_headers_json: + try: + parsed = _json.loads(env_headers_json) + if isinstance(parsed, dict): + resolved_headers = {str(k): str(v) for k, v in parsed.items()} + except Exception: + pass + elif env_api_key: + resolved_headers = {"x-api-key": env_api_key} + + return resolved_url, resolved_headers + + +def _register_svm_with_headers( + x402_client: Any, + signer: Any, + rpc_url: str, + rpc_headers: dict[str, str] | None, +) -> None: + """Register the SVM exact scheme on an x402 client, with optional + extra HTTP headers for the underlying Solana RPC. + + The x402 SDK's :func:`register_exact_svm_client` doesn't pass headers + through to ``solana.rpc.api.Client``, so when the user picks a gateway + that authenticates by header (Tatum, some Triton setups) we need a + pre-populated client cache. This function wires that up. + + If ``rpc_headers`` is ``None`` we delegate to the upstream helper so + behavior is unchanged for users on Helius-URL-style auth. + """ + if not rpc_headers: + register_exact_svm_client(x402_client, signer, rpc_url=rpc_url) + return + + # Header-auth path — pre-build SolanaClients with extra_headers and + # populate the scheme's _clients cache so it never falls back to the + # header-less default. + from solana.rpc.api import Client as SolanaClient + from x402.mechanisms.svm.exact.client import ExactSvmScheme + from x402.mechanisms.svm.exact.register import V1_NETWORKS + from x402.mechanisms.svm.exact.v1.client import ExactSvmSchemeV1 + + pre_client = SolanaClient(rpc_url, extra_headers=rpc_headers) + + def _populated(scheme): + # Hit several common network keys so the lazy _get_client never + # constructs a header-less SolanaClient. + for net in ("solana", "solana:mainnet", "solana-mainnet", "solana:devnet"): + scheme._clients[net] = pre_client + return scheme + + v2 = _populated(ExactSvmScheme(signer, rpc_url)) + v1 = _populated(ExactSvmSchemeV1(signer, rpc_url)) + + x402_client.register("solana:*", v2) + for network in V1_NETWORKS: + x402_client.register_v1(network, v1) + + +def _should_fallback_solana(exc: Exception) -> bool: + """Whether an exception during Solana streaming is retriable enough to + warrant trying the next ``fallback_models`` entry. Matches the Base + :func:`blockrun_llm.client._should_fallback` semantics: + + - Timeouts and network errors → fall back + - APIError with 5xx-ish status → fall back + - 4xx and PaymentError → propagate + + Defensive guard for issue #6: even when the exception is a transient + type (Timeout/Network), if the underlying reason is a permanent + payment classification (``transaction_simulation_failed``, etc.) we + do NOT fall back — re-signing a fresh request hits the same wall in + seconds. The first failure surfaces immediately. + + Also refuses anything already tagged by + :func:`blockrun_llm.client._mark_settled`: SPL USDC has left the wallet, and + the next model would sign a second transfer for the same call. + """ + if getattr(exc, _SETTLED_ATTR, False): + return False + # PaymentError always carries the gateway reason now (v0.32.0+). + if isinstance(exc, PaymentError): + return False + # Even for "transient" types, sniff the message for a permanent reason. + if _is_permanent_payment_error(str(exc)): + return False + if isinstance(exc, httpx.TimeoutException): + return True + if isinstance(exc, httpx.NetworkError): + return True + # 429 is retriable here for the same reason the TypeScript adapter treats it + # as transient: it means THIS upstream is saturated, and the next model in + # the chain is a different upstream. Observed live on the free tier — a + # rate-limited free model returned 429 and the three remaining free models + # in the ranked chain were never tried. Permanent payment failures and + # settled calls are refused above, before this line. + return bool(isinstance(exc, APIError) and exc.status_code in (429, 502, 503, 504, 522, 524)) + + +# Characters safe to interpolate into a single URL path segment. network / +# symbol / market / wallet address all get f-string'd into a paid endpoint +# path; a '/', '..', '?' or '#' would silently re-target the payment-signing +# request. These values often come from LLM output in agent use, so validate +# before building the URL. +_SAFE_PATH_SEGMENT_RE = re.compile(r"^[A-Za-z0-9._-]+$") + + +def _safe_path_segment(value: str, field: str) -> str: + """Return ``value`` if it is a single safe URL path segment, else raise.""" + if not value or not _SAFE_PATH_SEGMENT_RE.match(value): + raise ValueError( + f"{field} must contain only letters, digits, '.', '_' or '-' " f"(got {value!r})" + ) + return value + + +def _receipt_from_headers(headers: Any) -> str | None: + """Pull the x402 settlement tx hash from a paid response's headers.""" + if headers is None: + return None + return headers.get("x-payment-receipt") or headers.get("X-Payment-Receipt") + + +def _assert_same_payment_terms(signed_payload: Any, orig_amount: Any, orig_pay_to: Any) -> None: + """Guard a mid-poll re-sign: the fresh 402 challenge must charge the same + amount to the same recipient as the payment originally authorized for this + job. A gateway (buggy or hostile) that reprices or redirects the re-challenge + would otherwise extract an unbounded, unrelated payment from the wallet. + Raises :class:`PaymentError` on any mismatch so no signature is submitted.""" + accepted = signed_payload.accepted + if str(accepted.amount) != str(orig_amount) or accepted.pay_to != orig_pay_to: + raise PaymentError( + "Mid-poll re-sign challenge changed the payment terms " + f"(amount {orig_amount!r} -> {accepted.amount!r}, " + f"pay_to {orig_pay_to!r} -> {accepted.pay_to!r}); refusing to " + "authorize a different payment for the same job." + ) + + +class SolanaLLMClient: + """ + BlockRun LLM Client for Solana — pays via Solana USDC x402. + + Connects to sol.blockrun.ai by default. + """ + + SOLANA_API_URL = SOLANA_API_URL + + # Image generation slow-path polling. Models like ``openai/gpt-image-2`` + # or ``openai/dall-e-3`` routinely exceed the gateway's 30s inline window + # and come back as 202 + ``poll_url`` instead of the finished image. The + # SDK replays the same PAYMENT-SIGNATURE on every poll; settlement only + # happens on the first completed poll, so a poll-loop timeout = zero + # spend. Budget is conservative — most upstreams finish in 1-3 min. + IMAGE_POLL_INTERVAL_SECONDS = 5.0 + IMAGE_POLL_BUDGET_SECONDS = 300.0 + + # Video generation slow-path polling. Video always comes back as + # 202 + ``poll_url`` and can run far past the 600s x402 authorization + # window, so the poll loop re-signs a fresh PAYMENT-SIGNATURE (same + # wallet) when a poll 402s mid-flight. Settlement still only happens on + # the first completed poll, so a poll-loop timeout = zero spend. + VIDEO_DEFAULT_MODEL = "xai/grok-imagine-video" + VIDEO_POLL_INTERVAL_SECONDS = 5.0 + VIDEO_POLL_BUDGET_SECONDS = 900.0 + # Matches Base VideoClient.MAX_POLL_RESIGNS (2) — each re-sign is only used + # to refresh an expired blockhash, and every fresh signature is validated + # against the original payment terms before use. + MEDIA_POLL_MAX_RESIGNS = 2 + # Proactively re-sign the settlement authorization every N seconds during the + # poll loop so its recent-blockhash never ages out. The gateway settles only + # when upstream flips to "completed", and slow/flaky-status models (1080p + # Seedance) can bounce completed<->in_progress for minutes — long enough that + # a signature made earlier goes stale (blockhash lifetime ~60-90s) before the + # settling poll lands. 25s keeps every signature comfortably fresh. + MEDIA_RESIGN_FRESH_SECONDS = 25.0 + + # Media generation defaults (mirror the Base MusicClient/SpeechClient). + MUSIC_DEFAULT_MODEL = "minimax/music-2.5+" + SPEECH_DEFAULT_MODEL = "elevenlabs/flash-v2.5" + SOUNDFX_DEFAULT_MODEL = "elevenlabs/sound-effects" + + def __init__( + self, + private_key: str | None = None, + api_url: str = SOLANA_API_URL, + rpc_url: str | None = None, + timeout: float = DEFAULT_TIMEOUT, + image_timeout: float = DEFAULT_IMAGE_TIMEOUT, + search_timeout: float = DEFAULT_SEARCH_TIMEOUT, + rpc_headers: dict[str, str] | None = None, + transaction_log: bool | str | os.PathLike[str] | None = None, + max_cost_per_call: float | None = None, + max_session_cost: float | None = None, + ) -> None: + """Initialise the Solana client. + + ``timeout`` is the baseline (chat) HTTP timeout; ``image_timeout`` + and ``search_timeout`` tune the slower workloads independently — + mirroring the per-client tuning the Base SDK gets from having + separate ``LLMClient`` / ``ImageClient`` classes. Every public + method also takes a per-call ``timeout=`` override that wins over + all three. The historical single ``timeout=`` keyword still works + and now governs chat. + + ``rpc_url`` / ``rpc_headers`` fall back to the env vars + ``SOLANA_RPC_URL`` / ``SOLANA_RPC_HEADERS`` / ``SOLANA_RPC_API_KEY`` + when not passed explicitly (see :func:`_resolve_rpc_config`). + Default is the public mainnet-beta RPC if nothing is configured — + fine for low QPS, will 429 under burst load (~10-40 RPS). + + For production traffic point this at Helius / Tatum / QuickNode / + Triton. Tatum uses header-auth (``x-api-key``), which the upstream + x402 SDK doesn't pass through — we handle it here via + :func:`_register_svm_with_headers`. + + ``transaction_log`` mirrors :class:`LLMClient` — opt-in per-call + log written to a project folder (default ``./log/``) containing + the request, response, USD cost, and the on-chain settlement + signature returned by the facilitator. + """ + # An API key answers the chain question rather than being answered by + # it: api.blockrun.ai settles from credit, so there is no Solana + # transfer to sign, no wallet to load, and no reason to require the + # x402 SDK at all. Checked before the import guard for exactly that + # reason. + api_key = resolve_api_key(private_key) + if not api_key and not _HAS_X402: + raise ImportError( + "Solana payment requires the x402 SDK. " + "Install with: pip install blockrun-llm[solana]" + ) + from .solana_wallet import load_solana_wallet + + key = ( + None + if api_key + else ( + private_key + or os.environ.get("SOLANA_WALLET_KEY") + or load_solana_wallet() # disk: newest ~/.*/solana-wallet.json, else ~/.blockrun/.solana-session + ) + ) + if not api_key and not key: + raise missing_credential_error( + extra="Set SOLANA_WALLET_KEY, or keep a Solana wallet on disk " + "(~/./solana-wallet.json or ~/.blockrun/.solana-session)" + ) + self.api_key = api_key + self._private_key = key + if api_key: + # A key is answered by api.blockrun.ai, never by sol.blockrun.ai — + # so the Solana default must not reach the account rail. An + # api_url the caller actually typed still wins, as it does on + # every other client. + override = None if api_url == SOLANA_API_URL else api_url + self._api_url = api_key_base_url(override) + validate_api_url(self._api_url) + else: + validate_api_url(api_url) + self._api_url = api_url.rstrip("/") + # Model pricing cache for smart routing + self._model_pricing_cache: dict[str, dict[str, float]] | None = None + + # Resolve effective RPC URL + headers (explicit args > env vars > default). + resolved_url, resolved_headers = _resolve_rpc_config(rpc_url, rpc_headers) + self._rpc_url = resolved_url + self._rpc_headers = resolved_headers + + self._timeout = timeout + self._image_timeout = image_timeout + self._search_timeout = search_timeout + # httpx.Client carries the chat baseline as its default; image / + # search / per-call overrides are applied per request below. + self._client = httpx.Client(timeout=timeout, headers=auth_headers(api_key)) + self._session_total_usd = 0.0 + # Opt-in spend limits. None (the default) means unlimited, which is the + # behavior every release before 1.9.0 had: every 402 quote was signed + # automatically with nothing compared against anything. + self._max_cost_per_call = resolve_spend_limit( + max_cost_per_call, "BLOCKRUN_MAX_COST_PER_CALL" + ) + self._max_session_cost = resolve_spend_limit(max_session_cost, "BLOCKRUN_MAX_SESSION_COST") + self._session_calls = 0 + self._last_call_cost: float = 0.0 + self._address: str | None = None + + log_dir = _resolve_log_dir(transaction_log) + self._tx_logger: TransactionLogger | None = ( + TransactionLogger(log_dir) if log_dir is not None else None + ) + self._last_settlement: dict[str, Any] | None = None + # Response headers from the most recent raw paid POST — consumed by + # rpc()/music()/speech() to surface the settlement receipt + gateway + # metadata the shared JSON-only helper would otherwise drop. Read it + # immediately after the helper returns (no intervening await). + self._last_raw_headers: httpx.Headers | None = None + + # Account calls need neither an x402 client nor a local signer. Keep + # all shared request/receipt state above this branch initialized. + if api_key: + self._x402_client = None + self._payment_lock = threading.Lock() + return + + # Initialize x402 SDK client for Solana payment signing. + self._x402_client = x402ClientSync() + try: + signer = _create_signer(self._private_key) + except Exception as e: + # Parity with the Base client, which validates the resolved key up + # front: turn a malformed key (incl. one auto-loaded from disk) into + # a clean error instead of a raw base58/solders exception. + raise ValueError( + "Invalid Solana private key (expected a base58-encoded keypair " "or 32-byte seed)." + ) from e + _register_svm_with_headers(self._x402_client, signer, resolved_url, resolved_headers) + # x402ClientSync is NOT thread-safe: concurrent payment signing on one + # shared client races on nonce/authorization state. This lock serializes + # just the (fast) signing step so a single client can be shared across + # threads — see _sign_payment. + self._payment_lock = threading.Lock() + + def _sign_payment(self, payment_required: Any) -> Any: + """Thread-safe wrapper around ``x402_client.create_payment_payload``. + + Without this, sharing one ``SolanaLLMClient`` across threads (e.g. a + ThreadPoolExecutor issuing many concurrent paid requests from one wallet) + produces gateway rejections under load — ``authorization already used`` + (duplicate replay nonce) and ``invalid_exact_svm_payload_amount_mismatch`` + — because the underlying x402 client's nonce/auth state is mutated + concurrently. Serializing only this brief signing critical section fixes + it while the upstream streaming continues to run fully concurrently. + """ + with self._payment_lock: + return self._x402_client.create_payment_payload(payment_required) + + def _capture_settlement(self, response: httpx.Response) -> dict[str, Any] | None: + """Decode the x402 settlement header on a Solana paid response. + + Solana facilitators put the on-chain transaction signature in the + same ``PAYMENT-RESPONSE`` header EVM does — different chain id, same + wire format. ``None`` when no header is returned. + + Absence does NOT mean the call was free: this gateway's paid chat + path settles in parallel with the upstream call and re-raises at + once, so a charged-but-failed request answers before settlement + lands. See ``paid_request_error_prefix``. + """ + header = read_settlement_header(response.headers) + settlement = decode_settlement_header(header) + self._last_settlement = settlement + return settlement + + def _attach_receipt(self, data: Any) -> None: + """Inject the settlement tx hash from the most recent paid POST into a + raw response dict under ``txHash`` (mirrors the Base Music/Speech + clients). No-op on free responses (no receipt header).""" + tx_hash = _receipt_from_headers(self._last_raw_headers) + if tx_hash and isinstance(data, dict) and not data.get("txHash"): + data["txHash"] = tx_hash + + @property + def payment_mode(self) -> str: + """Which rail this client pays on: ``"apikey"`` or ``"wallet"``. + + Worth checking once at startup when both a key and a wallet are + configured in the environment: it is the difference between + spending credit and spending USDC.""" + return payment_mode(self) + + def get_wallet_address(self) -> str: + # No address on the account rail: payment comes from prepaid + # credit, so there is nothing to return but the empty string. + if self.api_key: + return "" + if not self._address: + self._address = get_solana_public_key(self._private_key) + return self._address + + def is_solana(self) -> bool: + return "sol.blockrun.ai" in self._api_url + + def get_balance(self) -> float: + # Returning 0 would be the worst available answer: it is + # indistinguishable from an empty wallet, and an agent gating on it + # would stop calling a well-funded account. + if self.api_key: + raise wallet_only("get_balance") + """Get USDC balance on Solana (matches LLMClient.get_balance() API).""" + from .solana_wallet import get_solana_usdc_balance + + return get_solana_usdc_balance(self.get_wallet_address(), rpc_url=self._rpc_url) + + def get_spending(self) -> dict[str, Any]: + return {"total_usd": self._session_total_usd, "calls": self._session_calls} + + def _billing_meta(self) -> dict[str, str | None]: + """Billing metadata for cost-log entries.""" + return { + "wallet": self.get_wallet_address(), + "network": "solana-mainnet" if self.is_solana() else "solana-other", + "client_kind": type(self).__name__, + } + + def _log_transaction( + self, + endpoint: str, + body: dict[str, Any], + response: Any, + cost_usd: float, + ) -> None: + """Append one row to the project-local transaction log when the + sync Solana client is constructed with ``transaction_log=…``. + + Consumes ``self._last_settlement`` so the on-chain Solana signature + captured from the paid retry is written exactly once per call.""" + logger = self._tx_logger + if logger is None: + return + settlement = self._last_settlement + self._last_settlement = None + try: + logger.log( + endpoint=endpoint, + request=body, + response=response, + cost_usd=cost_usd, + model=(body.get("model") if isinstance(body, dict) else None), + wallet=self.get_wallet_address(), + network="solana-mainnet" if self.is_solana() else "solana-other", + client_kind=type(self).__name__, + settlement=settlement, + ) + except Exception: + pass + + def _get_model_pricing(self) -> dict[str, dict[str, float]]: + """Model pricing for smart routing (cached for the client's lifetime).""" + if self._model_pricing_cache is not None: + return self._model_pricing_cache + pricing = build_model_pricing(self.list_models()) + self._model_pricing_cache = pricing + return pricing + + def route( + self, + prompt: str, + *, + system: str | None = None, + max_tokens: int | None = None, + routing_profile: RoutingProfile = "auto", + requires_structured_output: bool = False, + ) -> RoutingDecision: + """Inspect a Solana routing decision without making or paying for a call. + + Identical routing to the Base client — same Router Core engine, same + catalog — with the Solana x402 minimum applied to the cost estimate. + """ + decision = route_with_catalog( + prompt, + system, + max_tokens or DEFAULT_MAX_TOKENS, + self._get_model_pricing(), + routing_profile=routing_profile, + requires_structured_output=requires_structured_output, + minimum_payment_usd=SOLANA_MINIMUM_PAYMENT_USD, + ) + return RoutingDecision(**decision) + + def smart_chat( + self, + prompt: str, + *, + system: str | None = None, + max_tokens: int | None = None, + temperature: float | None = None, + routing_profile: RoutingProfile = "auto", + timeout: float | None = None, + ) -> SmartChatResponse: + """Smart chat with automatic model routing, paid on Solana. + + Uses BlockRun's Router Core portfolio strategy — the same engine the + Base client, the TypeScript SDK and the gateway run. Routing is local + (<1ms, no extra model call); only the payment leg differs by chain. + + Example: + result = client.smart_chat("What is 2+2?") + print(result.model) # 'google/gemini-2.5-flash' + print(result.routing.method) # 'portfolio' + """ + decision = route_with_catalog( + prompt, + system, + max_tokens or DEFAULT_MAX_TOKENS, + self._get_model_pricing(), + routing_profile=routing_profile, + minimum_payment_usd=SOLANA_MINIMUM_PAYMENT_USD, + ) + response = self.chat( + decision["model"], + prompt, + system=system, + max_tokens=max_tokens or DEFAULT_MAX_TOKENS, + temperature=temperature, + timeout=timeout, + fallback_models=decision.get("fallbacks") or None, + ) + return SmartChatResponse( + response=response, + model=decision["model"], + routing=RoutingDecision(**decision), + ) + + def smart_chat_completion( + self, + messages: list[dict[str, Any]], + *, + max_tokens: int | None = None, + temperature: float | None = None, + top_p: float | None = None, + search: bool = False, + search_parameters: dict[str, Any] | None = None, + tools: list[dict[str, Any]] | None = None, + tool_choice: Any | None = None, + timeout: float | None = None, + response_format: dict[str, Any] | None = None, + stop: str | list[str] | None = None, + fallback_models: list[str] | None = None, + routing_profile: RoutingProfile = "auto", + ) -> SmartChatCompletionResponse: + """Smart routing for a full message list, paid on Solana. + + Tools, tool_choice and response_format are part of the routing + decision, and capacity is checked against the whole transcript — see + :meth:`blockrun_llm.LLMClient.smart_chat_completion`. + """ + view = routing_text(messages) + decision = route_with_catalog( + view["prompt"], + view["system_prompt"], + max_tokens or DEFAULT_MAX_TOKENS, + self._get_model_pricing(), + routing_profile=routing_profile, + requires_structured_output=response_format is not None, + tools=tools, + tool_choice=tool_choice, + conversation_chars=view["conversation_chars"], + has_vision=view["has_vision"], + minimum_payment_usd=SOLANA_MINIMUM_PAYMENT_USD, + ) + response = self.chat_completion( + decision["model"], + messages, + max_tokens=max_tokens or DEFAULT_MAX_TOKENS, + temperature=temperature, + top_p=top_p, + search=search, + search_parameters=search_parameters, + tools=tools, + tool_choice=tool_choice, + timeout=timeout, + response_format=response_format, + stop=stop, + # An explicit caller-supplied chain wins over the routed one. + fallback_models=fallback_models or decision.get("fallbacks") or None, + ) + return SmartChatCompletionResponse( + response=response, + model=decision["model"], + routing=RoutingDecision(**decision), + ) + + def chat( + self, + model: str, + prompt: str, + system: str | None = None, + max_tokens: int = DEFAULT_MAX_TOKENS, + temperature: float | None = None, + search: bool = False, + timeout: float | None = None, + response_format: dict[str, Any] | None = None, + stop: str | list[str] | None = None, + fallback_models: list[str] | None = None, + ) -> str: + """Simple 1-line chat.""" + messages: list[dict[str, str]] = [] + if system: + messages.append({"role": "system", "content": system}) + messages.append({"role": "user", "content": prompt}) + result = self.chat_completion( + model, + messages, + max_tokens=max_tokens, + temperature=temperature, + search=search, + timeout=timeout, + response_format=response_format, + stop=stop, + fallback_models=fallback_models, + ) + return result.choices[0].message.content or "" + + def chat_completion( + self, + model: str, + messages: list[dict[str, Any]], + max_tokens: int = DEFAULT_MAX_TOKENS, + temperature: float | None = None, + top_p: float | None = None, + search: bool = False, + search_parameters: dict[str, Any] | None = None, + tools: list[dict[str, Any]] | None = None, + tool_choice: Any | None = None, + timeout: float | None = None, + response_format: dict[str, Any] | None = None, + stop: str | list[str] | None = None, + fallback_models: list[str] | None = None, + ) -> ChatResponse: + """Full chat completion (OpenAI-compatible). + + Supports OpenAI-style function calling via ``tools`` / + ``tool_choice`` — the BlockRun gateway forwards them to the + upstream model unchanged (Base and Solana use the same backend + schema; the only chain difference is the payment leg). + + ``timeout`` overrides the per-call HTTP timeout (defaults to the + client's chat baseline, ``DEFAULT_CHAT_TIMEOUT``). Raise it for + large ``max_tokens`` runs against slow models. + """ + # `blockrun/auto` | `blockrun/eco` | `blockrun/premium` are routing + # profiles rather than models — hand the turn to the routed path. + virtual_profile = routing_profile_for_model(model) + if virtual_profile is not None: + return self.smart_chat_completion( + messages, + max_tokens=max_tokens, + temperature=temperature, + top_p=top_p, + search=search, + search_parameters=search_parameters, + tools=tools, + tool_choice=tool_choice, + timeout=timeout, + response_format=response_format, + stop=stop, + fallback_models=fallback_models, + routing_profile=virtual_profile, # type: ignore[arg-type] + ).response + + validate_max_tokens(max_tokens) + body: dict[str, Any] = {"model": model, "messages": messages, "max_tokens": max_tokens} + if temperature is not None: + body["temperature"] = temperature + if top_p is not None: + body["top_p"] = top_p + if search_parameters: + body["search_parameters"] = search_parameters + elif search: + body["search_parameters"] = {"mode": "on"} + if tools is not None: + body["tools"] = tools + if tool_choice is not None: + body["tool_choice"] = tool_choice + if response_format is not None: + body["response_format"] = response_format + if stop is not None: + body["stop"] = stop + + # Walk [model, *fallback_models] on retriable errors (timeouts, 5xx, + # network) exactly as the streaming path and the Base client do. A + # settled payment is never retried — _should_fallback_solana refuses + # anything tagged as settled, so the next model cannot sign a second + # transfer for the same call. + attempts = [model, *(fallback_models or [])] + last_exc: Exception | None = None + for i, attempt_model in enumerate(attempts): + body["model"] = attempt_model + try: + return self._request_with_payment("/v1/chat/completions", body, timeout=timeout) + except Exception as exc: + if not _should_fallback_solana(exc) or i + 1 >= len(attempts): + raise + last_exc = exc + sys.stderr.write( + f"[blockrun_llm] solana {attempt_model} -> {attempts[i + 1]} " + f"({type(exc).__name__}: {str(exc)[:80]})\n" + ) + assert last_exc is not None + raise last_exc + + def close(self) -> None: + """Close the HTTP client.""" + self._client.close() + + def list_models(self) -> list[dict[str, Any]]: + resp = self._client.get(f"{self._api_url}/v1/models") + resp.raise_for_status() + return resp.json().get("data", []) + + @staticmethod + def _extract_payment_header(response: httpx.Response) -> str | None: + """Extract x402 payment header from a 402 response (header or body).""" + payment_header = response.headers.get("payment-required") + if not payment_header: + try: + import base64 + import json + + resp_body = response.json() + if resp_body.get("accepts") or resp_body.get("x402Version"): + payment_header = base64.b64encode(json.dumps(resp_body).encode()).decode() + except Exception: + pass + return payment_header + + # ------------------------------------------------------------------ + # Streaming (SSE) chat completions + # ------------------------------------------------------------------ + + # Retry policy mirrors LLMClient. ``1 + len(_STREAM_5XX_BACKOFFS)`` tries + # per phase (probe / paid-retry), exponential backoff in seconds. + _STREAM_5XX_STATUSES = (500, 502, 503, 504) + _STREAM_5XX_BACKOFFS = (1.0, 2.0, 4.0) + + # Whole-request payment retry: on a PRE-BROADCAST payment rejection + # (concurrent single-wallet replay-nonce / underpaid amount binding / + # verify-phase flake), re-run the ENTIRE paid request — fresh 402 probe + + # fresh signature (new nonce, correct amount, current blockhash) — but only + # before the first chunk is yielded. This is what gets concurrent load to + # ~100% success; the per-call signing lock alone can't recover a transient + # or amount failure once it has happened. + # + # A settlement failure is NEVER retried (see _is_safe_resign_error), so no + # attempt here can pay twice: every retried rejection is one the gateway + # refused before broadcasting. + _MAX_PAYMENT_RETRIES = 4 + _PAYMENT_RETRY_BACKOFFS = (0.25, 0.5, 1.0, 2.0) + + def chat_completion_stream( + self, + model: str, + messages: list[dict[str, Any]], + *, + max_tokens: int = DEFAULT_MAX_TOKENS, + temperature: float | None = None, + top_p: float | None = None, + search: bool = False, + search_parameters: dict[str, Any] | None = None, + tools: list[dict[str, Any]] | None = None, + tool_choice: Any | None = None, + response_format: dict[str, Any] | None = None, + stop: str | list[str] | None = None, + fallback_models: list[str] | None = None, + timeout: float | None = None, + ) -> Iterator[ChatCompletionChunk]: + """ + Stream a chat completion via Server-Sent Events, paid in Solana USDC + via x402. Mirrors :meth:`LLMClient.chat_completion_stream` semantics: + + - Yields one :class:`ChatCompletionChunk` per ``data:`` line until + the upstream emits ``data: [DONE]``. + - Free models stream on the first request; paid models do the + 402 → sign locally with the SVM signer → retry with + ``PAYMENT-SIGNATURE`` dance before the first chunk. + - 5xx upstream errors are retried in-band with exponential + backoff (1s / 2s / 4s). + - ``fallback_models`` walks the chain on retriable errors, but + only **before** the first chunk has been yielded (mid-stream + fallback would concatenate two distinct responses). + - ``tools`` / ``tool_choice`` work the same as on Base — the + gateway forwards them to the upstream model regardless of + chain. + + Note: ``search_parameters`` is rejected by the BlockRun gateway in + stream mode (HTTP 400). Codex / GPT-5.4-Pro also can't stream. + """ + validate_max_tokens(max_tokens) + body: dict[str, Any] = { + "model": model, + "messages": messages, + "stream": True, + "max_tokens": max_tokens, + } + if temperature is not None: + body["temperature"] = temperature + if top_p is not None: + body["top_p"] = top_p + if search_parameters: + body["search_parameters"] = search_parameters + elif search: + body["search_parameters"] = {"mode": "on"} + if tools is not None: + body["tools"] = tools + if tool_choice is not None: + body["tool_choice"] = tool_choice + if response_format is not None: + body["response_format"] = response_format + if stop is not None: + body["stop"] = stop + + attempts = [model, *(fallback_models or [])] + last_exc: Exception | None = None + + for i, attempt_model in enumerate(attempts): + body["model"] = attempt_model + inner = self._stream_with_payment("/v1/chat/completions", body, timeout=timeout) + chunks_yielded = 0 + try: + for chunk in inner: + chunks_yielded += 1 + yield chunk + return # finished cleanly + except Exception as exc: + if chunks_yielded > 0: + raise # mid-stream — can't fall back + if not _should_fallback_solana(exc): + raise + last_exc = exc + if i + 1 < len(attempts): + next_model = attempts[i + 1] + sys.stderr.write( + f"[blockrun_llm] solana stream {attempt_model} -> " + f"{next_model} ({type(exc).__name__}: {str(exc)[:80]})\n" + ) + assert last_exc is not None + raise last_exc + + def _stream_with_payment( + self, + endpoint: str, + body: dict[str, Any], + timeout: float | None = None, + ) -> Iterator[ChatCompletionChunk]: + """Whole-request payment-retry wrapper around :meth:`_stream_once`. + + Re-runs the entire paid request (fresh 402 probe + fresh signature) on a + PRE-BROADCAST payment rejection (:func:`_is_safe_resign_error`), but only + before the first chunk is yielded — once the 200 stream starts, + :meth:`_stream_once` returns without raising, so output is never + replayed. A settlement failure is terminal. See _MAX_PAYMENT_RETRIES. + """ + import time + + for payment_attempt in range(self._MAX_PAYMENT_RETRIES + 1): + yielded = 0 + try: + for chunk in self._stream_once(endpoint, body, timeout=timeout): + yielded += 1 + yield chunk + return + except PaymentError as exc: + if ( + yielded > 0 + or not _is_safe_resign_error(exc) + or payment_attempt >= self._MAX_PAYMENT_RETRIES + ): + raise + time.sleep( + self._PAYMENT_RETRY_BACKOFFS[ + min(payment_attempt, len(self._PAYMENT_RETRY_BACKOFFS) - 1) + ] + ) + + def _stream_once( + self, + endpoint: str, + body: dict[str, Any], + timeout: float | None = None, + ) -> Iterator[ChatCompletionChunk]: + """402 → sign (SVM) → retry → SSE iter. Same shape as the Base + :meth:`LLMClient._stream_with_payment`; differs only in the + signing path (we go through the x402 SDK's SVM client).""" + url = f"{self._api_url}{endpoint}" + req_headers = {"Content-Type": "application/json", "User-Agent": _get_user_agent()} + eff_timeout = timeout if timeout is not None else self._timeout + + backoffs = self._STREAM_5XX_BACKOFFS + + # ----- Phase 1: probe (no payment header) ----- + payment_headers: dict[str, str] | None = None + cost_usd = 0.0 + + for attempt in range(len(backoffs) + 1): + with self._client.stream( + "POST", url, json=body, headers=req_headers, timeout=eff_timeout + ) as resp1: + if resp1.status_code == 200: + # Free model — stream directly. + yield from self._iter_sse_chunks(resp1) + return + resp1.read() + if resp1.status_code == 402: + # Account rail: a 402 is the account being out of credit, not a + # challenge to sign. Nothing here can sign, so say so plainly. + raise_for_api_key_402(resp1, self.api_key) + payment_headers, cost_usd = self._sign_payment_from_response(resp1) + break + if resp1.status_code in self._STREAM_5XX_STATUSES and attempt < len(backoffs): + import time + + time.sleep(backoffs[attempt]) + continue + self._raise_stream_error(resp1, after_payment=False) + else: + raise APIError("solana stream probe exhausted retries", 0, None) + + # ----- Phase 2: stream with PAYMENT-SIGNATURE ----- + assert payment_headers is not None + try: + for attempt in range(len(backoffs) + 1): + with self._client.stream( + "POST", url, json=body, headers=payment_headers, timeout=eff_timeout + ) as resp2: + if resp2.status_code == 200: + if cost_usd > 0: + self._session_calls += 1 + self._session_total_usd += cost_usd + self._last_call_cost = cost_usd + self._capture_settlement(resp2) + yield from self._iter_and_archive(resp2, body, cost_usd) + return + resp2.read() + if resp2.status_code == 402: + # Account rail: a 402 is the account being out of credit, not a + # challenge to sign. Nothing here can sign, so say so plainly. + raise_for_api_key_402(resp2, self.api_key) + raise build_payment_rejected_error(resp2) + if resp2.status_code in self._STREAM_5XX_STATUSES and attempt < len(backoffs): + import time + + time.sleep(backoffs[attempt]) + continue + self._raise_stream_error(resp2, after_payment=True) + + except (httpx.HTTPError, APIError) as exc: + # Signed above; SPL USDC is gone. Do not let the fallback + # chain buy a retry on the next model. Re-raise bare so the + # traceback and __context__ survive. + _mark_settled(exc) + raise + + def _iter_and_archive( + self, + response: httpx.Response, + body: dict[str, Any], + cost_usd: float, + ) -> Iterator[ChatCompletionChunk]: + """Yield SSE chunks; on stream completion, archive the assembled + response to ``~/.blockrun/data/`` and append a row to + ``~/.blockrun/cost_log.jsonl``. Paid streaming calls now show up + in the same audit trail as non-stream paid calls. + + ``cost_usd == 0`` skips the archive (free models / unauth probe).""" + assembled_id: str | None = None + assembled_model: str | None = None + assembled_created: int = 0 + content_parts: list[str] = [] + finish_reason: str | None = None + usage_dict: dict[str, Any] | None = None + + for chunk in self._iter_sse_chunks(response): + if chunk.choices: + choice = chunk.choices[0] + content = stream_choice_content(choice) + if content: + content_parts.append(content) + fr = stream_choice_finish_reason(choice) + if fr: + finish_reason = fr + if assembled_id is None: + _id, _model, _created = chunk_meta(chunk) + if _id: + assembled_id = _id + assembled_model = _model + assembled_created = _created + _usage = chunk_usage_dict(chunk) + if _usage is not None: + usage_dict = _usage + # Race-free per-call x402 charge — see LLMClient._iter_and_archive. + chunk.cost_usd = cost_usd + yield chunk + + if cost_usd > 0: + from .cache import save_to_cache + + response_data: dict[str, Any] = { + "id": assembled_id or "stream", + "object": "chat.completion", + "created": assembled_created or int(__import__("time").time()), + "model": assembled_model or body.get("model"), + "choices": [ + { + "index": 0, + "message": { + "role": "assistant", + "content": "".join(content_parts), + }, + "finish_reason": finish_reason, + } + ], + "stream": True, + } + if usage_dict: + response_data["usage"] = usage_dict + try: + save_to_cache( + "/v1/chat/completions", + body, + response_data, + cost_usd=cost_usd, + **self._billing_meta(), + ) + except Exception: + pass + self._log_transaction("/v1/chat/completions", body, response_data, cost_usd) + + @staticmethod + def _iter_sse_chunks(response: httpx.Response) -> Iterator[ChatCompletionChunk]: + """OpenAI-format SSE parser. ``data: \\n\\n`` lines, terminated + by ``data: [DONE]``. Malformed chunks are skipped, not raised.""" + for raw_line in response.iter_lines(): + if not raw_line or not raw_line.startswith("data: "): + continue + payload = raw_line[6:].strip() + if payload == "[DONE]": + return + try: + chunk_dict = _json.loads(payload) + except Exception: + continue + try: + yield ChatCompletionChunk(**chunk_dict) + except Exception: + yield ChatCompletionChunk.model_construct(**chunk_dict) + + def _sign_payment_from_response( + self, + response: httpx.Response, + ) -> tuple[dict[str, str], float]: + """Extract a 402 response's payment requirements, sign locally with + the SVM x402 client, return ``(headers_with_PAYMENT_SIGNATURE, + cost_usd)``. Mirrors the inline logic in + :meth:`_handle_payment_and_retry` but returns headers instead of + making the retry POST itself — lets the streaming path open an + SSE connection for the retry.""" + payment_header = self._extract_payment_header(response) + if not payment_header: + raise PaymentError("402 response but no payment requirements found") + + payment_required = decode_payment_required_header(payment_header) + payment_payload = self._sign_payment(payment_required) + # Before the paid request goes out. Signing alone moves nothing; the + # gateway submitting the signed authorization does, so refusing here + # means nothing settles. + _enforce_spend_limits(self, float(payment_payload.accepted.amount) / 1e6) + encoded_payment = encode_payment_signature_header(payment_payload) + + cost_usd = float(payment_payload.accepted.amount) / 1e6 + + return ( + { + "Content-Type": "application/json", + "User-Agent": _get_user_agent(), + "PAYMENT-SIGNATURE": encoded_payment, + }, + cost_usd, + ) + + @staticmethod + def _raise_stream_error(response: httpx.Response, *, after_payment: bool) -> None: + try: + error_body = response.json() + except Exception: + error_body = {"error": "Stream request failed"} + prefix = paid_request_error_prefix(response.headers) if after_payment else "API error" + raise APIError( + f"{prefix}: {response.status_code}", + response.status_code, + sanitize_error_response(error_body), + retry_after=retry_after_of(response), + ) + + def _request_with_payment( + self, endpoint: str, body: dict[str, Any], timeout: float | None = None + ) -> ChatResponse: + """Whole-request payment-retry wrapper around :meth:`_request_once`. + + Re-runs the entire paid request (fresh 402 probe + fresh signature) on a + PRE-BROADCAST payment rejection — concurrent replay-nonce, underpaid + amount binding, or a verify-phase flake — so a shared client under + concurrent load reaches ~100%. Settlement failures are terminal: settle + may already have broadcast, so re-signing could pay twice for one + request. See :func:`_is_safe_resign_error` and _MAX_PAYMENT_RETRIES. + """ + import time + + for payment_attempt in range(self._MAX_PAYMENT_RETRIES + 1): + try: + return self._request_once(endpoint, body, timeout=timeout) + except PaymentError as exc: + if not _is_safe_resign_error(exc) or payment_attempt >= self._MAX_PAYMENT_RETRIES: + raise + time.sleep( + self._PAYMENT_RETRY_BACKOFFS[ + min(payment_attempt, len(self._PAYMENT_RETRY_BACKOFFS) - 1) + ] + ) + + raise PaymentError( # pragma: no cover - bounded loop always returns or raises + "Payment retry loop exhausted without a result." + ) + + def _request_once( + self, endpoint: str, body: dict[str, Any], timeout: float | None = None + ) -> ChatResponse: + url = f"{self._api_url}{endpoint}" + headers = {"Content-Type": "application/json", "User-Agent": _get_user_agent()} + eff_timeout = timeout if timeout is not None else self._timeout + + response = self._client.post(url, json=body, headers=headers, timeout=eff_timeout) + + # Auto-retry on transient server errors + if response.status_code in (502, 503): + import time + + time.sleep(1) + response = self._client.post(url, json=body, headers=headers, timeout=eff_timeout) + + if response.status_code == 402: + # Account rail: a 402 is the account being out of credit, not a + # challenge to sign. Nothing here can sign, so say so plainly. + raise_for_api_key_402(response, self.api_key) + # Past this point the SPL USDC transfer has been signed. Tag + # anything that escapes so no fallback chain can buy a retry. + try: + return self._handle_payment_and_retry(url, body, response, timeout=eff_timeout) + except (httpx.HTTPError, APIError) as exc: + _mark_settled(exc) + raise + + if not response.is_success: + try: + error_body = response.json() + except Exception: + error_body = {"error": "Request failed"} + raise APIError( + f"API error: {response.status_code}", + response.status_code, + sanitize_error_response(error_body), + retry_after=retry_after_of(response), + ) + + return ChatResponse(**response.json()) + + def _handle_payment_and_retry( + self, + url: str, + body: dict[str, Any], + response: httpx.Response, + timeout: float | None = None, + ) -> ChatResponse: + eff_timeout = timeout if timeout is not None else self._timeout + payment_header = self._extract_payment_header(response) + if not payment_header: + raise PaymentError("402 response but no payment requirements found") + + # Use x402 SDK to decode 402 response and create signed payment + payment_required = decode_payment_required_header(payment_header) + payment_payload = self._sign_payment(payment_required) + # Before the paid request goes out. Signing alone moves nothing; the + # gateway submitting the signed authorization does, so refusing here + # means nothing settles. + _enforce_spend_limits(self, float(payment_payload.accepted.amount) / 1e6, body.get("model")) + encoded_payment = encode_payment_signature_header(payment_payload) + + payment_headers = { + "Content-Type": "application/json", + "User-Agent": _get_user_agent(), + "PAYMENT-SIGNATURE": encoded_payment, + } + + # Retry with payment, with one automatic retry on 502/503 + retry_response = self._client.post( + url, json=body, headers=payment_headers, timeout=eff_timeout + ) + if retry_response.status_code in (502, 503): + import time + + time.sleep(1) + retry_response = self._client.post( + url, json=body, headers=payment_headers, timeout=eff_timeout + ) + + if retry_response.status_code == 402: + # Account rail: a 402 is the account being out of credit, not a + # challenge to sign. Nothing here can sign, so say so plainly. + raise_for_api_key_402(retry_response, self.api_key) + raise build_payment_rejected_error(retry_response) + + if not retry_response.is_success: + try: + error_body = retry_response.json() + except Exception: + error_body = {"error": "Request failed"} + raise APIError( + f"{paid_request_error_prefix(retry_response.headers)}: {retry_response.status_code}", + retry_response.status_code, + sanitize_error_response(error_body), + retry_after=retry_after_of(retry_response), + ) + + cost_usd = float(payment_payload.accepted.amount) / 1e6 + self._session_calls += 1 + self._session_total_usd += cost_usd + self._last_call_cost = cost_usd + self._capture_settlement(retry_response) + + # Save full response locally + response_data = retry_response.json() + from .cache import save_to_cache + + save_to_cache( + "/v1/chat/completions", + body, + response_data, + cost_usd=cost_usd, + **self._billing_meta(), + ) + self._log_transaction("/v1/chat/completions", body, response_data, cost_usd) + + return ChatResponse(**response_data) + + def _request_with_payment_raw( + self, endpoint: str, body: dict[str, Any], timeout: float | None = None + ) -> dict[str, Any]: + """Bounded fresh-signature retry wrapper for raw POST endpoints.""" + import time + + for payment_attempt in range(self._MAX_PAYMENT_RETRIES + 1): + try: + return self._request_with_payment_raw_once(endpoint, body, timeout=timeout) + except PaymentError as exc: + if not _is_safe_resign_error(exc) or payment_attempt >= self._MAX_PAYMENT_RETRIES: + raise + time.sleep( + self._PAYMENT_RETRY_BACKOFFS[ + min(payment_attempt, len(self._PAYMENT_RETRY_BACKOFFS) - 1) + ] + ) + + raise PaymentError( # pragma: no cover - bounded loop always returns or raises + "Payment retry loop exhausted without a result." + ) + + def _request_with_payment_raw_once( + self, endpoint: str, body: dict[str, Any], timeout: float | None = None + ) -> dict[str, Any]: + """Make a request with Solana x402 payment, returning raw JSON.""" + from .cache import get_cached, save_to_cache + + # Check cache first — don't pay twice for same data + cached = get_cached(endpoint, body) + if cached is not None: + return cached + + # Reset per-call receipt headers; only a paid retry repopulates them, so + # a free/cached model can't inherit a prior call's settlement receipt. + self._last_raw_headers = None + + url = f"{self._api_url}{endpoint}" + headers = {"Content-Type": "application/json", "User-Agent": _get_user_agent()} + eff_timeout = timeout if timeout is not None else self._timeout + + response = self._client.post(url, json=body, headers=headers, timeout=eff_timeout) + + # Auto-retry on transient server errors + if response.status_code in (502, 503): + import time + + time.sleep(1) + response = self._client.post(url, json=body, headers=headers, timeout=eff_timeout) + + if response.status_code == 402: + # Account rail: a 402 is the account being out of credit, not a + # challenge to sign. Nothing here can sign, so say so plainly. + raise_for_api_key_402(response, self.api_key) + # Past this point the SPL USDC transfer has been signed. Tag + # anything that escapes so no fallback chain can buy a retry. + try: + result = self._handle_payment_and_retry_raw( + url, body, response, timeout=eff_timeout + ) + except (httpx.HTTPError, APIError) as exc: + _mark_settled(exc) + raise + save_to_cache( + endpoint, + body, + result, + cost_usd=self._last_call_cost, + **self._billing_meta(), + ) + self._log_transaction(endpoint, body, result, self._last_call_cost) + return result + + if not response.is_success: + try: + error_body = response.json() + except Exception: + error_body = {"error": "Request failed"} + raise APIError( + f"API error: {response.status_code}", + response.status_code, + sanitize_error_response(error_body), + retry_after=retry_after_of(response), + ) + + return response.json() + + def _handle_payment_and_retry_raw( + self, + url: str, + body: dict[str, Any], + response: httpx.Response, + timeout: float | None = None, + ) -> dict[str, Any]: + """Handle 402 for raw endpoints with Solana payment.""" + eff_timeout = timeout if timeout is not None else self._timeout + payment_header = self._extract_payment_header(response) + if not payment_header: + raise PaymentError("402 response but no payment requirements found") + + # Use x402 SDK to decode 402 response and create signed payment + payment_required = decode_payment_required_header(payment_header) + payment_payload = self._sign_payment(payment_required) + # Before the paid request goes out. Signing alone moves nothing; the + # gateway submitting the signed authorization does, so refusing here + # means nothing settles. + _enforce_spend_limits(self, float(payment_payload.accepted.amount) / 1e6, body.get("model")) + encoded_payment = encode_payment_signature_header(payment_payload) + + payment_headers = { + "Content-Type": "application/json", + "User-Agent": _get_user_agent(), + "PAYMENT-SIGNATURE": encoded_payment, + } + + # Retry with payment, with one automatic retry on 502/503 + retry_response = self._client.post( + url, json=body, headers=payment_headers, timeout=eff_timeout + ) + if retry_response.status_code in (502, 503): + import time + + time.sleep(1) + retry_response = self._client.post( + url, json=body, headers=payment_headers, timeout=eff_timeout + ) + + if retry_response.status_code == 402: + # Account rail: a 402 is the account being out of credit, not a + # challenge to sign. Nothing here can sign, so say so plainly. + raise_for_api_key_402(retry_response, self.api_key) + raise build_payment_rejected_error(retry_response) + + if not retry_response.is_success: + try: + error_body = retry_response.json() + except Exception: + error_body = {"error": "Request failed"} + raise APIError( + f"{paid_request_error_prefix(retry_response.headers)}: {retry_response.status_code}", + retry_response.status_code, + sanitize_error_response(error_body), + retry_after=retry_after_of(retry_response), + ) + + cost_usd = float(payment_payload.accepted.amount) / 1e6 + self._session_calls += 1 + self._session_total_usd += cost_usd + self._last_call_cost = cost_usd + self._capture_settlement(retry_response) + self._last_raw_headers = retry_response.headers + + return retry_response.json() + + def _get_with_payment_raw( + self, + endpoint: str, + params: dict[str, Any] | None = None, + timeout: float | None = None, + ) -> dict[str, Any]: + """Bounded fresh-signature retry wrapper for raw GET endpoints.""" + import time + + for payment_attempt in range(self._MAX_PAYMENT_RETRIES + 1): + try: + return self._get_with_payment_raw_once(endpoint, params=params, timeout=timeout) + except PaymentError as exc: + if not _is_safe_resign_error(exc) or payment_attempt >= self._MAX_PAYMENT_RETRIES: + raise + time.sleep( + self._PAYMENT_RETRY_BACKOFFS[ + min(payment_attempt, len(self._PAYMENT_RETRY_BACKOFFS) - 1) + ] + ) + + raise PaymentError( # pragma: no cover - bounded loop always returns or raises + "Payment retry loop exhausted without a result." + ) + + def _get_with_payment_raw_once( + self, + endpoint: str, + params: dict[str, Any] | None = None, + timeout: float | None = None, + ) -> dict[str, Any]: + """GET with Solana x402 payment, returning raw JSON.""" + from .cache import get_cached, save_to_cache + + cache_key_body = params or {} + cached = get_cached(endpoint, cache_key_body) + if cached is not None: + return cached + + url = f"{self._api_url}{endpoint}" + headers = {"User-Agent": _get_user_agent()} + eff_timeout = timeout if timeout is not None else self._timeout + + response = self._client.get(url, params=params, headers=headers, timeout=eff_timeout) + + if response.status_code in (502, 503): + import time + + time.sleep(1) + response = self._client.get(url, params=params, headers=headers, timeout=eff_timeout) + + if response.status_code == 402: + # Account rail: a 402 is the account being out of credit, not a + # challenge to sign. Nothing here can sign, so say so plainly. + raise_for_api_key_402(response, self.api_key) + result = self._handle_get_payment_and_retry(url, params, response, timeout=eff_timeout) + save_to_cache( + endpoint, + cache_key_body, + result, + cost_usd=self._last_call_cost, + **self._billing_meta(), + ) + self._log_transaction(endpoint, cache_key_body, result, self._last_call_cost) + return result + + if not response.is_success: + try: + error_body = response.json() + except Exception: + error_body = {"error": "Request failed"} + raise APIError( + f"API error: {response.status_code}", + response.status_code, + sanitize_error_response(error_body), + retry_after=retry_after_of(response), + ) + + return response.json() + + def _handle_get_payment_and_retry( + self, + url: str, + params: dict[str, Any] | None, + response: httpx.Response, + timeout: float | None = None, + ) -> dict[str, Any]: + """Handle 402 for GET endpoints with Solana payment.""" + eff_timeout = timeout if timeout is not None else self._timeout + payment_header = self._extract_payment_header(response) + if not payment_header: + raise PaymentError("402 response but no payment requirements found") + + payment_required = decode_payment_required_header(payment_header) + payment_payload = self._sign_payment(payment_required) + # Before the paid request goes out. Signing alone moves nothing; the + # gateway submitting the signed authorization does, so refusing here + # means nothing settles. + _enforce_spend_limits(self, float(payment_payload.accepted.amount) / 1e6) + encoded_payment = encode_payment_signature_header(payment_payload) + + payment_headers = { + "User-Agent": _get_user_agent(), + "PAYMENT-SIGNATURE": encoded_payment, + } + + retry_response = self._client.get( + url, params=params, headers=payment_headers, timeout=eff_timeout + ) + if retry_response.status_code in (502, 503): + import time + + time.sleep(1) + retry_response = self._client.get( + url, params=params, headers=payment_headers, timeout=eff_timeout + ) + + if retry_response.status_code == 402: + # Account rail: a 402 is the account being out of credit, not a + # challenge to sign. Nothing here can sign, so say so plainly. + raise_for_api_key_402(retry_response, self.api_key) + raise build_payment_rejected_error(retry_response) + + if not retry_response.is_success: + try: + error_body = retry_response.json() + except Exception: + error_body = {"error": "Request failed"} + raise APIError( + f"{paid_request_error_prefix(retry_response.headers)}: {retry_response.status_code}", + retry_response.status_code, + sanitize_error_response(error_body), + retry_after=retry_after_of(retry_response), + ) + + cost_usd = float(payment_payload.accepted.amount) / 1e6 + self._session_calls += 1 + self._session_total_usd += cost_usd + self._last_call_cost = cost_usd + self._capture_settlement(retry_response) + self._last_raw_headers = retry_response.headers + + return retry_response.json() + + def _absolute_url(self, url: str) -> str: + """Resolve a server-supplied relative ``poll_url`` against the API host. + + Poll URLs come back as ``/api/v1/images/generations/``; our + configured ``api_url`` already includes the trailing ``/api`` so + we strip it once to avoid ``/api/api/...``. + """ + if self.api_key: + # api.blockrun.ai serves these routes at /v1/... and answers + # /api/v1/... with wrong_host, so the gateway-minted prefix has to + # come off. Shared with the Base clients, which also pins the + # Authorization header to the gateway's own origin. + return resolve_poll_url(url, self._api_url, self.api_key) + base = self._api_url.removesuffix("/api") + if url.startswith(("http://", "https://")): + # The poll loop sends (and re-signs) the wallet's PAYMENT-SIGNATURE + # against this URL, so an absolute poll_url is pinned to the API + # host+scheme — a gateway response pointing it elsewhere would leak + # the signed payment off-host. + poll, api = httpx.URL(url), httpx.URL(base) + if (poll.scheme, poll.host) != (api.scheme, api.host): + raise APIError( + "Refusing an absolute poll_url on a different host/scheme than " + f"the API ({poll.scheme}://{poll.host} != {api.scheme}://{api.host}); " + "the signed payment header must not be sent off-host.", + 502, + {"poll_url": url}, + ) + return url + return f"{base}{url}" + + def _request_image_with_payment( + self, + endpoint: str, + body: dict[str, Any], + timeout: float | None = None, + *, + poll_budget_seconds: float | None = None, + poll_interval_seconds: float | None = None, + max_resigns: int = 0, + label: str = "Image", + ) -> dict[str, Any]: + """Sign + submit + poll wrapper for async media generation. + + Shared by :meth:`image` (5-min budget, no mid-poll re-signing needed) + and :meth:`video` (15-min budget, ``max_resigns`` re-signs to survive + the 600s x402 authorization window). ``poll_budget_seconds`` / + ``poll_interval_seconds`` default to the image constants; ``label`` + only tunes error text. + + Why this exists instead of reusing ``_request_with_payment_raw``: + the gateway falls back to an async ``202 + poll_url`` flow when a + model exceeds the 30s inline window (gpt-image-2, dall-e-3, slow + nano-banana-pro 4K, etc.). The raw helper treats 202 as success and + feeds the job-stub JSON to ``ImageResponse(**data)``, which then + raises a Pydantic validation error because the ``data`` field + isn't populated until the upstream finishes. + + Flow: + + 1. Probe POST → expect 402 (payment required) from the gateway. + 2. Sign the x402 SVM payload locally; resubmit with PAYMENT-SIGNATURE. + 3. Fast path: 200 with the finished image → settle inline. + 4. Slow path: 202 with ``{id, poll_url, status: queued}`` → loop + GET poll_url with the PAYMENT-SIGNATURE until status = ``completed``. + If a poll 402s (settlement failed, e.g. stale blockhash), re-GET + poll_url for a fresh challenge and re-sign (up to ``max_resigns``). + Settlement happens on the first completed poll; giving up before + then costs the caller nothing. + + Returns the raw response JSON from the final completed response. + """ + import time as _time + + from .cache import get_cached, save_to_cache + + cached = get_cached(endpoint, body) + if cached is not None: + return cached + + url = f"{self._api_url}{endpoint}" + probe_headers = {"Content-Type": "application/json", "User-Agent": _get_user_agent()} + eff_timeout = timeout if timeout is not None else self._image_timeout + + # Step 1: probe — expect 402 unless the model is free or cached upstream. + probe = self._client.post(url, json=body, headers=probe_headers, timeout=eff_timeout) + if probe.status_code in (502, 503): + _time.sleep(1) + probe = self._client.post(url, json=body, headers=probe_headers, timeout=eff_timeout) + + # Account rail: a 402 here is "out of credit", not a challenge to sign. + # Checked before the x402 branch below, which has no signer to reach for + # and, without the optional SDK installed, no decoder either. + raise_for_api_key_402(probe, self.api_key) + + if probe.status_code != 402: + if not probe.is_success: + try: + error_body = probe.json() + except Exception: + error_body = {"error": "Request failed"} + raise APIError( + f"Image request: HTTP {probe.status_code}", + probe.status_code, + sanitize_error_response(error_body), + retry_after=retry_after_of(probe), + ) + # Free / cached upstream — return whatever the gateway gave us. + return probe.json() + + # Step 2: sign x402 SVM payload. + payment_header_str = self._extract_payment_header(probe) + if not payment_header_str: + raise PaymentError("402 response but no payment requirements found") + + payment_required = decode_payment_required_header(payment_header_str) + payment_payload_obj = self._sign_payment(payment_required) + encoded_payment = encode_payment_signature_header(payment_payload_obj) + cost_usd = float(payment_payload_obj.accepted.amount) / 1e6 + # Terms this job is authorized to pay — any mid-poll re-sign must match. + orig_amount = payment_payload_obj.accepted.amount + orig_pay_to = payment_payload_obj.accepted.pay_to + + paid_headers = { + "Content-Type": "application/json", + "User-Agent": _get_user_agent(), + "PAYMENT-SIGNATURE": encoded_payment, + } + + # Step 3: submit with signature. + submit_resp = self._client.post(url, json=body, headers=paid_headers, timeout=eff_timeout) + if submit_resp.status_code in (502, 503): + _time.sleep(1) + submit_resp = self._client.post( + url, json=body, headers=paid_headers, timeout=eff_timeout + ) + + if submit_resp.status_code == 402: + # Account rail: a 402 is the account being out of credit, not a + # challenge to sign. Nothing here can sign, so say so plainly. + raise_for_api_key_402(submit_resp, self.api_key) + raise build_payment_rejected_error(submit_resp) + + if submit_resp.status_code == 200: + # Fast path — image was produced inline. + self._session_calls += 1 + self._session_total_usd += cost_usd + self._last_call_cost = cost_usd + self._capture_settlement(submit_resp) + data = submit_resp.json() + save_to_cache(endpoint, body, data, cost_usd=cost_usd, **self._billing_meta()) + self._log_transaction(endpoint, body, data, cost_usd) + return data + + if submit_resp.status_code != 202: + try: + error_body = submit_resp.json() + except Exception: + error_body = {"error": "Request failed"} + raise APIError( + f"Image request failed: {paid_request_error_prefix(submit_resp.headers)}: HTTP {submit_resp.status_code}", + submit_resp.status_code, + sanitize_error_response(error_body), + retry_after=retry_after_of(submit_resp), + ) + + # Step 4: slow path — poll until completed (or budget exhausted). + try: + submit_data = submit_resp.json() + except Exception: + submit_data = {} + + poll_url_rel = submit_data.get("poll_url") + job_id = submit_data.get("id") + if not poll_url_rel: + raise APIError( + "Slow-path 202 missing poll_url", + 202, + {"response": submit_data}, + ) + poll_url = self._absolute_url(poll_url_rel) + poll_headers = { + "User-Agent": _get_user_agent(), + "PAYMENT-SIGNATURE": encoded_payment, + } + + budget = ( + poll_budget_seconds + if poll_budget_seconds is not None + else self.IMAGE_POLL_BUDGET_SECONDS + ) + interval = ( + poll_interval_seconds + if poll_interval_seconds is not None + else self.IMAGE_POLL_INTERVAL_SECONDS + ) + deadline = _time.monotonic() + budget + last_status = submit_data.get("status", "queued") + resigns_left = max_resigns + last_resign_at = _time.monotonic() + + while _time.monotonic() < deadline: + _time.sleep(interval) + + # Keep the settlement blockhash fresh (poll-based media path only, + # gated on max_resigns). Re-sign the ORIGINAL challenge — same amount/pay_to, + # only a freshly-fetched blockhash — so that whenever upstream flips to + # "completed" the signature is 0 + and _time.monotonic() - last_resign_at >= self.MEDIA_RESIGN_FRESH_SECONDS + ): + try: + fresh_payload = self._sign_payment(payment_required) + poll_headers["PAYMENT-SIGNATURE"] = encode_payment_signature_header( + fresh_payload + ) + last_resign_at = _time.monotonic() + except Exception: + # Best-effort only: a failed proactive re-sign (RPC hiccup, + # SolanaRpcException, etc.) must never abort the poll loop — + # we simply keep the prior signature (pre-fix behaviour). + pass + + poll_resp = self._client.get(poll_url, headers=poll_headers, timeout=eff_timeout) + try: + poll_data = poll_resp.json() + except Exception: + poll_data = {} + last_status = poll_data.get("status", last_status) + + if poll_resp.status_code == 402: + # Account rail: a 402 is the account being out of credit, not a + # challenge to sign. Nothing here can sign, so say so plainly. + raise_for_api_key_402(poll_resp, self.api_key) + # Mid-poll 402 = settlement of the signed payment failed. For + # long jobs this is almost always a stale blockhash: the payment + # was signed at submit time, but the on-chain settlement only + # runs once the job completes, and by then the signed + # transaction's recent-blockhash can be expired — the facilitator + # reports ``transaction_simulation_failed``. The failing poll + # response carries NO fresh challenge, so re-GET poll_url WITHOUT + # the stale signature to solicit a fresh 402 (new blockhash), + # re-sign, and keep polling. Mirrors the Base VideoClient. A + # fresh signature that 402s again is a genuine payment problem. + if resigns_left > 0: + resigns_left -= 1 + resign_payload = None + try: + challenge = self._client.get( + poll_url, + headers={"User-Agent": _get_user_agent()}, + timeout=eff_timeout, + ) + resign_header = self._extract_payment_header(challenge) + if challenge.status_code == 402 and resign_header: + resign_required = decode_payment_required_header(resign_header) + resign_payload = self._sign_payment(resign_required) + except (PaymentError, httpx.HTTPError): + # Challenge GET failed, or signing was rejected — fall + # through to surface the gateway's real 402 reason rather + # than masking it with a network/signing error. Nothing + # settled here. + resign_payload = None + if resign_payload is not None: + # Refuse a re-challenge that reprices or redirects the + # payment vs. what this job originally authorized. This + # PaymentError must propagate (NOT fall through to the + # generic 402). The guard also pins the amount, so the + # submit-time cost_usd stays correct for the ledger. + _assert_same_payment_terms(resign_payload, orig_amount, orig_pay_to) + poll_headers["PAYMENT-SIGNATURE"] = encode_payment_signature_header( + resign_payload + ) + continue + raise build_payment_rejected_error(poll_resp) + + if last_status == "failed": + raise APIError( + f"{label} failed upstream: {poll_data.get('error', 'unknown')}", + poll_resp.status_code, + sanitize_error_response(poll_data if isinstance(poll_data, dict) else {}), + retry_after=retry_after_of(poll_resp), + ) + + # Terminal success is keyed on status, NOT the HTTP code — the + # gateway settles the moment a poll reports completed, so a + # completed-but-non-200 poll (which the caller was already charged + # for) must still be treated as success. + if last_status == "completed": + tx_hash = poll_resp.headers.get("x-payment-receipt") or poll_resp.headers.get( + "X-Payment-Receipt" + ) + if tx_hash and isinstance(poll_data, dict) and not poll_data.get("txHash"): + poll_data["txHash"] = tx_hash + self._session_calls += 1 + self._session_total_usd += cost_usd + self._last_call_cost = cost_usd + self._capture_settlement(poll_resp) + save_to_cache(endpoint, body, poll_data, cost_usd=cost_usd, **self._billing_meta()) + self._log_transaction(endpoint, body, poll_data, cost_usd) + return poll_data + + if poll_resp.status_code in (202, 504): + # 202 = still queued/in_progress; 504 = transient upstream + # hiccup. Both are retriable inside the budget. + continue + + if poll_resp.status_code != 200: + try: + error_body = poll_resp.json() + except Exception: + error_body = {"error": "Request failed"} + raise APIError( + f"{label} poll failed: HTTP {poll_resp.status_code}", + poll_resp.status_code, + sanitize_error_response(error_body), + retry_after=retry_after_of(poll_resp), + ) + + raise APIError( + ( + f"{label} did not complete within {budget:.0f}s " + f"(last status: {last_status}). Settlement only happens on " + "completion, so no payment was taken. The job stays claimable " + "for ~48h — re-poll poll_url with a fresh signature from the " + "same wallet to fetch (and settle) the finished result." + ), + 504, + {"id": job_id, "last_status": last_status, "poll_url": poll_url}, + ) + + def image( + self, + prompt: str, + *, + model: str = "google/nano-banana", + size: str = "1024x1024", + n: int = 1, + quality: str | None = None, + timeout: float | None = None, + ) -> ImageResponse: + """Generate an image from a text prompt (Solana payment). + + Supports the same model catalog as ``ImageClient.generate`` on Base: + ``google/nano-banana``, ``google/nano-banana-pro``, + ``openai/dall-e-3``, ``openai/gpt-image-1``, ``openai/gpt-image-2``, + ``zai/cogview-4``, ``xai/grok-imagine-image``, + ``xai/grok-imagine-image-pro``, ``black-forest/flux-1.1-pro``. + + Slow models (gpt-image-2, dall-e-3) trigger the gateway's async + 202 + poll flow; the client polls transparently until completion + and only settles on the final completed poll. If the poll budget + (``IMAGE_POLL_BUDGET_SECONDS``, 5 min) is exhausted, an + :class:`APIError` 504 is raised and **no payment is taken**. + + Args: + quality: ``low`` / ``medium`` / ``high`` / ``auto`` — latency vs + fidelity, ``openai/gpt-image-*`` only. ``low`` meaningfully + cuts generation time. Solana only: the Base gateway has no + such field, so ``ImageClient`` deliberately omits it rather + than accept a value that would be silently dropped. + + Raises: + ValueError: If ``quality`` is not one of the four accepted values. + """ + body: dict[str, Any] = { + "model": model, + "prompt": prompt, + "size": size, + "n": n, + } + validate_image_quality(quality) + if quality is not None: + body["quality"] = quality + data = self._request_image_with_payment("/v1/images/generations", body, timeout=timeout) + return ImageResponse(**data) + + def image_edit( + self, + prompt: str, + image: str | list[str], + *, + model: str = "openai/gpt-image-2", + mask: str | None = None, + size: str = "1024x1024", + n: int = 1, + quality: str | None = None, + timeout: float | None = None, + ) -> ImageResponse: + """Edit an image using img2img (Solana payment). ``image`` may be a + single data URI or a list of 1-4 data URIs for multi-image fusion + (openai/* up to 4, google/* up to 3). + + Like :meth:`image`, this handles the gateway's async 202 + poll + slow path transparently — settlement only happens on completion. + + Args: + quality: ``low`` / ``medium`` / ``high`` / ``auto``, as in + :meth:`image` — ``openai/gpt-image-*`` only, Solana only. + + Raises: + ValueError: If ``quality`` is not one of the four accepted values. + """ + body: dict[str, Any] = { + "model": model, + "prompt": prompt, + "image": image, + "size": size, + "n": n, + } + if mask is not None: + body["mask"] = mask + validate_image_quality(quality) + if quality is not None: + body["quality"] = quality + + data = self._request_image_with_payment("/v1/images/image2image", body, timeout=timeout) + return ImageResponse(**data) + + # ------------------------------------------------------------------ + # Video generation (Solana payment) — async 202 + poll, mid-poll re-sign + # ------------------------------------------------------------------ + + def video( + self, + prompt: str, + *, + model: str | None = None, + image_url: str | None = None, + last_frame_url: str | None = None, + reference_image_urls: list[str] | None = None, + reference_videos: list[dict[str, str]] | None = None, + reference_audios: list[dict[str, str]] | None = None, + bitrate_mode: str | None = None, + output_format: str | None = None, + camera_fixed: bool | None = None, + safety_identifier: str | None = None, + real_face_asset_id: str | None = None, + duration_seconds: int | None = None, + aspect_ratio: str | None = None, + resolution: str | None = None, + generate_audio: bool | None = None, + seed: int | None = None, + watermark: bool | None = None, + return_last_frame: bool | None = None, + input_type: str | None = None, + budget_seconds: float | None = None, + timeout: float | None = None, + ) -> VideoResponse: + """Generate a video clip from a text prompt (Solana payment). + + Mirrors ``VideoClient.generate`` on Base: submits an async job and + polls until the clip is ready (typical 60-180s). Settlement only + happens on the first completed poll, so a poll-budget timeout takes + **no payment** and leaves the job claimable ~48h. Default model is + ``xai/grok-imagine-video``. + + Args: + input_type: Optional assertion of the seed mode — ``text`` / + ``image`` / ``first_last_frame`` / ``reference``. The gateway + rejects (400, unbilled) if it disagrees with the seed fields + sent, turning a silent wrong-mode clip into an error. See + ``VideoClient.generate``. + """ + body = self._build_video_body( + prompt, + model=model, + image_url=image_url, + last_frame_url=last_frame_url, + reference_image_urls=reference_image_urls, + reference_videos=reference_videos, + reference_audios=reference_audios, + bitrate_mode=bitrate_mode, + output_format=output_format, + camera_fixed=camera_fixed, + safety_identifier=safety_identifier, + real_face_asset_id=real_face_asset_id, + duration_seconds=duration_seconds, + aspect_ratio=aspect_ratio, + resolution=resolution, + generate_audio=generate_audio, + seed=seed, + watermark=watermark, + return_last_frame=return_last_frame, + input_type=input_type, + ) + + data = self._request_image_with_payment( + "/v1/videos/generations", + body, + timeout=timeout, + poll_budget_seconds=( + budget_seconds if budget_seconds is not None else self.VIDEO_POLL_BUDGET_SECONDS + ), + poll_interval_seconds=self.VIDEO_POLL_INTERVAL_SECONDS, + max_resigns=self.MEDIA_POLL_MAX_RESIGNS, + label="Video generation", + ) + return VideoResponse(**data) + + def video_from_content( + self, + content: list[dict[str, Any]], + *, + model: str | None = None, + budget_seconds: float | None = None, + timeout: float | None = None, + **options: Any, + ) -> VideoResponse: + """Generate a video from a Seedance ``content[]`` body (Solana payment). + + Targets ``POST /v1/videos`` (the multimodal ``content`` array shape). + Prefer :meth:`video` for structured kwargs; this exists for migrating + existing ``content[]`` payloads unchanged. + """ + if not content: + raise ValueError("content must be a non-empty list of Seedance content items.") + body: dict[str, Any] = {"content": content, **options} + if model is not None: + body["model"] = model + data = self._request_image_with_payment( + "/v1/videos", + body, + timeout=timeout, + poll_budget_seconds=( + budget_seconds if budget_seconds is not None else self.VIDEO_POLL_BUDGET_SECONDS + ), + poll_interval_seconds=self.VIDEO_POLL_INTERVAL_SECONDS, + max_resigns=self.MEDIA_POLL_MAX_RESIGNS, + label="Video generation", + ) + return VideoResponse(**data) + + # ------------------------------------------------------------------ + # Music generation (Solana payment) + # ------------------------------------------------------------------ + + def music( + self, + prompt: str, + *, + model: str | None = None, + instrumental: bool = True, + lyrics: str | None = None, + timeout: float | None = None, + ) -> MusicResponse: + """Generate a music track from a text prompt (Solana payment). + + Mirrors ``MusicClient.generate`` on Base. Takes 1-3 minutes; the + returned CDN URL is valid ~24h. Default model ``minimax/music-2.5+``. + """ + if instrumental and lyrics and lyrics.strip(): + raise ValueError("Cannot specify lyrics when instrumental is True") + body: dict[str, Any] = { + "model": model or self.MUSIC_DEFAULT_MODEL, + "prompt": prompt, + "instrumental": instrumental, + } + if lyrics and lyrics.strip(): + body["lyrics"] = lyrics.strip() + data = self._request_with_payment_raw("/v1/audio/generations", body, timeout=timeout) + self._attach_receipt(data) + return MusicResponse(**data) + + # ------------------------------------------------------------------ + # Speech / TTS + sound effects (Solana payment) + # ------------------------------------------------------------------ + + def speech( + self, + input: str, + *, + model: str | None = None, + voice: str | None = None, + response_format: str | None = None, + speed: float | None = None, + timeout: float | None = None, + ) -> SpeechResponse: + """Synthesize speech from text (Solana payment). + + Mirrors ``SpeechClient.generate`` on Base. Synchronous; price scales + with character count. Default model ``elevenlabs/flash-v2.5``, default + voice ``sarah``. + """ + body: dict[str, Any] = { + "model": model or self.SPEECH_DEFAULT_MODEL, + "input": input, + } + if voice: + body["voice"] = voice + if response_format: + body["response_format"] = response_format + if speed is not None: + body["speed"] = speed + data = self._request_with_payment_raw("/v1/audio/speech", body, timeout=timeout) + self._attach_receipt(data) + return SpeechResponse(**data) + + def sound_effect( + self, + text: str, + *, + model: str | None = None, + duration_seconds: float | None = None, + prompt_influence: float | None = None, + response_format: str | None = None, + timeout: float | None = None, + ) -> SpeechResponse: + """Generate a cinematic sound effect from a text prompt (Solana + payment). Mirrors ``SpeechClient.sound_effect``. Flat $0.05, <=22s.""" + body: dict[str, Any] = { + "model": model or self.SOUNDFX_DEFAULT_MODEL, + "text": text, + } + if duration_seconds is not None: + body["duration_seconds"] = duration_seconds + if prompt_influence is not None: + body["prompt_influence"] = prompt_influence + if response_format: + body["response_format"] = response_format + data = self._request_with_payment_raw("/v1/audio/sound-effects", body, timeout=timeout) + self._attach_receipt(data) + return SpeechResponse(**data) + + def list_voices(self) -> list[dict[str, Any]]: + """List available speech voices (free).""" + url = f"{self._api_url}/v1/audio/voices" + resp = self._client.get( + url, headers={"User-Agent": _get_user_agent()}, timeout=DEFAULT_FAST_TIMEOUT + ) + if resp.status_code != 200: + try: + error_body = resp.json() + except Exception: + error_body = {"error": "Request failed"} + raise APIError( + f"List voices failed: HTTP {resp.status_code}", + resp.status_code, + sanitize_error_response(error_body), + retry_after=retry_after_of(resp), + ) + data = resp.json() + # Gateway wraps the voice list under "data" (mirrors SpeechClient.list_voices). + return data.get("data", []) if isinstance(data, dict) else data + + # ------------------------------------------------------------------ + # Virtual Portrait enrollment (Solana payment) + # ------------------------------------------------------------------ + + def portrait_enroll(self, name: str, image_url: str) -> PortraitEnrollment: + """Enroll a Virtual Portrait ($0.01 USDC, one-time). Returns the + ``ta_xxxxxxxx`` asset id usable as ``real_face_asset_id`` in + :meth:`video`. Mirrors ``PortraitClient.enroll``.""" + if not name or not name.strip(): + raise ValueError("name is required (1-64 chars)") + if len(name) > 64: + raise ValueError(f"name must be 64 chars or fewer (got {len(name)})") + if not image_url or not image_url.lower().startswith(("https://", "http://")): + raise ValueError("image_url must be an http(s) URL") + body: dict[str, Any] = {"name": name, "image_url": image_url} + data = self._request_with_payment_raw("/v1/portrait/enroll", body) + return PortraitEnrollment(**data) + + def list_portraits(self, wallet_address: str | None = None) -> PortraitList: + """List Virtual Portraits enrolled by a wallet (free, rate-limited).""" + addr = _safe_path_segment(wallet_address or self.get_wallet_address(), "wallet_address") + url = f"{self._api_url}/v1/wallet/{addr}/portraits" + resp = self._client.get( + url, headers={"User-Agent": _get_user_agent()}, timeout=DEFAULT_FAST_TIMEOUT + ) + if resp.status_code != 200: + try: + error_body = resp.json() + except Exception: + error_body = {"error": "Request failed"} + raise APIError( + "Portrait listing failed", + resp.status_code, + sanitize_error_response(error_body), + retry_after=retry_after_of(resp), + ) + return PortraitList(**resp.json()) + + # ------------------------------------------------------------------ + # RealFace enrollment (Solana payment) + # ------------------------------------------------------------------ + + def realface_init(self, name: str, group_id: str | None = None) -> RealFaceInit: + """Start/refresh a RealFace enrollment (free, rate-limited). Returns + the ``group_id`` and an ``h5_link`` (render as a QR for the real + person's phone liveness check).""" + if not name or not name.strip(): + raise ValueError("name is required (1-64 chars)") + if len(name) > 64: + raise ValueError(f"name must be 64 chars or fewer (got {len(name)})") + if group_id is not None and not _GROUP_ID_RE.match(group_id): + raise ValueError("group_id must look like 'legacy_rf_'") + body: dict[str, Any] = {"name": name} + if group_id: + body["groupId"] = group_id + url = f"{self._api_url}/v1/realface/init" + resp = self._client.post( + url, + json=body, + headers={"Content-Type": "application/json", "User-Agent": _get_user_agent()}, + timeout=DEFAULT_FAST_TIMEOUT, + ) + if resp.status_code != 200: + try: + error_body = resp.json() + except Exception: + error_body = {"error": "Request failed"} + raise APIError( + "RealFace init failed", + resp.status_code, + sanitize_error_response(error_body), + retry_after=retry_after_of(resp), + ) + return RealFaceInit(**resp.json()) + + def realface_status(self, group_id: str) -> RealFaceStatus: + """Poll a RealFace group's state (free, rate-limited).""" + if not group_id or not _GROUP_ID_RE.match(group_id): + raise ValueError("group_id must look like 'legacy_rf_'") + url = f"{self._api_url}/v1/realface/status" + resp = self._client.get( + url, + params={"groupId": group_id}, + headers={"User-Agent": _get_user_agent()}, + timeout=DEFAULT_FAST_TIMEOUT, + ) + if resp.status_code != 200: + try: + error_body = resp.json() + except Exception: + error_body = {"error": "Request failed"} + raise APIError( + "RealFace status check failed", + resp.status_code, + sanitize_error_response(error_body), + retry_after=retry_after_of(resp), + ) + return RealFaceStatus(**resp.json()) + + def realface_wait_for_active( + self, + group_id: str, + timeout_seconds: float = 180.0, + poll_interval_seconds: float = 4.0, + ) -> RealFaceStatus: + """Block until the RealFace group is active (person finished the phone + liveness check). Convenience wrapper around :meth:`realface_status`.""" + import time as _time + + if poll_interval_seconds <= 0: + raise ValueError("poll_interval_seconds must be positive") + deadline = _time.monotonic() + timeout_seconds + while True: + state = self.realface_status(group_id) + if state.ready_to_finalize: + return state + if _time.monotonic() + poll_interval_seconds >= deadline: + raise TimeoutError( + f"RealFace group {group_id} not active after {timeout_seconds:.0f}s " + f"(last status: {state.status!r})." + ) + _time.sleep(poll_interval_seconds) + + def realface_enroll(self, name: str, image_url: str, group_id: str) -> RealFaceEnrollment: + """Finalize a RealFace enrollment ($0.01 USDC). Requires the group to + be active (see :meth:`realface_wait_for_active`).""" + if not name or not name.strip(): + raise ValueError("name is required (1-64 chars)") + if len(name) > 64: + raise ValueError(f"name must be 64 chars or fewer (got {len(name)})") + if not image_url or not image_url.lower().startswith(("https://", "http://")): + raise ValueError("image_url must be an http(s) URL") + if not group_id or not _GROUP_ID_RE.match(group_id): + raise ValueError("group_id must look like 'legacy_rf_'") + body: dict[str, Any] = {"name": name, "image_url": image_url, "group_id": group_id} + data = self._request_with_payment_raw("/v1/realface/enroll", body) + return RealFaceEnrollment(**data) + + def list_realfaces(self, wallet_address: str | None = None) -> RealFaceList: + """List RealFace assets enrolled by a wallet (free, rate-limited).""" + addr = _safe_path_segment(wallet_address or self.get_wallet_address(), "wallet_address") + url = f"{self._api_url}/v1/wallet/{addr}/realfaces" + resp = self._client.get( + url, headers={"User-Agent": _get_user_agent()}, timeout=DEFAULT_FAST_TIMEOUT + ) + if resp.status_code != 200: + try: + error_body = resp.json() + except Exception: + error_body = {"error": "Request failed"} + raise APIError( + "RealFace listing failed", + resp.status_code, + sanitize_error_response(error_body), + retry_after=retry_after_of(resp), + ) + return RealFaceList(**resp.json()) + + # ------------------------------------------------------------------ + # Pyth market data (Solana payment for paid categories) + # ------------------------------------------------------------------ + + @staticmethod + def _build_video_body( + prompt: str, + *, + model: str | None, + image_url: str | None, + last_frame_url: str | None, + reference_image_urls: list[str] | None, + reference_videos: list[dict[str, str]] | None = None, + reference_audios: list[dict[str, str]] | None = None, + bitrate_mode: str | None = None, + output_format: str | None = None, + camera_fixed: bool | None = None, + safety_identifier: str | None = None, + real_face_asset_id: str | None, + duration_seconds: int | None, + aspect_ratio: str | None, + resolution: str | None, + generate_audio: bool | None, + seed: int | None, + watermark: bool | None, + return_last_frame: bool | None, + input_type: str | None, + ) -> dict[str, Any]: + """Validate video kwargs and build the request body. Shared by the sync + and async ``video()`` so their validation and payload never drift. + + Every param is required (pass None to omit) precisely so a caller can't + silently drop one — the drift this builder exists to prevent.""" + if image_url and real_face_asset_id: + raise ValueError( + "image_url and real_face_asset_id are mutually exclusive; pass at most one." + ) + if last_frame_url and not image_url: + raise ValueError( + "last_frame_url requires image_url: image_url seeds the FIRST frame and " + "last_frame_url the FINAL frame — send both." + ) + if last_frame_url and real_face_asset_id: + raise ValueError( + "last_frame_url and real_face_asset_id are mutually exclusive; " + "first-and-last-frame uses image_url + last_frame_url." + ) + if reference_image_urls: + if image_url or last_frame_url or real_face_asset_id: + raise ValueError( + "reference_image_urls is mutually exclusive with image_url, " + "last_frame_url, and real_face_asset_id." + ) + image_limit = 30 if (model or "").removeprefix("bytedance/") == "seedance-2.5" else 9 + if len(reference_image_urls) > image_limit: + raise ValueError(f"reference_image_urls accepts at most {image_limit} images.") + if (reference_videos or reference_audios) and ( + image_url or last_frame_url or real_face_asset_id + ): + raise ValueError( + "reference media is mutually exclusive with frame-seed inputs; use reference_image_urls." + ) + for clips in (reference_videos, reference_audios): + if clips is not None: + if not 1 <= len(clips) <= 3: + raise ValueError("reference media accepts 1 to 3 clips per type.") + if any( + not isinstance(clip, dict) + or not isinstance(clip.get("url"), str) + or not clip["url"].startswith(("https://", "http://")) + or clip.get("role", "reference") != "reference" + for clip in clips + ): + raise ValueError( + "reference clips require an http(s) URL and optional reference role." + ) + if real_face_asset_id is not None and not real_face_asset_id.startswith("ta_"): + raise ValueError( + "real_face_asset_id must start with 'ta_' " + "(a Virtual Portrait or RealFace asset id, e.g. 'ta_abc123xyz')" + ) + validate_video_input_type(input_type) + + body: dict[str, Any] = { + "model": model or SolanaLLMClient.VIDEO_DEFAULT_MODEL, + "prompt": prompt, + } + if image_url: + body["image_url"] = image_url + if last_frame_url: + body["last_frame_url"] = last_frame_url + if reference_image_urls: + body["reference_image_urls"] = reference_image_urls + if reference_videos is not None: + body["reference_videos"] = reference_videos + if reference_audios is not None: + body["reference_audios"] = reference_audios + if bitrate_mode is not None: + body["bitrate_mode"] = bitrate_mode + if output_format is not None: + body["output_format"] = output_format + if camera_fixed is not None: + body["camera_fixed"] = camera_fixed + if safety_identifier is not None: + body["safety_identifier"] = safety_identifier + if real_face_asset_id: + body["real_face_asset_id"] = real_face_asset_id + if duration_seconds is not None: + body["duration_seconds"] = duration_seconds + if aspect_ratio is not None: + body["aspect_ratio"] = aspect_ratio + if resolution is not None: + body["resolution"] = resolution + if generate_audio is not None: + body["generate_audio"] = generate_audio + if seed is not None: + body["seed"] = seed + if watermark is not None: + body["watermark"] = watermark + if return_last_frame is not None: + body["return_last_frame"] = return_last_frame + if input_type is not None: + body["input_type"] = input_type + return body + + @staticmethod + def _rpc_response( + data: Any, headers: httpx.Headers | None, fallback_network: str + ) -> RpcResponse: + """Build an RpcResponse, surfacing gateway metadata from the paid + response headers (canonical network, cache hit, settlement tx) exactly + like the Base RPCClient. Strips body keys that would collide with those + metadata kwargs.""" + if not isinstance(data, dict): + data = {"result": data} + else: + data = {k: v for k, v in data.items() if k not in ("network", "cache_hit", "tx_hash")} + hdrs = headers if headers is not None else httpx.Headers() + return RpcResponse( + **data, + network=hdrs.get("x-network") or fallback_network, + cache_hit=(hdrs.get("x-cache", "") or "").upper() == "HIT", + tx_hash=_receipt_from_headers(hdrs), + ) + + @staticmethod + def _price_category_path( + category: str, market: str | None, kind: str, symbol: str | None + ) -> str: + if category == "stocks": + if not market: + raise ValueError("market is required for category='stocks' (e.g. market='us')") + base = f"/v1/stocks/{_safe_path_segment(market, 'market')}" + elif category in ("crypto", "fx", "commodity", "usstock"): + base = f"/v1/{category}" + else: + raise ValueError(f"Unknown category: {category}") + if symbol is None: + return f"{base}/{kind}" + return f"{base}/{kind}/{_safe_path_segment(symbol.upper(), 'symbol')}" + + def price( + self, + category: Category, + symbol: str, + *, + market: Market | None = None, + session: Session | None = None, + ) -> PricePoint: + """Fetch a realtime Pyth price quote (Solana payment for paid + categories). ``market`` is required for ``category='stocks'``.""" + endpoint = self._price_category_path(category, market, "price", symbol) + params: dict[str, Any] = {} + if session is not None: + params["session"] = session + data = self._get_with_payment_raw( + endpoint, params=params or None, timeout=DEFAULT_FAST_TIMEOUT + ) + return PricePoint( + symbol=data.get("symbol", symbol.upper()), + price=data.get("price"), + publish_time=data.get("publishTime"), + confidence=data.get("confidence"), + feed_id=data.get("feedId"), + **{ + k: v + for k, v in data.items() + if k not in {"symbol", "price", "publishTime", "confidence", "feedId"} + }, + ) + + def price_history( + self, + category: Category, + symbol: str, + *, + resolution: Resolution = "D", + from_ts: int, + to_ts: int, + market: Market | None = None, + session: Session | None = None, + ) -> PriceHistoryResponse: + """Fetch OHLC bars between two Unix timestamps (seconds).""" + endpoint = self._price_category_path(category, market, "history", symbol) + params: dict[str, Any] = {"resolution": resolution, "from": from_ts, "to": to_ts} + if session is not None: + params["session"] = session + data = self._get_with_payment_raw(endpoint, params=params, timeout=DEFAULT_FAST_TIMEOUT) + return PriceHistoryResponse( + symbol=data.get("symbol", symbol.upper()), + resolution=data.get("resolution", resolution), + bars=data.get("bars", []), + **{k: v for k, v in data.items() if k not in {"symbol", "resolution", "bars"}}, + ) + + def list_symbols( + self, + category: Category, + *, + q: str | None = None, + limit: int = 100, + market: Market | None = None, + ) -> SymbolListResponse: + """List available symbols in a Pyth category (free discovery).""" + endpoint = self._price_category_path(category, market, "list", None) + params: dict[str, Any] = {"limit": limit} + if q: + params["q"] = q + data = self._get_with_payment_raw(endpoint, params=params, timeout=DEFAULT_FAST_TIMEOUT) + if isinstance(data, list): + return SymbolListResponse(symbols=data, count=len(data)) + return SymbolListResponse( + symbols=data.get("symbols", data.get("feeds", [])), + count=data.get("count"), + **{k: v for k, v in data.items() if k not in {"symbols", "feeds", "count"}}, + ) + + # ------------------------------------------------------------------ + # Multi-chain JSON-RPC (Solana payment) + # ------------------------------------------------------------------ + + def rpc( + self, + network: str, + method: str, + params: list[Any] | None = None, + *, + id: str | int = 1, + ) -> RpcResponse: + """Make a single JSON-RPC 2.0 call (Solana payment, flat $0.002). + + Mirrors ``RPCClient.call``. ``network`` may be a chain name or alias + (``eth``, ``sol``, ``base`` …); the gateway resolves it. + """ + _safe_path_segment(network, "network") + body: dict[str, Any] = {"jsonrpc": "2.0", "id": id, "method": method} + if params is not None: + body["params"] = params + data = self._request_with_payment_raw(f"/v1/rpc/{network}", body) + return self._rpc_response(data, self._last_raw_headers, network) + + def rpc_batch(self, network: str, requests: list[dict[str, Any]]) -> list[RpcResponse]: + """Make a JSON-RPC 2.0 batch call (Solana payment, $0.002 x N).""" + if not requests: + raise ValueError("batch requires at least one request") + _safe_path_segment(network, "network") + body: list[dict[str, Any]] = [] + for i, req in enumerate(requests): + if "method" not in req: + raise ValueError(f"batch request {i} is missing 'method'") + body.append({"jsonrpc": "2.0", "id": i + 1, **req}) + data = self._request_with_payment_raw(f"/v1/rpc/{network}", body) # type: ignore[arg-type] + headers = self._last_raw_headers + if not isinstance(data, list): + data = [data] + return [self._rpc_response(item, headers, network) for item in data] + + def search( + self, + query: str, + *, + sources: list[str] | None = None, + max_results: int = 10, + from_date: str | None = None, + to_date: str | None = None, + timeout: float | None = None, + ) -> SearchResult: + """Standalone search (Solana payment). + + ``timeout`` overrides the per-call HTTP timeout (defaults to + ``DEFAULT_SEARCH_TIMEOUT`` — deep web/X tool-use can run minutes). + """ + body: dict[str, Any] = { + "query": query, + "max_results": max_results, + } + if sources is not None: + body["sources"] = sources + if from_date is not None: + body["from_date"] = from_date + if to_date is not None: + body["to_date"] = to_date + + eff_timeout = timeout if timeout is not None else self._search_timeout + data = self._request_with_payment_raw("/v1/search", body, timeout=eff_timeout) + return SearchResult(**data) + + # ── Prediction Markets (Powered by Predexon) ──────────────────────────── + + def pm(self, path: str, **params: Any) -> dict[str, Any]: + """Query Predexon prediction market data (GET, Solana payment). Powered by Predexon.""" + return self._get_with_payment_raw(f"/v1/pm/{path}", params or None) + + def pm_query(self, path: str, query: dict[str, Any]) -> dict[str, Any]: + """Structured query for Predexon data (POST, Solana payment). Powered by Predexon.""" + return self._request_with_payment_raw(f"/v1/pm/{path}", query) + + def pm_markets(self, **params: Any) -> dict[str, Any]: + """RETIRED — ``/v1/pm/markets`` no longer exists. + + Predexon sunset market matching on 2026-07-20 and the whole + canonical layer went with it, so this path returns 410 upstream. + Use ``pm("markets/search", q=...)`` for cross-venue lookups. + + Kept as a raising stub rather than deleted so upgrading does not + break imports or attribute access; calling it fails immediately + instead of after a paid round trip. + + :raises RetiredEndpointError: always. + """ + raise RetiredEndpointError( + "/v1/pm/markets was sunset by Predexon on 2026-07-20 (upstream 410). Use pm('markets/search', q=...) for cross-venue lookups." + ) + + def pm_listings(self, **params: Any) -> dict[str, Any]: + """RETIRED — ``/v1/pm/markets/listings`` no longer exists. + + Predexon sunset market matching on 2026-07-20 and the whole + canonical layer went with it, so this path returns 410 upstream. + Use ``pm("markets/search", q=...)`` for cross-venue lookups. + + Kept as a raising stub rather than deleted so upgrading does not + break imports or attribute access; calling it fails immediately + instead of after a paid round trip. + + :raises RetiredEndpointError: always. + """ + raise RetiredEndpointError( + "/v1/pm/markets/listings was sunset by Predexon on 2026-07-20 (upstream 410). Use pm('markets/search', q=...) for cross-venue lookups." + ) + + def pm_outcome(self, predexon_id: str) -> dict[str, Any]: + """RETIRED — ``/v1/pm/outcomes/{predexon_id}`` no longer exists. + + Predexon sunset market matching on 2026-07-20 and the whole + canonical layer went with it, so this path returns 410 upstream. + Use ``pm("markets/search", q=...)`` for cross-venue lookups. + + Kept as a raising stub rather than deleted so upgrading does not + break imports or attribute access; calling it fails immediately + instead of after a paid round trip. + + :raises RetiredEndpointError: always. + """ + raise RetiredEndpointError( + "/v1/pm/outcomes/{predexon_id} was sunset by Predexon on 2026-07-20 (upstream 410). Use pm('markets/search', q=...) for cross-venue lookups." + ) + + def pm_polymarket_markets(self, **params: Any) -> dict[str, Any]: + """List Polymarket markets (Predexon v2). Tier 1 ($0.001/call).""" + return self.pm("polymarket/markets", **params) + + def pm_polymarket_events(self, **params: Any) -> dict[str, Any]: + """List Polymarket events (Predexon v2). Tier 1 ($0.001/call).""" + return self.pm("polymarket/events", **params) + + def pm_polymarket_markets_keyset(self, **params: Any) -> dict[str, Any]: + """Polymarket markets with cursor-based keyset pagination. Tier 1 ($0.001/call).""" + return self.pm("polymarket/markets/keyset", **params) + + def pm_polymarket_events_keyset(self, **params: Any) -> dict[str, Any]: + """Polymarket events with cursor-based keyset pagination. Tier 1 ($0.001/call).""" + return self.pm("polymarket/events/keyset", **params) + + def pm_polymarket_positions(self, **params: Any) -> dict[str, Any]: + """Polymarket open positions (per-wallet, market-level PnL). + Tier 1 ($0.001/call).""" + return self.pm("polymarket/positions", **params) + + def pm_polymarket_trades(self, **params: Any) -> dict[str, Any]: + """Recent Polymarket trades. Tier 1 ($0.001/call).""" + return self.pm("polymarket/trades", **params) + + def pm_polymarket_leaderboard(self, **params: Any) -> dict[str, Any]: + """Polymarket trader leaderboard. Tier 1 ($0.001/call).""" + return self.pm("polymarket/leaderboard", **params) + + def pm_kalshi_markets(self, **params: Any) -> dict[str, Any]: + """List Kalshi markets. Tier 1 ($0.001/call).""" + return self.pm("kalshi/markets", **params) + + def pm_limitless_markets(self, **params: Any) -> dict[str, Any]: + """List Limitless markets. Tier 1 ($0.001/call).""" + return self.pm("limitless/markets", **params) + + def pm_sports_categories(self) -> dict[str, Any]: + """List available sports categories. Tier 1 ($0.001/call). + + .. warning:: + Upstream is returning 500 for every ``sports/*`` path as of + 2026-08-04. The route still resolves, so this keeps working the + moment Predexon restores it, but do not build on it yet. + """ + return self.pm("sports/categories") + + def pm_sports_markets(self, **params: Any) -> dict[str, Any]: + """List sports markets grouped by game. Tier 1 ($0.001/call). + + .. warning:: + Upstream is returning 500 for every ``sports/*`` path as of + 2026-08-04. The route still resolves, so this keeps working the + moment Predexon restores it, but do not build on it yet. + """ + return self.pm("sports/markets", **params) + + def pm_wallet_identity(self, wallet: str) -> dict[str, Any]: + """Identity + profile for one wallet. Tier 2 ($0.005/call).""" + return self.pm(f"polymarket/wallet/identity/{wallet}") + + def pm_wallet_identities(self, addresses: list[str]) -> dict[str, Any]: + """Bulk identity for up to 200 wallet addresses. Tier 2 ($0.005/call).""" + return self.pm_query("polymarket/wallet/identities", {"addresses": addresses}) + + def pm_wallet_cluster(self, address: str) -> dict[str, Any]: + """Wallet-cluster discovery (on-chain transfers + identity proofs). + Tier 2 ($0.005/call).""" + return self.pm(f"polymarket/wallet/{address}/cluster") + + # ── Exa Web Search (Powered by Exa) ───────────────────────────────────── + + def exa(self, path: str, body: dict[str, Any]) -> dict[str, Any]: + """Generic Exa endpoint proxy (POST, Solana payment). Powered by Exa. + + Args: + path: Exa endpoint — one of: "search", "find-similar", "contents", "answer" + body: Request body (see Exa API docs) + + Example:: + + result = client.exa("search", {"query": "latest AI research", "numResults": 5}) + """ + return self._request_with_payment_raw(f"/v1/exa/{path}", body, timeout=self._search_timeout) + + def exa_search(self, query: str, **kwargs: Any) -> dict[str, Any]: + """Neural and keyword web search via Exa (Solana payment, $0.01/request). + + Args: + query: Search query string + **kwargs: Additional Exa parameters (numResults, category, useAutoprompt, etc.) + + Example:: + + results = client.exa_search("latest AI papers", numResults=5) + """ + return self._request_with_payment_raw( + "/v1/exa/search", {"query": query, **kwargs}, timeout=self._search_timeout + ) + + def exa_find_similar(self, url: str, **kwargs: Any) -> dict[str, Any]: + """Find pages semantically similar to a given URL via Exa (Solana payment, $0.01/request). + + Args: + url: URL to find similar pages for + **kwargs: Additional Exa parameters (numResults, etc.) + + Example:: + + results = client.exa_find_similar("https://openai.com/research/gpt-4", numResults=5) + """ + return self._request_with_payment_raw( + "/v1/exa/find-similar", {"url": url, **kwargs}, timeout=self._search_timeout + ) + + def exa_contents(self, urls: list[str], **kwargs: Any) -> dict[str, Any]: + """Extract full text content from URLs via Exa (Solana payment, $0.002/URL). + + Args: + urls: List of URLs to extract content from + **kwargs: Additional Exa parameters (text, highlights, summary, etc.) + + Example:: + + data = client.exa_contents(["https://arxiv.org/abs/2303.08774"]) + """ + return self._request_with_payment_raw( + "/v1/exa/contents", {"urls": urls, **kwargs}, timeout=self._search_timeout + ) + + def exa_answer(self, query: str, **kwargs: Any) -> dict[str, Any]: + """AI-generated answer grounded in live web search via Exa (Solana payment, $0.01/request). + + Args: + query: Question to answer + **kwargs: Additional Exa parameters + + Example:: + + answer = client.exa_answer("What is the current state of AI safety research?") + """ + return self._request_with_payment_raw( + "/v1/exa/answer", {"query": query, **kwargs}, timeout=self._search_timeout + ) + + # ── DefiLlama (DeFi protocols / TVL / yields / prices) ────────────────── + + def defi(self, path: str, **params: Any) -> dict[str, Any]: + """Query DefiLlama DeFi data (GET, Solana payment). $0.005/call + ($0.001 for prices/{coins}).""" + return self._get_with_payment_raw(f"/v1/defillama/{path}", params or None) + + def defi_protocols(self) -> dict[str, Any]: + """All DeFi protocols with TVL ($0.005/call).""" + return self.defi("protocols") + + def defi_protocol(self, slug: str) -> dict[str, Any]: + """Single protocol details + historical TVL ($0.005/call).""" + return self.defi(f"protocol/{slug}") + + def defi_chains(self) -> dict[str, Any]: + """Current TVL of every chain ($0.005/call).""" + return self.defi("chains") + + def defi_yields(self, **params: Any) -> dict[str, Any]: + """Yield pools with APY/TVL ($0.005/call).""" + return self.defi("yields", **params) + + def defi_prices(self, coins: list[str] | str) -> dict[str, Any]: + """Token price lookup ($0.001/call).""" + joined = ",".join(coins) if isinstance(coins, list) else coins + return self.defi(f"prices/{joined}") + + # ── 0x DEX (swap quotes + gasless) — free passthrough ─────────────────── + + def dex( + self, + path: str, + *, + method: str = "GET", + body: dict[str, Any] | None = None, + **params: Any, + ) -> dict[str, Any]: + """Query the 0x Swap / Gasless APIs (free — no x402 payment).""" + endpoint = f"/v1/zerox/{path}" + if method.upper() == "POST": + return self._request_with_payment_raw(endpoint, body or {}) + return self._get_with_payment_raw(endpoint, params or None) + + def dex_price(self, **params: Any) -> dict[str, Any]: + """Indicative Permit2 swap price — no commitment (free).""" + return self.dex("price", **params) + + def dex_quote(self, **params: Any) -> dict[str, Any]: + """Firm Permit2 swap quote with permit2.eip712 + tx data (free).""" + return self.dex("quote", **params) + + def dex_gasless_price(self, **params: Any) -> dict[str, Any]: + """Gasless indicative price quote (free).""" + return self.dex("gasless/price", **params) + + def dex_gasless_quote(self, **params: Any) -> dict[str, Any]: + """Gasless firm quote — returns trade.eip712 to sign (free).""" + return self.dex("gasless/quote", **params) + + def dex_gasless_submit(self, body: dict[str, Any]) -> dict[str, Any]: + """Submit a signed gasless trade; the 0x relayer pays gas (free).""" + return self.dex("gasless/submit", method="POST", body=body) + + def dex_gasless_status(self, trade_hash: str) -> dict[str, Any]: + """Poll a gasless trade's status by tradeHash (free).""" + return self.dex(f"gasless/status/{trade_hash}") + + def dex_chains(self) -> dict[str, Any]: + """Chains where the Swap API is supported (free).""" + return self.dex("swap/chains") + + def dex_gasless_chains(self) -> dict[str, Any]: + """Chains where the Gasless API is supported (free).""" + return self.dex("gasless/chains") + + # ── Modal Sandbox (pay-per-call cloud compute) ─────────────────────────── + + def modal(self, path: str, body: dict[str, Any] | None = None) -> dict[str, Any]: + """Call the Modal sandbox compute API (POST, Solana payment).""" + return self._request_with_payment_raw(f"/v1/modal/{path}", body or {}) + + def modal_sandbox_create(self, **body: Any) -> dict[str, Any]: + """Create a sandboxed compute environment ($0.01 CPU / $0.05 GPU).""" + return self.modal("sandbox/create", body) + + def modal_sandbox_exec( + self, sandbox_id: str, command: list[str], **body: Any + ) -> dict[str, Any]: + """Execute a command in a sandbox; returns stdout/stderr ($0.001).""" + return self.modal("sandbox/exec", {"sandbox_id": sandbox_id, "command": command, **body}) + + def modal_sandbox_status(self, sandbox_id: str) -> dict[str, Any]: + """Check a sandbox's status ($0.001).""" + return self.modal("sandbox/status", {"sandbox_id": sandbox_id}) + + def modal_sandbox_terminate(self, sandbox_id: str) -> dict[str, Any]: + """Terminate a sandbox ($0.001).""" + return self.modal("sandbox/terminate", {"sandbox_id": sandbox_id}) + + +# =========================================================================== +# AsyncSolanaLLMClient — async mirror of SolanaLLMClient (chat only, v0.22.0) +# =========================================================================== +# +# Scope for the first release: chat completions, sync **and** streaming. Image, +# music, video, exa, predexon are sync-only on Solana for now — same as the +# Solana sync class shipped initially. They can be added in follow-up releases. + + +class AsyncSolanaLLMClient: + """ + Async BlockRun Solana LLM Client — pays via Solana USDC x402. + + Mirrors :class:`SolanaLLMClient` but exposes ``await``-able methods so + Python ``asyncio`` callers (FastAPI handlers, LiteLLM Proxy, etc.) don't + have to thread-pool around blocking I/O. + + Usage:: + + client = AsyncSolanaLLMClient() # SOLANA_WALLET_KEY env + resp = await client.chat_completion( + "openai/gpt-5.5", + [{"role": "user", "content": "gm Solana"}], + ) + await client.close() + """ + + SOLANA_API_URL = SOLANA_API_URL + _STREAM_5XX_STATUSES = SolanaLLMClient._STREAM_5XX_STATUSES + _STREAM_5XX_BACKOFFS = SolanaLLMClient._STREAM_5XX_BACKOFFS + _MAX_PAYMENT_RETRIES = SolanaLLMClient._MAX_PAYMENT_RETRIES + _PAYMENT_RETRY_BACKOFFS = SolanaLLMClient._PAYMENT_RETRY_BACKOFFS + + def __init__( + self, + private_key: str | None = None, + api_url: str = SOLANA_API_URL, + rpc_url: str | None = None, + timeout: float = DEFAULT_TIMEOUT, + image_timeout: float = DEFAULT_IMAGE_TIMEOUT, + search_timeout: float = DEFAULT_SEARCH_TIMEOUT, + rpc_headers: dict[str, str] | None = None, + transaction_log: bool | str | os.PathLike[str] | None = None, + max_cost_per_call: float | None = None, + max_session_cost: float | None = None, + ) -> None: + """Async mirror of :class:`SolanaLLMClient.__init__`. Same env-var + fallback for ``rpc_url`` / ``rpc_headers`` — see + :func:`_resolve_rpc_config`. ``transaction_log`` works the same way + — opt-in per-call log to a project folder (default ``./log/``).""" + # An API key answers the chain question rather than being answered by + # it: api.blockrun.ai settles from credit, so there is no Solana + # transfer to sign, no wallet to load, and no reason to require the + # x402 SDK at all. Checked before the import guard for exactly that + # reason. + api_key = resolve_api_key(private_key) + if not api_key and not _HAS_X402: + raise ImportError( + "Solana payment requires the x402 SDK. " + "Install with: pip install blockrun-llm[solana]" + ) + from .solana_wallet import load_solana_wallet + + key = ( + None + if api_key + else ( + private_key + or os.environ.get("SOLANA_WALLET_KEY") + or load_solana_wallet() # disk: newest ~/.*/solana-wallet.json, else ~/.blockrun/.solana-session + ) + ) + if not api_key and not key: + raise missing_credential_error( + extra="Set SOLANA_WALLET_KEY, or keep a Solana wallet on disk " + "(~/./solana-wallet.json or ~/.blockrun/.solana-session)" + ) + self.api_key = api_key + self._private_key = key + if api_key: + # A key is answered by api.blockrun.ai, never by sol.blockrun.ai — + # so the Solana default must not reach the account rail. An + # api_url the caller actually typed still wins, as it does on + # every other client. + override = None if api_url == SOLANA_API_URL else api_url + self._api_url = api_key_base_url(override) + validate_api_url(self._api_url) + else: + validate_api_url(api_url) + self._api_url = api_url.rstrip("/") + # Model pricing cache for smart routing + self._model_pricing_cache: dict[str, dict[str, float]] | None = None + + resolved_url, resolved_headers = _resolve_rpc_config(rpc_url, rpc_headers) + self._rpc_url = resolved_url + self._rpc_headers = resolved_headers + + self._timeout = timeout + self._image_timeout = image_timeout + self._search_timeout = search_timeout + self._client = httpx.AsyncClient(timeout=timeout, headers=auth_headers(api_key)) + self._session_total_usd = 0.0 + # Opt-in spend limits. None (the default) means unlimited, which is the + # behavior every release before 1.9.0 had: every 402 quote was signed + # automatically with nothing compared against anything. + self._max_cost_per_call = resolve_spend_limit( + max_cost_per_call, "BLOCKRUN_MAX_COST_PER_CALL" + ) + self._max_session_cost = resolve_spend_limit(max_session_cost, "BLOCKRUN_MAX_SESSION_COST") + self._session_calls = 0 + self._last_call_cost: float = 0.0 + self._address: str | None = None + + log_dir = _resolve_log_dir(transaction_log) + self._tx_logger: TransactionLogger | None = ( + TransactionLogger(log_dir) if log_dir is not None else None + ) + self._last_settlement: dict[str, Any] | None = None + # Response headers from the most recent raw paid POST — consumed by + # rpc()/music()/speech() to surface the settlement receipt + gateway + # metadata the shared JSON-only helper would otherwise drop. Read it + # immediately after the helper returns (no intervening await). + self._last_raw_headers: httpx.Headers | None = None + + if api_key: + self._x402_client = None + self._payment_lock = None + return + + # Async x402 client + same SVM signer the sync class uses. + from x402 import x402Client # local import to keep optional dep clean + + self._x402_client = x402Client() + try: + signer = _create_signer(self._private_key) + except Exception as e: + # Parity with the Base client, which validates the resolved key up + # front: turn a malformed key (incl. one auto-loaded from disk) into + # a clean error instead of a raw base58/solders exception. + raise ValueError( + "Invalid Solana private key (expected a base58-encoded keypair " "or 32-byte seed)." + ) from e + _register_svm_with_headers(self._x402_client, signer, resolved_url, resolved_headers) + # Lazily created on first sign (avoids binding asyncio.Lock to a loop at + # construction time). Serializes the async signing critical section so a + # shared client is safe across concurrent coroutines — see _sign_payment. + self._payment_lock: asyncio.Lock | None = None + + async def _sign_payment(self, payment_required: Any) -> Any: + """Task-safe async wrapper around ``x402_client.create_payment_payload``. + + Mirrors the sync :meth:`SolanaLLMClient._sign_payment`: concurrent + coroutines sharing one client would otherwise race on the x402 client's + nonce/auth state and trip replay / amount-mismatch rejections under load. + """ + if self._payment_lock is None: + self._payment_lock = asyncio.Lock() + async with self._payment_lock: + return await self._x402_client.create_payment_payload(payment_required) + + def _capture_settlement(self, response: httpx.Response) -> dict[str, Any] | None: + """Async-Solana twin of :meth:`SolanaLLMClient._capture_settlement`.""" + header = read_settlement_header(response.headers) + settlement = decode_settlement_header(header) + self._last_settlement = settlement + return settlement + + def _attach_receipt(self, data: Any) -> None: + """Inject the settlement tx hash from the most recent paid POST into a + raw response dict under ``txHash`` (mirrors the Base Music/Speech + clients). No-op on free responses (no receipt header).""" + tx_hash = _receipt_from_headers(self._last_raw_headers) + if tx_hash and isinstance(data, dict) and not data.get("txHash"): + data["txHash"] = tx_hash + + def _log_transaction( + self, + endpoint: str, + body: dict[str, Any], + response: Any, + cost_usd: float, + ) -> None: + """Async-Solana twin of :meth:`SolanaLLMClient._log_transaction`.""" + logger = self._tx_logger + if logger is None: + return + settlement = self._last_settlement + self._last_settlement = None + try: + logger.log( + endpoint=endpoint, + request=body, + response=response, + cost_usd=cost_usd, + model=(body.get("model") if isinstance(body, dict) else None), + wallet=self.get_wallet_address(), + network="solana-mainnet" if self.is_solana() else "solana-other", + client_kind=type(self).__name__, + settlement=settlement, + ) + except Exception: + pass + + # ------------------------------------------------------------------ + # Lifecycle + # ------------------------------------------------------------------ + + async def close(self) -> None: + await self._client.aclose() + + async def __aenter__(self) -> Self: + return self + + async def __aexit__(self, *_exc: object) -> None: + await self.close() + + # ------------------------------------------------------------------ + # Identity / state + # ------------------------------------------------------------------ + + @property + def payment_mode(self) -> str: + """Which rail this client pays on: ``"apikey"`` or ``"wallet"``. + + Worth checking once at startup when both a key and a wallet are + configured in the environment: it is the difference between + spending credit and spending USDC.""" + return payment_mode(self) + + def get_wallet_address(self) -> str: + # No address on the account rail: payment comes from prepaid + # credit, so there is nothing to return but the empty string. + if self.api_key: + return "" + if not self._address: + self._address = get_solana_public_key(self._private_key) + return self._address + + def is_solana(self) -> bool: + return "sol.blockrun.ai" in self._api_url + + def get_spending(self) -> dict[str, Any]: + return {"total_usd": self._session_total_usd, "calls": self._session_calls} + + def _billing_meta(self) -> dict[str, str | None]: + return { + "wallet": self.get_wallet_address(), + "network": "solana-mainnet" if self.is_solana() else "solana-other", + "client_kind": type(self).__name__, + } + + # ------------------------------------------------------------------ + # Non-streaming chat + # ------------------------------------------------------------------ + + async def _get_model_pricing(self) -> dict[str, dict[str, float]]: + """Model pricing for smart routing (cached for the client's lifetime).""" + if self._model_pricing_cache is not None: + return self._model_pricing_cache + pricing = build_model_pricing(await self.list_models()) + self._model_pricing_cache = pricing + return pricing + + async def route( + self, + prompt: str, + *, + system: str | None = None, + max_tokens: int | None = None, + routing_profile: RoutingProfile = "auto", + requires_structured_output: bool = False, + ) -> RoutingDecision: + """Inspect a Solana routing decision without making or paying for a call.""" + decision = route_with_catalog( + prompt, + system, + max_tokens or DEFAULT_MAX_TOKENS, + await self._get_model_pricing(), + routing_profile=routing_profile, + requires_structured_output=requires_structured_output, + minimum_payment_usd=SOLANA_MINIMUM_PAYMENT_USD, + ) + return RoutingDecision(**decision) + + async def smart_chat( + self, + prompt: str, + *, + system: str | None = None, + max_tokens: int | None = None, + temperature: float | None = None, + routing_profile: RoutingProfile = "auto", + timeout: float | None = None, + ) -> SmartChatResponse: + """Async smart chat with automatic model routing, paid on Solana.""" + decision = route_with_catalog( + prompt, + system, + max_tokens or DEFAULT_MAX_TOKENS, + await self._get_model_pricing(), + routing_profile=routing_profile, + minimum_payment_usd=SOLANA_MINIMUM_PAYMENT_USD, + ) + response = await self.chat( + decision["model"], + prompt, + system=system, + max_tokens=max_tokens or DEFAULT_MAX_TOKENS, + temperature=temperature, + timeout=timeout, + fallback_models=decision.get("fallbacks") or None, + ) + return SmartChatResponse( + response=response, + model=decision["model"], + routing=RoutingDecision(**decision), + ) + + async def smart_chat_completion( + self, + messages: list[dict[str, Any]], + *, + max_tokens: int | None = None, + temperature: float | None = None, + top_p: float | None = None, + search: bool = False, + search_parameters: dict[str, Any] | None = None, + tools: list[dict[str, Any]] | None = None, + tool_choice: Any | None = None, + timeout: float | None = None, + response_format: dict[str, Any] | None = None, + stop: str | list[str] | None = None, + fallback_models: list[str] | None = None, + routing_profile: RoutingProfile = "auto", + ) -> SmartChatCompletionResponse: + """Async smart routing for a full message list, paid on Solana.""" + view = routing_text(messages) + decision = route_with_catalog( + view["prompt"], + view["system_prompt"], + max_tokens or DEFAULT_MAX_TOKENS, + await self._get_model_pricing(), + routing_profile=routing_profile, + requires_structured_output=response_format is not None, + tools=tools, + tool_choice=tool_choice, + conversation_chars=view["conversation_chars"], + has_vision=view["has_vision"], + minimum_payment_usd=SOLANA_MINIMUM_PAYMENT_USD, + ) + response = await self.chat_completion( + decision["model"], + messages, + max_tokens=max_tokens or DEFAULT_MAX_TOKENS, + temperature=temperature, + top_p=top_p, + search=search, + search_parameters=search_parameters, + tools=tools, + tool_choice=tool_choice, + timeout=timeout, + response_format=response_format, + stop=stop, + fallback_models=fallback_models or decision.get("fallbacks") or None, + ) + return SmartChatCompletionResponse( + response=response, + model=decision["model"], + routing=RoutingDecision(**decision), + ) + + async def chat( + self, + model: str, + prompt: str, + system: str | None = None, + max_tokens: int = DEFAULT_MAX_TOKENS, + temperature: float | None = None, + search: bool = False, + timeout: float | None = None, + response_format: dict[str, Any] | None = None, + stop: str | list[str] | None = None, + fallback_models: list[str] | None = None, + ) -> str: + messages: list[dict[str, str]] = [] + if system: + messages.append({"role": "system", "content": system}) + messages.append({"role": "user", "content": prompt}) + result = await self.chat_completion( + model, + messages, + max_tokens=max_tokens, + temperature=temperature, + search=search, + timeout=timeout, + response_format=response_format, + stop=stop, + fallback_models=fallback_models, + ) + return result.choices[0].message.content or "" + + async def chat_completion( + self, + model: str, + messages: list[dict[str, Any]], + max_tokens: int = DEFAULT_MAX_TOKENS, + temperature: float | None = None, + top_p: float | None = None, + search: bool = False, + search_parameters: dict[str, Any] | None = None, + tools: list[dict[str, Any]] | None = None, + tool_choice: Any | None = None, + timeout: float | None = None, + response_format: dict[str, Any] | None = None, + stop: str | list[str] | None = None, + fallback_models: list[str] | None = None, + ) -> ChatResponse: + # `blockrun/auto` | `blockrun/eco` | `blockrun/premium` select a routing + # profile rather than a model. + virtual_profile = routing_profile_for_model(model) + if virtual_profile is not None: + return ( + await self.smart_chat_completion( + messages, + max_tokens=max_tokens, + temperature=temperature, + top_p=top_p, + search=search, + search_parameters=search_parameters, + tools=tools, + tool_choice=tool_choice, + timeout=timeout, + response_format=response_format, + stop=stop, + fallback_models=fallback_models, + routing_profile=virtual_profile, # type: ignore[arg-type] + ) + ).response + + validate_max_tokens(max_tokens) + body: dict[str, Any] = {"model": model, "messages": messages, "max_tokens": max_tokens} + if temperature is not None: + body["temperature"] = temperature + if top_p is not None: + body["top_p"] = top_p + if search_parameters: + body["search_parameters"] = search_parameters + elif search: + body["search_parameters"] = {"mode": "on"} + if tools is not None: + body["tools"] = tools + if tool_choice is not None: + body["tool_choice"] = tool_choice + if response_format is not None: + body["response_format"] = response_format + if stop is not None: + body["stop"] = stop + + # Same recovery walk as the sync client: transient upstream failures + # step to the next ranked model, a settled payment never retries. + attempts = [model, *(fallback_models or [])] + last_exc: Exception | None = None + for i, attempt_model in enumerate(attempts): + body["model"] = attempt_model + try: + return await self._request_with_payment( + "/v1/chat/completions", body, timeout=timeout + ) + except Exception as exc: + if not _should_fallback_solana(exc) or i + 1 >= len(attempts): + raise + last_exc = exc + sys.stderr.write( + f"[blockrun_llm] solana {attempt_model} -> {attempts[i + 1]} " + f"({type(exc).__name__}: {str(exc)[:80]})\n" + ) + assert last_exc is not None + raise last_exc + + async def list_models(self) -> list[dict[str, Any]]: + resp = await self._client.get(f"{self._api_url}/v1/models") + resp.raise_for_status() + return resp.json().get("data", []) + + # ------------------------------------------------------------------ + # Streaming chat + # ------------------------------------------------------------------ + + async def chat_completion_stream( + self, + model: str, + messages: list[dict[str, Any]], + *, + max_tokens: int = DEFAULT_MAX_TOKENS, + temperature: float | None = None, + top_p: float | None = None, + search: bool = False, + search_parameters: dict[str, Any] | None = None, + tools: list[dict[str, Any]] | None = None, + tool_choice: Any | None = None, + response_format: dict[str, Any] | None = None, + stop: str | list[str] | None = None, + fallback_models: list[str] | None = None, + timeout: float | None = None, + ) -> AsyncSolanaIterator: + """Async streaming. Same protocol semantics as the sync + :meth:`SolanaLLMClient.chat_completion_stream`; only the iteration + protocol differs (``async for``).""" + validate_max_tokens(max_tokens) + body: dict[str, Any] = { + "model": model, + "messages": messages, + "stream": True, + "max_tokens": max_tokens, + } + if temperature is not None: + body["temperature"] = temperature + if top_p is not None: + body["top_p"] = top_p + if search_parameters: + body["search_parameters"] = search_parameters + elif search: + body["search_parameters"] = {"mode": "on"} + if tools is not None: + body["tools"] = tools + if tool_choice is not None: + body["tool_choice"] = tool_choice + if response_format is not None: + body["response_format"] = response_format + if stop is not None: + body["stop"] = stop + + attempts = [model, *(fallback_models or [])] + last_exc: Exception | None = None + + for i, attempt_model in enumerate(attempts): + body["model"] = attempt_model + inner = self._stream_with_payment("/v1/chat/completions", body, timeout=timeout) + chunks_yielded = 0 + try: + async for chunk in inner: + chunks_yielded += 1 + yield chunk + return + except Exception as exc: + if chunks_yielded > 0: + raise + if not _should_fallback_solana(exc): + raise + last_exc = exc + if i + 1 < len(attempts): + next_model = attempts[i + 1] + sys.stderr.write( + f"[blockrun_llm] async solana stream {attempt_model} -> " + f"{next_model} ({type(exc).__name__}: {str(exc)[:80]})\n" + ) + assert last_exc is not None + raise last_exc + + async def _stream_with_payment( + self, + endpoint: str, + body: dict[str, Any], + timeout: float | None = None, + ): + """Whole-request payment-retry wrapper around :meth:`_stream_once` + (async). Re-runs the paid request on a PRE-BROADCAST payment rejection, + only before the first chunk is yielded; a settlement failure is + terminal. See _MAX_PAYMENT_RETRIES.""" + for payment_attempt in range(self._MAX_PAYMENT_RETRIES + 1): + yielded = 0 + try: + async for chunk in self._stream_once(endpoint, body, timeout=timeout): + yielded += 1 + yield chunk + return + except PaymentError as exc: + if ( + yielded > 0 + or not _is_safe_resign_error(exc) + or payment_attempt >= self._MAX_PAYMENT_RETRIES + ): + raise + await asyncio.sleep( + self._PAYMENT_RETRY_BACKOFFS[ + min(payment_attempt, len(self._PAYMENT_RETRY_BACKOFFS) - 1) + ] + ) + + async def _stream_once( + self, + endpoint: str, + body: dict[str, Any], + timeout: float | None = None, + ): + """Async version of :meth:`SolanaLLMClient._stream_once`.""" + url = f"{self._api_url}{endpoint}" + req_headers = {"Content-Type": "application/json", "User-Agent": _get_user_agent()} + eff_timeout = timeout if timeout is not None else self._timeout + backoffs = self._STREAM_5XX_BACKOFFS + + # ----- Phase 1: probe (no payment header) ----- + payment_headers: dict[str, str] | None = None + cost_usd = 0.0 + + for attempt in range(len(backoffs) + 1): + async with self._client.stream( + "POST", url, json=body, headers=req_headers, timeout=eff_timeout + ) as resp1: + if resp1.status_code == 200: + async for chunk in self._aiter_sse_chunks(resp1): + yield chunk + return + await resp1.aread() + if resp1.status_code == 402: + # Account rail: a 402 is the account being out of credit, not a + # challenge to sign. Nothing here can sign, so say so plainly. + raise_for_api_key_402(resp1, self.api_key) + payment_headers, cost_usd = await self._sign_payment_from_response(resp1) + break + if resp1.status_code in self._STREAM_5XX_STATUSES and attempt < len(backoffs): + import asyncio + + await asyncio.sleep(backoffs[attempt]) + continue + self._raise_stream_error(resp1, after_payment=False) + else: + raise APIError("solana stream probe exhausted retries", 0, None) + + # ----- Phase 2: stream with PAYMENT-SIGNATURE ----- + assert payment_headers is not None + try: + for attempt in range(len(backoffs) + 1): + async with self._client.stream( + "POST", url, json=body, headers=payment_headers, timeout=eff_timeout + ) as resp2: + if resp2.status_code == 200: + if cost_usd > 0: + self._session_calls += 1 + self._session_total_usd += cost_usd + self._last_call_cost = cost_usd + self._capture_settlement(resp2) + async for chunk in self._aiter_and_archive(resp2, body, cost_usd): + yield chunk + return + await resp2.aread() + if resp2.status_code == 402: + # Account rail: a 402 is the account being out of credit, not a + # challenge to sign. Nothing here can sign, so say so plainly. + raise_for_api_key_402(resp2, self.api_key) + raise build_payment_rejected_error(resp2) + if resp2.status_code in self._STREAM_5XX_STATUSES and attempt < len(backoffs): + import asyncio + + await asyncio.sleep(backoffs[attempt]) + continue + self._raise_stream_error(resp2, after_payment=True) + + except (httpx.HTTPError, APIError) as exc: + # Signed above; SPL USDC is gone. Do not let the fallback + # chain buy a retry on the next model. Re-raise bare so the + # traceback and __context__ survive. + _mark_settled(exc) + raise + + @staticmethod + async def _aiter_sse_chunks(response: httpx.Response): + async for raw_line in response.aiter_lines(): + if not raw_line or not raw_line.startswith("data: "): + continue + payload = raw_line[6:].strip() + if payload == "[DONE]": + return + try: + chunk_dict = _json.loads(payload) + except Exception: + continue + try: + yield ChatCompletionChunk(**chunk_dict) + except Exception: + yield ChatCompletionChunk.model_construct(**chunk_dict) + + async def _aiter_and_archive( + self, + response: httpx.Response, + body: dict[str, Any], + cost_usd: float, + ): + """Async version of :meth:`SolanaLLMClient._iter_and_archive`.""" + assembled_id: str | None = None + assembled_model: str | None = None + assembled_created: int = 0 + content_parts: list[str] = [] + finish_reason: str | None = None + usage_dict: dict[str, Any] | None = None + + async for chunk in self._aiter_sse_chunks(response): + if chunk.choices: + choice = chunk.choices[0] + content = stream_choice_content(choice) + if content: + content_parts.append(content) + fr = stream_choice_finish_reason(choice) + if fr: + finish_reason = fr + if assembled_id is None: + _id, _model, _created = chunk_meta(chunk) + if _id: + assembled_id = _id + assembled_model = _model + assembled_created = _created + _usage = chunk_usage_dict(chunk) + if _usage is not None: + usage_dict = _usage + # Race-free per-call x402 charge — see LLMClient._iter_and_archive. + chunk.cost_usd = cost_usd + yield chunk + + if cost_usd > 0: + from .cache import save_to_cache + + response_data: dict[str, Any] = { + "id": assembled_id or "stream", + "object": "chat.completion", + "created": assembled_created or int(__import__("time").time()), + "model": assembled_model or body.get("model"), + "choices": [ + { + "index": 0, + "message": { + "role": "assistant", + "content": "".join(content_parts), + }, + "finish_reason": finish_reason, + } + ], + "stream": True, + } + if usage_dict: + response_data["usage"] = usage_dict + try: + save_to_cache( + "/v1/chat/completions", + body, + response_data, + cost_usd=cost_usd, + **self._billing_meta(), + ) + except Exception: + pass + self._log_transaction("/v1/chat/completions", body, response_data, cost_usd) + + # ------------------------------------------------------------------ + # Payment + transport helpers + # ------------------------------------------------------------------ + + async def _sign_payment_from_response( + self, + response: httpx.Response, + ) -> tuple[dict[str, str], float]: + payment_header = SolanaLLMClient._extract_payment_header(response) + if not payment_header: + raise PaymentError("402 response but no payment requirements found") + payment_required = decode_payment_required_header(payment_header) + payment_payload = await self._sign_payment(payment_required) + # See the sync path: refusing here means nothing is ever sent. + _enforce_spend_limits(self, float(payment_payload.accepted.amount) / 1e6) + encoded_payment = encode_payment_signature_header(payment_payload) + cost_usd = float(payment_payload.accepted.amount) / 1e6 + return ( + { + "Content-Type": "application/json", + "User-Agent": _get_user_agent(), + "PAYMENT-SIGNATURE": encoded_payment, + }, + cost_usd, + ) + + # Reuse the sync class's pure helper — it doesn't touch async state. + _raise_stream_error = SolanaLLMClient._raise_stream_error + + async def _request_with_payment( + self, endpoint: str, body: dict[str, Any], timeout: float | None = None + ) -> ChatResponse: + """Whole-request payment-retry wrapper around :meth:`_request_once` + (async). Same policy as the sync path — a PRE-BROADCAST payment + rejection re-runs the entire request with a fresh signature; a + settlement failure is terminal. See _MAX_PAYMENT_RETRIES.""" + for payment_attempt in range(self._MAX_PAYMENT_RETRIES + 1): + try: + return await self._request_once(endpoint, body, timeout=timeout) + except PaymentError as exc: + if not _is_safe_resign_error(exc) or payment_attempt >= self._MAX_PAYMENT_RETRIES: + raise + await asyncio.sleep( + self._PAYMENT_RETRY_BACKOFFS[ + min(payment_attempt, len(self._PAYMENT_RETRY_BACKOFFS) - 1) + ] + ) + + raise PaymentError( # pragma: no cover - bounded loop always returns or raises + "Payment retry loop exhausted without a result." + ) + + async def _request_once( + self, endpoint: str, body: dict[str, Any], timeout: float | None = None + ) -> ChatResponse: + url = f"{self._api_url}{endpoint}" + headers = {"Content-Type": "application/json", "User-Agent": _get_user_agent()} + eff_timeout = timeout if timeout is not None else self._timeout + + response = await self._client.post(url, json=body, headers=headers, timeout=eff_timeout) + if response.status_code in (502, 503): + import asyncio + + await asyncio.sleep(1) + response = await self._client.post(url, json=body, headers=headers, timeout=eff_timeout) + + if response.status_code == 402: + # Account rail: a 402 is the account being out of credit, not a + # challenge to sign. Nothing here can sign, so say so plainly. + raise_for_api_key_402(response, self.api_key) + # Past this point the SPL USDC transfer has been signed. Tag + # anything that escapes so no fallback chain can buy a retry. + try: + return await self._handle_payment_and_retry( + url, body, response, timeout=eff_timeout + ) + except (httpx.HTTPError, APIError) as exc: + _mark_settled(exc) + raise + + if not response.is_success: + try: + error_body = response.json() + except Exception: + error_body = {"error": "Request failed"} + raise APIError( + f"API error: {response.status_code}", + response.status_code, + sanitize_error_response(error_body), + retry_after=retry_after_of(response), + ) + return ChatResponse(**response.json()) + + async def _handle_payment_and_retry( + self, + url: str, + body: dict[str, Any], + response: httpx.Response, + timeout: float | None = None, + ) -> ChatResponse: + eff_timeout = timeout if timeout is not None else self._timeout + payment_headers, cost_usd = await self._sign_payment_from_response(response) + + retry_response = await self._client.post( + url, json=body, headers=payment_headers, timeout=eff_timeout + ) + if retry_response.status_code in (502, 503): + import asyncio + + await asyncio.sleep(1) + retry_response = await self._client.post( + url, json=body, headers=payment_headers, timeout=eff_timeout + ) + + if retry_response.status_code == 402: + # Account rail: a 402 is the account being out of credit, not a + # challenge to sign. Nothing here can sign, so say so plainly. + raise_for_api_key_402(retry_response, self.api_key) + raise build_payment_rejected_error(retry_response) + if not retry_response.is_success: + try: + error_body = retry_response.json() + except Exception: + error_body = {"error": "Request failed"} + raise APIError( + f"{paid_request_error_prefix(retry_response.headers)}: {retry_response.status_code}", + retry_response.status_code, + sanitize_error_response(error_body), + retry_after=retry_after_of(retry_response), + ) + + self._session_calls += 1 + self._session_total_usd += cost_usd + self._last_call_cost = cost_usd + self._capture_settlement(retry_response) + + response_data = retry_response.json() + from .cache import save_to_cache + + save_to_cache( + "/v1/chat/completions", + body, + response_data, + cost_usd=cost_usd, + **self._billing_meta(), + ) + self._log_transaction("/v1/chat/completions", body, response_data, cost_usd) + return ChatResponse(**response_data) + + # ── Raw passthrough request helpers (async, Solana payment) ───────────── + + async def _request_with_payment_raw( + self, endpoint: str, body: dict[str, Any], timeout: float | None = None + ) -> dict[str, Any]: + """Bounded fresh-signature retry wrapper for raw POST endpoints.""" + for payment_attempt in range(self._MAX_PAYMENT_RETRIES + 1): + try: + return await self._request_with_payment_raw_once(endpoint, body, timeout=timeout) + except PaymentError as exc: + if not _is_safe_resign_error(exc) or payment_attempt >= self._MAX_PAYMENT_RETRIES: + raise + await asyncio.sleep( + self._PAYMENT_RETRY_BACKOFFS[ + min(payment_attempt, len(self._PAYMENT_RETRY_BACKOFFS) - 1) + ] + ) + + raise PaymentError( # pragma: no cover - bounded loop always returns or raises + "Payment retry loop exhausted without a result." + ) + + async def _request_with_payment_raw_once( + self, endpoint: str, body: dict[str, Any], timeout: float | None = None + ) -> dict[str, Any]: + """POST with Solana x402 payment, returning raw JSON (async mirror of + the sync :class:`SolanaLLMClient` helper).""" + from .cache import get_cached, save_to_cache + + cached = get_cached(endpoint, body) + if cached is not None: + return cached + + # Reset per-call receipt headers; only a paid retry repopulates them, so + # a free/cached model can't inherit a prior call's settlement receipt. + self._last_raw_headers = None + + url = f"{self._api_url}{endpoint}" + headers = {"Content-Type": "application/json", "User-Agent": _get_user_agent()} + eff_timeout = timeout if timeout is not None else self._timeout + + response = await self._client.post(url, json=body, headers=headers, timeout=eff_timeout) + if response.status_code in (502, 503): + await asyncio.sleep(1) + response = await self._client.post(url, json=body, headers=headers, timeout=eff_timeout) + + if response.status_code == 402: + # Account rail: a 402 is the account being out of credit, not a + # challenge to sign. Nothing here can sign, so say so plainly. + raise_for_api_key_402(response, self.api_key) + payment_headers, cost_usd = await self._sign_payment_from_response(response) + retry_response = await self._client.post( + url, json=body, headers=payment_headers, timeout=eff_timeout + ) + if retry_response.status_code in (502, 503): + await asyncio.sleep(1) + retry_response = await self._client.post( + url, json=body, headers=payment_headers, timeout=eff_timeout + ) + if retry_response.status_code == 402: + raise build_payment_rejected_error(retry_response) + if not retry_response.is_success: + try: + error_body = retry_response.json() + except Exception: + error_body = {"error": "Request failed"} + raise APIError( + f"{paid_request_error_prefix(retry_response.headers)}: {retry_response.status_code}", + retry_response.status_code, + sanitize_error_response(error_body), + retry_after=retry_after_of(retry_response), + ) + self._session_calls += 1 + self._session_total_usd += cost_usd + self._last_call_cost = cost_usd + self._capture_settlement(retry_response) + self._last_raw_headers = retry_response.headers + result = retry_response.json() + save_to_cache(endpoint, body, result, cost_usd=cost_usd, **self._billing_meta()) + self._log_transaction(endpoint, body, result, cost_usd) + return result + + if not response.is_success: + try: + error_body = response.json() + except Exception: + error_body = {"error": "Request failed"} + raise APIError( + f"API error: {response.status_code}", + response.status_code, + sanitize_error_response(error_body), + retry_after=retry_after_of(response), + ) + return response.json() + + async def _get_with_payment_raw( + self, + endpoint: str, + params: dict[str, Any] | None = None, + timeout: float | None = None, + ) -> dict[str, Any]: + """Bounded fresh-signature retry wrapper for raw GET endpoints.""" + for payment_attempt in range(self._MAX_PAYMENT_RETRIES + 1): + try: + return await self._get_with_payment_raw_once( + endpoint, params=params, timeout=timeout + ) + except PaymentError as exc: + if not _is_safe_resign_error(exc) or payment_attempt >= self._MAX_PAYMENT_RETRIES: + raise + await asyncio.sleep( + self._PAYMENT_RETRY_BACKOFFS[ + min(payment_attempt, len(self._PAYMENT_RETRY_BACKOFFS) - 1) + ] + ) + + raise PaymentError( # pragma: no cover - bounded loop always returns or raises + "Payment retry loop exhausted without a result." + ) + + async def _get_with_payment_raw_once( + self, + endpoint: str, + params: dict[str, Any] | None = None, + timeout: float | None = None, + ) -> dict[str, Any]: + """GET with Solana x402 payment, returning raw JSON (async).""" + from .cache import get_cached, save_to_cache + + cache_key_body = params or {} + cached = get_cached(endpoint, cache_key_body) + if cached is not None: + return cached + + url = f"{self._api_url}{endpoint}" + headers = {"User-Agent": _get_user_agent()} + eff_timeout = timeout if timeout is not None else self._timeout + + response = await self._client.get(url, params=params, headers=headers, timeout=eff_timeout) + if response.status_code in (502, 503): + await asyncio.sleep(1) + response = await self._client.get( + url, params=params, headers=headers, timeout=eff_timeout + ) + + if response.status_code == 402: + # Account rail: a 402 is the account being out of credit, not a + # challenge to sign. Nothing here can sign, so say so plainly. + raise_for_api_key_402(response, self.api_key) + payment_headers, cost_usd = await self._sign_payment_from_response(response) + retry_response = await self._client.get( + url, params=params, headers=payment_headers, timeout=eff_timeout + ) + if retry_response.status_code in (502, 503): + await asyncio.sleep(1) + retry_response = await self._client.get( + url, params=params, headers=payment_headers, timeout=eff_timeout + ) + if retry_response.status_code == 402: + raise build_payment_rejected_error(retry_response) + if not retry_response.is_success: + try: + error_body = retry_response.json() + except Exception: + error_body = {"error": "Request failed"} + raise APIError( + f"{paid_request_error_prefix(retry_response.headers)}: {retry_response.status_code}", + retry_response.status_code, + sanitize_error_response(error_body), + retry_after=retry_after_of(retry_response), + ) + self._session_calls += 1 + self._session_total_usd += cost_usd + self._last_call_cost = cost_usd + self._capture_settlement(retry_response) + result = retry_response.json() + save_to_cache( + endpoint, cache_key_body, result, cost_usd=cost_usd, **self._billing_meta() + ) + self._log_transaction(endpoint, cache_key_body, result, cost_usd) + return result + + if not response.is_success: + try: + error_body = response.json() + except Exception: + error_body = {"error": "Request failed"} + raise APIError( + f"API error: {response.status_code}", + response.status_code, + sanitize_error_response(error_body), + retry_after=retry_after_of(response), + ) + return response.json() + + # ── Standalone search (Grok Live Search) ──────────────────────────────── + + async def search( + self, + query: str, + *, + sources: list[str] | None = None, + max_results: int = 10, + from_date: str | None = None, + to_date: str | None = None, + timeout: float | None = None, + ) -> SearchResult: + """Standalone search (Solana payment). + + ``timeout`` overrides the per-call HTTP timeout (defaults to + ``DEFAULT_SEARCH_TIMEOUT`` — deep web/X tool-use can run minutes). + """ + body: dict[str, Any] = { + "query": query, + "max_results": max_results, + } + if sources is not None: + body["sources"] = sources + if from_date is not None: + body["from_date"] = from_date + if to_date is not None: + body["to_date"] = to_date + + eff_timeout = timeout if timeout is not None else self._search_timeout + data = await self._request_with_payment_raw("/v1/search", body, timeout=eff_timeout) + return SearchResult(**data) + + # ── Balance ───────────────────────────────────────────────────────────── + + async def get_balance(self) -> float: + # Returning 0 would be the worst available answer: it is + # indistinguishable from an empty wallet, and an agent gating on it + # would stop calling a well-funded account. + if self.api_key: + raise wallet_only("get_balance") + """Get USDC balance on Solana (async; matches the sync client API). + + The underlying RPC read is synchronous, so it runs in a worker thread + to avoid blocking the event loop. + """ + from .solana_wallet import get_solana_usdc_balance + + return await asyncio.to_thread( + get_solana_usdc_balance, self.get_wallet_address(), rpc_url=self._rpc_url + ) + + # ── Image generation + editing ────────────────────────────────────────── + + async def image( + self, + prompt: str, + *, + model: str = "google/nano-banana", + size: str = "1024x1024", + n: int = 1, + quality: str | None = None, + timeout: float | None = None, + ) -> ImageResponse: + """Generate an image from a text prompt (Solana payment). + + Slow models (gpt-image-2, dall-e-3, nano-banana-pro 4K) trigger the + gateway's async 202 + poll flow; this polls transparently until + completion and only settles on the final completed poll. If the poll + budget is exhausted an :class:`APIError` 504 is raised and **no payment + is taken**. + + Args: + quality: ``low`` / ``medium`` / ``high`` / ``auto``, + ``openai/gpt-image-*`` only. See :meth:`SolanaLLMClient.image`. + + Raises: + ValueError: If ``quality`` is not one of the four accepted values. + """ + body: dict[str, Any] = { + "model": model, + "prompt": prompt, + "size": size, + "n": n, + } + validate_image_quality(quality) + if quality is not None: + body["quality"] = quality + data = await self._request_image_with_payment( + "/v1/images/generations", body, timeout=timeout + ) + return ImageResponse(**data) + + async def image_edit( + self, + prompt: str, + image: str | list[str], + *, + model: str = "openai/gpt-image-2", + mask: str | None = None, + size: str = "1024x1024", + n: int = 1, + quality: str | None = None, + timeout: float | None = None, + ) -> ImageResponse: + """Edit an image using img2img (Solana payment). ``image`` may be a + single data URI or a list of 1-4 data URIs for multi-image fusion + (openai/* up to 4, google/* up to 3). Handles the async 202 + poll + slow path transparently — settlement only happens on completion. + + Args: + quality: ``low`` / ``medium`` / ``high`` / ``auto``, + ``openai/gpt-image-*`` only. See :meth:`SolanaLLMClient.image`. + + Raises: + ValueError: If ``quality`` is not one of the four accepted values. + """ + body: dict[str, Any] = { + "model": model, + "prompt": prompt, + "image": image, + "size": size, + "n": n, + } + if mask is not None: + body["mask"] = mask + validate_image_quality(quality) + if quality is not None: + body["quality"] = quality + + data = await self._request_image_with_payment( + "/v1/images/image2image", body, timeout=timeout + ) + return ImageResponse(**data) + + def _absolute_url(self, url: str) -> str: + """Resolve a server-supplied relative ``poll_url`` against the API host + (``api_url`` already includes the trailing ``/api`` — strip it once).""" + if self.api_key: + # api.blockrun.ai serves these routes at /v1/... and answers + # /api/v1/... with wrong_host, so the gateway-minted prefix has to + # come off. Shared with the Base clients, which also pins the + # Authorization header to the gateway's own origin. + return resolve_poll_url(url, self._api_url, self.api_key) + base = self._api_url.removesuffix("/api") + if url.startswith(("http://", "https://")): + # The poll loop sends (and re-signs) the wallet's PAYMENT-SIGNATURE + # against this URL, so an absolute poll_url is pinned to the API + # host+scheme — a gateway response pointing it elsewhere would leak + # the signed payment off-host. + poll, api = httpx.URL(url), httpx.URL(base) + if (poll.scheme, poll.host) != (api.scheme, api.host): + raise APIError( + "Refusing an absolute poll_url on a different host/scheme than " + f"the API ({poll.scheme}://{poll.host} != {api.scheme}://{api.host}); " + "the signed payment header must not be sent off-host.", + 502, + {"poll_url": url}, + ) + return url + return f"{base}{url}" + + # ── Video / music / speech / enrollment / market data (async) ────────── + + async def video( + self, + prompt: str, + *, + model: str | None = None, + image_url: str | None = None, + last_frame_url: str | None = None, + reference_image_urls: list[str] | None = None, + reference_videos: list[dict[str, str]] | None = None, + reference_audios: list[dict[str, str]] | None = None, + bitrate_mode: str | None = None, + output_format: str | None = None, + camera_fixed: bool | None = None, + safety_identifier: str | None = None, + real_face_asset_id: str | None = None, + duration_seconds: int | None = None, + aspect_ratio: str | None = None, + resolution: str | None = None, + generate_audio: bool | None = None, + seed: int | None = None, + watermark: bool | None = None, + return_last_frame: bool | None = None, + input_type: str | None = None, + budget_seconds: float | None = None, + timeout: float | None = None, + ) -> VideoResponse: + """Generate a video clip (Solana payment). Async mirror of + :meth:`SolanaLLMClient.video`.""" + body = SolanaLLMClient._build_video_body( + prompt, + model=model, + image_url=image_url, + last_frame_url=last_frame_url, + reference_image_urls=reference_image_urls, + reference_videos=reference_videos, + reference_audios=reference_audios, + bitrate_mode=bitrate_mode, + output_format=output_format, + camera_fixed=camera_fixed, + safety_identifier=safety_identifier, + real_face_asset_id=real_face_asset_id, + duration_seconds=duration_seconds, + aspect_ratio=aspect_ratio, + resolution=resolution, + generate_audio=generate_audio, + seed=seed, + watermark=watermark, + return_last_frame=return_last_frame, + input_type=input_type, + ) + + data = await self._request_image_with_payment( + "/v1/videos/generations", + body, + timeout=timeout, + poll_budget_seconds=( + budget_seconds + if budget_seconds is not None + else SolanaLLMClient.VIDEO_POLL_BUDGET_SECONDS + ), + poll_interval_seconds=SolanaLLMClient.VIDEO_POLL_INTERVAL_SECONDS, + max_resigns=SolanaLLMClient.MEDIA_POLL_MAX_RESIGNS, + label="Video generation", + ) + return VideoResponse(**data) + + async def video_from_content( + self, + content: list[dict[str, Any]], + *, + model: str | None = None, + budget_seconds: float | None = None, + timeout: float | None = None, + **options: Any, + ) -> VideoResponse: + """Generate a video from a Seedance ``content[]`` body (Solana payment).""" + if not content: + raise ValueError("content must be a non-empty list of Seedance content items.") + body: dict[str, Any] = {"content": content, **options} + if model is not None: + body["model"] = model + data = await self._request_image_with_payment( + "/v1/videos", + body, + timeout=timeout, + poll_budget_seconds=( + budget_seconds + if budget_seconds is not None + else SolanaLLMClient.VIDEO_POLL_BUDGET_SECONDS + ), + poll_interval_seconds=SolanaLLMClient.VIDEO_POLL_INTERVAL_SECONDS, + max_resigns=SolanaLLMClient.MEDIA_POLL_MAX_RESIGNS, + label="Video generation", + ) + return VideoResponse(**data) + + async def music( + self, + prompt: str, + *, + model: str | None = None, + instrumental: bool = True, + lyrics: str | None = None, + timeout: float | None = None, + ) -> MusicResponse: + """Generate a music track (Solana payment).""" + if instrumental and lyrics and lyrics.strip(): + raise ValueError("Cannot specify lyrics when instrumental is True") + body: dict[str, Any] = { + "model": model or SolanaLLMClient.MUSIC_DEFAULT_MODEL, + "prompt": prompt, + "instrumental": instrumental, + } + if lyrics and lyrics.strip(): + body["lyrics"] = lyrics.strip() + data = await self._request_with_payment_raw("/v1/audio/generations", body, timeout=timeout) + self._attach_receipt(data) + return MusicResponse(**data) + + async def speech( + self, + input: str, + *, + model: str | None = None, + voice: str | None = None, + response_format: str | None = None, + speed: float | None = None, + timeout: float | None = None, + ) -> SpeechResponse: + """Synthesize speech from text (Solana payment).""" + body: dict[str, Any] = { + "model": model or SolanaLLMClient.SPEECH_DEFAULT_MODEL, + "input": input, + } + if voice: + body["voice"] = voice + if response_format: + body["response_format"] = response_format + if speed is not None: + body["speed"] = speed + data = await self._request_with_payment_raw("/v1/audio/speech", body, timeout=timeout) + self._attach_receipt(data) + return SpeechResponse(**data) + + async def sound_effect( + self, + text: str, + *, + model: str | None = None, + duration_seconds: float | None = None, + prompt_influence: float | None = None, + response_format: str | None = None, + timeout: float | None = None, + ) -> SpeechResponse: + """Generate a cinematic sound effect (Solana payment).""" + body: dict[str, Any] = { + "model": model or SolanaLLMClient.SOUNDFX_DEFAULT_MODEL, + "text": text, + } + if duration_seconds is not None: + body["duration_seconds"] = duration_seconds + if prompt_influence is not None: + body["prompt_influence"] = prompt_influence + if response_format: + body["response_format"] = response_format + data = await self._request_with_payment_raw( + "/v1/audio/sound-effects", body, timeout=timeout + ) + self._attach_receipt(data) + return SpeechResponse(**data) + + async def list_voices(self) -> list[dict[str, Any]]: + """List available speech voices (free).""" + url = f"{self._api_url}/v1/audio/voices" + resp = await self._client.get( + url, headers={"User-Agent": _get_user_agent()}, timeout=DEFAULT_FAST_TIMEOUT + ) + if resp.status_code != 200: + try: + error_body = resp.json() + except Exception: + error_body = {"error": "Request failed"} + raise APIError( + f"List voices failed: HTTP {resp.status_code}", + resp.status_code, + sanitize_error_response(error_body), + retry_after=retry_after_of(resp), + ) + data = resp.json() + # Gateway wraps the voice list under "data" (mirrors SpeechClient.list_voices). + return data.get("data", []) if isinstance(data, dict) else data + + async def portrait_enroll(self, name: str, image_url: str) -> PortraitEnrollment: + """Enroll a Virtual Portrait ($0.01 USDC). Returns a ``ta_`` asset id.""" + if not name or not name.strip(): + raise ValueError("name is required (1-64 chars)") + if len(name) > 64: + raise ValueError(f"name must be 64 chars or fewer (got {len(name)})") + if not image_url or not image_url.lower().startswith(("https://", "http://")): + raise ValueError("image_url must be an http(s) URL") + data = await self._request_with_payment_raw( + "/v1/portrait/enroll", {"name": name, "image_url": image_url} + ) + return PortraitEnrollment(**data) + + async def list_portraits(self, wallet_address: str | None = None) -> PortraitList: + """List Virtual Portraits enrolled by a wallet (free, rate-limited).""" + addr = _safe_path_segment(wallet_address or self.get_wallet_address(), "wallet_address") + url = f"{self._api_url}/v1/wallet/{addr}/portraits" + resp = await self._client.get( + url, headers={"User-Agent": _get_user_agent()}, timeout=DEFAULT_FAST_TIMEOUT + ) + if resp.status_code != 200: + try: + error_body = resp.json() + except Exception: + error_body = {"error": "Request failed"} + raise APIError( + "Portrait listing failed", + resp.status_code, + sanitize_error_response(error_body), + retry_after=retry_after_of(resp), + ) + return PortraitList(**resp.json()) + + async def realface_init(self, name: str, group_id: str | None = None) -> RealFaceInit: + """Start/refresh a RealFace enrollment (free, rate-limited).""" + if not name or not name.strip(): + raise ValueError("name is required (1-64 chars)") + if len(name) > 64: + raise ValueError(f"name must be 64 chars or fewer (got {len(name)})") + if group_id is not None and not _GROUP_ID_RE.match(group_id): + raise ValueError("group_id must look like 'legacy_rf_'") + body: dict[str, Any] = {"name": name} + if group_id: + body["groupId"] = group_id + url = f"{self._api_url}/v1/realface/init" + resp = await self._client.post( + url, + json=body, + headers={"Content-Type": "application/json", "User-Agent": _get_user_agent()}, + timeout=DEFAULT_FAST_TIMEOUT, + ) + if resp.status_code != 200: + try: + error_body = resp.json() + except Exception: + error_body = {"error": "Request failed"} + raise APIError( + "RealFace init failed", + resp.status_code, + sanitize_error_response(error_body), + retry_after=retry_after_of(resp), + ) + return RealFaceInit(**resp.json()) + + async def realface_status(self, group_id: str) -> RealFaceStatus: + """Poll a RealFace group's state (free, rate-limited).""" + if not group_id or not _GROUP_ID_RE.match(group_id): + raise ValueError("group_id must look like 'legacy_rf_'") + url = f"{self._api_url}/v1/realface/status" + resp = await self._client.get( + url, + params={"groupId": group_id}, + headers={"User-Agent": _get_user_agent()}, + timeout=DEFAULT_FAST_TIMEOUT, + ) + if resp.status_code != 200: + try: + error_body = resp.json() + except Exception: + error_body = {"error": "Request failed"} + raise APIError( + "RealFace status check failed", + resp.status_code, + sanitize_error_response(error_body), + retry_after=retry_after_of(resp), + ) + return RealFaceStatus(**resp.json()) + + async def realface_wait_for_active( + self, + group_id: str, + timeout_seconds: float = 180.0, + poll_interval_seconds: float = 4.0, + ) -> RealFaceStatus: + """Block until the RealFace group is active (person finished the phone + liveness check).""" + import time as _time + + if poll_interval_seconds <= 0: + raise ValueError("poll_interval_seconds must be positive") + deadline = _time.monotonic() + timeout_seconds + while True: + state = await self.realface_status(group_id) + if state.ready_to_finalize: + return state + if _time.monotonic() + poll_interval_seconds >= deadline: + raise TimeoutError( + f"RealFace group {group_id} not active after {timeout_seconds:.0f}s " + f"(last status: {state.status!r})." + ) + await asyncio.sleep(poll_interval_seconds) + + async def realface_enroll(self, name: str, image_url: str, group_id: str) -> RealFaceEnrollment: + """Finalize a RealFace enrollment ($0.01 USDC).""" + if not name or not name.strip(): + raise ValueError("name is required (1-64 chars)") + if len(name) > 64: + raise ValueError(f"name must be 64 chars or fewer (got {len(name)})") + if not image_url or not image_url.lower().startswith(("https://", "http://")): + raise ValueError("image_url must be an http(s) URL") + if not group_id or not _GROUP_ID_RE.match(group_id): + raise ValueError("group_id must look like 'legacy_rf_'") + data = await self._request_with_payment_raw( + "/v1/realface/enroll", {"name": name, "image_url": image_url, "group_id": group_id} + ) + return RealFaceEnrollment(**data) + + async def list_realfaces(self, wallet_address: str | None = None) -> RealFaceList: + """List RealFace assets enrolled by a wallet (free, rate-limited).""" + addr = _safe_path_segment(wallet_address or self.get_wallet_address(), "wallet_address") + url = f"{self._api_url}/v1/wallet/{addr}/realfaces" + resp = await self._client.get( + url, headers={"User-Agent": _get_user_agent()}, timeout=DEFAULT_FAST_TIMEOUT + ) + if resp.status_code != 200: + try: + error_body = resp.json() + except Exception: + error_body = {"error": "Request failed"} + raise APIError( + "RealFace listing failed", + resp.status_code, + sanitize_error_response(error_body), + retry_after=retry_after_of(resp), + ) + return RealFaceList(**resp.json()) + + async def price( + self, + category: Category, + symbol: str, + *, + market: Market | None = None, + session: Session | None = None, + ) -> PricePoint: + """Fetch a realtime Pyth price quote (Solana payment for paid categories).""" + endpoint = SolanaLLMClient._price_category_path(category, market, "price", symbol) + params: dict[str, Any] = {} + if session is not None: + params["session"] = session + data = await self._get_with_payment_raw( + endpoint, params=params or None, timeout=DEFAULT_FAST_TIMEOUT + ) + return PricePoint( + symbol=data.get("symbol", symbol.upper()), + price=data.get("price"), + publish_time=data.get("publishTime"), + confidence=data.get("confidence"), + feed_id=data.get("feedId"), + **{ + k: v + for k, v in data.items() + if k not in {"symbol", "price", "publishTime", "confidence", "feedId"} + }, + ) + + async def price_history( + self, + category: Category, + symbol: str, + *, + resolution: Resolution = "D", + from_ts: int, + to_ts: int, + market: Market | None = None, + session: Session | None = None, + ) -> PriceHistoryResponse: + """Fetch OHLC bars between two Unix timestamps (seconds).""" + endpoint = SolanaLLMClient._price_category_path(category, market, "history", symbol) + params: dict[str, Any] = {"resolution": resolution, "from": from_ts, "to": to_ts} + if session is not None: + params["session"] = session + data = await self._get_with_payment_raw( + endpoint, params=params, timeout=DEFAULT_FAST_TIMEOUT + ) + return PriceHistoryResponse( + symbol=data.get("symbol", symbol.upper()), + resolution=data.get("resolution", resolution), + bars=data.get("bars", []), + **{k: v for k, v in data.items() if k not in {"symbol", "resolution", "bars"}}, + ) + + async def list_symbols( + self, + category: Category, + *, + q: str | None = None, + limit: int = 100, + market: Market | None = None, + ) -> SymbolListResponse: + """List available symbols in a Pyth category (free discovery).""" + endpoint = SolanaLLMClient._price_category_path(category, market, "list", None) + params: dict[str, Any] = {"limit": limit} + if q: + params["q"] = q + data = await self._get_with_payment_raw( + endpoint, params=params, timeout=DEFAULT_FAST_TIMEOUT + ) + if isinstance(data, list): + return SymbolListResponse(symbols=data, count=len(data)) + return SymbolListResponse( + symbols=data.get("symbols", data.get("feeds", [])), + count=data.get("count"), + **{k: v for k, v in data.items() if k not in {"symbols", "feeds", "count"}}, + ) + + async def rpc( + self, + network: str, + method: str, + params: list[Any] | None = None, + *, + id: str | int = 1, + ) -> RpcResponse: + """Make a single JSON-RPC 2.0 call (Solana payment, flat $0.002).""" + _safe_path_segment(network, "network") + body: dict[str, Any] = {"jsonrpc": "2.0", "id": id, "method": method} + if params is not None: + body["params"] = params + data = await self._request_with_payment_raw(f"/v1/rpc/{network}", body) + return SolanaLLMClient._rpc_response(data, self._last_raw_headers, network) + + async def rpc_batch(self, network: str, requests: list[dict[str, Any]]) -> list[RpcResponse]: + """Make a JSON-RPC 2.0 batch call (Solana payment, $0.002 x N).""" + if not requests: + raise ValueError("batch requires at least one request") + _safe_path_segment(network, "network") + body: list[dict[str, Any]] = [] + for i, req in enumerate(requests): + if "method" not in req: + raise ValueError(f"batch request {i} is missing 'method'") + body.append({"jsonrpc": "2.0", "id": i + 1, **req}) + data = await self._request_with_payment_raw(f"/v1/rpc/{network}", body) # type: ignore[arg-type] + headers = self._last_raw_headers + if not isinstance(data, list): + data = [data] + return [SolanaLLMClient._rpc_response(item, headers, network) for item in data] + + async def _request_image_with_payment( + self, + endpoint: str, + body: dict[str, Any], + timeout: float | None = None, + *, + poll_budget_seconds: float | None = None, + poll_interval_seconds: float | None = None, + max_resigns: int = 0, + label: str = "Image", + ) -> dict[str, Any]: + """Async sign + submit + poll wrapper for async media generation — the + async mirror of the sync :class:`SolanaLLMClient` helper. Shared by + :meth:`image` and :meth:`video` (``max_resigns`` re-signs to survive + the 600s x402 authorization window on long video polls). + """ + import time as _time + + from .cache import get_cached, save_to_cache + + cached = get_cached(endpoint, body) + if cached is not None: + return cached + + url = f"{self._api_url}{endpoint}" + probe_headers = {"Content-Type": "application/json", "User-Agent": _get_user_agent()} + eff_timeout = timeout if timeout is not None else self._image_timeout + + # Step 1: probe — expect 402 unless the model is free or cached upstream. + probe = await self._client.post(url, json=body, headers=probe_headers, timeout=eff_timeout) + if probe.status_code in (502, 503): + await asyncio.sleep(1) + probe = await self._client.post( + url, json=body, headers=probe_headers, timeout=eff_timeout + ) + + # Account rail: a 402 here is "out of credit", not a challenge to sign. + # Checked before the x402 branch below, which has no signer to reach for + # and, without the optional SDK installed, no decoder either. + raise_for_api_key_402(probe, self.api_key) + + if probe.status_code != 402: + if not probe.is_success: + try: + error_body = probe.json() + except Exception: + error_body = {"error": "Request failed"} + raise APIError( + f"Image request: HTTP {probe.status_code}", + probe.status_code, + sanitize_error_response(error_body), + retry_after=retry_after_of(probe), + ) + return probe.json() + + # Step 2: sign x402 SVM payload (reuse the encoded signature on polls). + # Inlined rather than _sign_payment_from_response so the original payment + # terms are captured for the mid-poll re-sign guard below. + probe_payment_header = SolanaLLMClient._extract_payment_header(probe) + if not probe_payment_header: + raise PaymentError("402 response but no payment requirements found") + payment_required = decode_payment_required_header(probe_payment_header) + payment_payload_obj = await self._sign_payment(payment_required) + encoded_payment = encode_payment_signature_header(payment_payload_obj) + cost_usd = float(payment_payload_obj.accepted.amount) / 1e6 + # Terms this job is authorized to pay — any mid-poll re-sign must match. + orig_amount = payment_payload_obj.accepted.amount + orig_pay_to = payment_payload_obj.accepted.pay_to + payment_headers = { + "Content-Type": "application/json", + "User-Agent": _get_user_agent(), + "PAYMENT-SIGNATURE": encoded_payment, + } + + # Step 3: submit with signature. + submit_resp = await self._client.post( + url, json=body, headers=payment_headers, timeout=eff_timeout + ) + if submit_resp.status_code in (502, 503): + await asyncio.sleep(1) + submit_resp = await self._client.post( + url, json=body, headers=payment_headers, timeout=eff_timeout + ) + + if submit_resp.status_code == 402: + # Account rail: a 402 is the account being out of credit, not a + # challenge to sign. Nothing here can sign, so say so plainly. + raise_for_api_key_402(submit_resp, self.api_key) + raise build_payment_rejected_error(submit_resp) + + if submit_resp.status_code == 200: + # Fast path — image produced inline. + self._session_calls += 1 + self._session_total_usd += cost_usd + self._last_call_cost = cost_usd + self._capture_settlement(submit_resp) + data = submit_resp.json() + save_to_cache(endpoint, body, data, cost_usd=cost_usd, **self._billing_meta()) + self._log_transaction(endpoint, body, data, cost_usd) + return data + + if submit_resp.status_code != 202: + try: + error_body = submit_resp.json() + except Exception: + error_body = {"error": "Request failed"} + raise APIError( + f"Image request failed: {paid_request_error_prefix(submit_resp.headers)}: HTTP {submit_resp.status_code}", + submit_resp.status_code, + sanitize_error_response(error_body), + retry_after=retry_after_of(submit_resp), + ) + + # Step 4: slow path — poll until completed (or budget exhausted). + try: + submit_data = submit_resp.json() + except Exception: + submit_data = {} + + poll_url_rel = submit_data.get("poll_url") + job_id = submit_data.get("id") + if not poll_url_rel: + raise APIError("Slow-path 202 missing poll_url", 202, {"response": submit_data}) + poll_url = self._absolute_url(poll_url_rel) + poll_headers = { + "User-Agent": _get_user_agent(), + "PAYMENT-SIGNATURE": encoded_payment, + } + + budget = ( + poll_budget_seconds + if poll_budget_seconds is not None + else SolanaLLMClient.IMAGE_POLL_BUDGET_SECONDS + ) + interval = ( + poll_interval_seconds + if poll_interval_seconds is not None + else SolanaLLMClient.IMAGE_POLL_INTERVAL_SECONDS + ) + deadline = _time.monotonic() + budget + last_status = submit_data.get("status", "queued") + resigns_left = max_resigns + last_resign_at = _time.monotonic() + + while _time.monotonic() < deadline: + await asyncio.sleep(interval) + + # Keep the settlement blockhash fresh (poll-based media path only, + # gated on max_resigns) — mirror of the sync helper. Re-sign the + # ORIGINAL challenge (same amount/ + # pay_to, fresh blockhash) every MEDIA_RESIGN_FRESH_SECONDS so a slow / + # flaky-status model (1080p Seedance) can't age the signature out + # before the settling "completed" poll lands. Only completed settles. + if ( + max_resigns > 0 + and _time.monotonic() - last_resign_at >= SolanaLLMClient.MEDIA_RESIGN_FRESH_SECONDS + ): + try: + fresh_payload = await self._sign_payment(payment_required) + poll_headers["PAYMENT-SIGNATURE"] = encode_payment_signature_header( + fresh_payload + ) + last_resign_at = _time.monotonic() + except Exception: + pass + + poll_resp = await self._client.get(poll_url, headers=poll_headers, timeout=eff_timeout) + try: + poll_data = poll_resp.json() + except Exception: + poll_data = {} + last_status = poll_data.get("status", last_status) + + if poll_resp.status_code == 402: + # Account rail: a 402 is the account being out of credit, not a + # challenge to sign. Nothing here can sign, so say so plainly. + raise_for_api_key_402(poll_resp, self.api_key) + # Mid-poll 402 = settlement failed, almost always a stale + # blockhash (the payment was signed at submit time but only + # settles when the job completes; by then the signed tx's + # recent-blockhash can be expired -> transaction_simulation_failed). + # The failing poll carries NO fresh challenge, so re-GET poll_url + # WITHOUT the stale signature to solicit a fresh 402 (new + # blockhash), re-sign, and keep polling. Mirrors the sync helper / + # Base VideoClient. + if resigns_left > 0: + resigns_left -= 1 + resign_payload = None + try: + challenge = await self._client.get( + poll_url, + headers={"User-Agent": _get_user_agent()}, + timeout=eff_timeout, + ) + resign_header = SolanaLLMClient._extract_payment_header(challenge) + if challenge.status_code == 402 and resign_header: + resign_required = decode_payment_required_header(resign_header) + resign_payload = await self._sign_payment(resign_required) + except (PaymentError, httpx.HTTPError): + # Challenge GET or re-sign failed — surface the gateway's + # real 402 reason, not a network/signing error. + resign_payload = None + if resign_payload is not None: + # Refuse a re-challenge that reprices or redirects the + # payment vs. what this job originally authorized. This + # PaymentError must propagate (NOT fall through to the + # generic 402); the guard also pins the amount, so the + # submit-time cost_usd stays correct for the ledger. + _assert_same_payment_terms(resign_payload, orig_amount, orig_pay_to) + poll_headers["PAYMENT-SIGNATURE"] = encode_payment_signature_header( + resign_payload + ) + continue + raise build_payment_rejected_error(poll_resp) + + if last_status == "failed": + raise APIError( + f"{label} failed upstream: {poll_data.get('error', 'unknown')}", + poll_resp.status_code, + sanitize_error_response(poll_data if isinstance(poll_data, dict) else {}), + retry_after=retry_after_of(poll_resp), + ) + + # Terminal success is keyed on status, NOT the HTTP code (see the + # sync helper) — a completed-but-non-200 poll is still success. + if last_status == "completed": + tx_hash = poll_resp.headers.get("x-payment-receipt") or poll_resp.headers.get( + "X-Payment-Receipt" + ) + if tx_hash and isinstance(poll_data, dict) and not poll_data.get("txHash"): + poll_data["txHash"] = tx_hash + self._session_calls += 1 + self._session_total_usd += cost_usd + self._last_call_cost = cost_usd + self._capture_settlement(poll_resp) + save_to_cache(endpoint, body, poll_data, cost_usd=cost_usd, **self._billing_meta()) + self._log_transaction(endpoint, body, poll_data, cost_usd) + return poll_data + + if poll_resp.status_code in (202, 504): + continue + + if poll_resp.status_code != 200: + try: + error_body = poll_resp.json() + except Exception: + error_body = {"error": "Request failed"} + raise APIError( + f"{label} poll failed: HTTP {poll_resp.status_code}", + poll_resp.status_code, + sanitize_error_response(error_body), + retry_after=retry_after_of(poll_resp), + ) + + raise APIError( + ( + f"{label} did not complete within {budget:.0f}s " + f"(last status: {last_status}). Settlement only happens on " + "completion, so no payment was taken. The job stays claimable " + "for ~48h — re-poll poll_url with a fresh signature from the " + "same wallet to fetch (and settle) the finished result." + ), + 504, + {"id": job_id, "last_status": last_status, "poll_url": poll_url}, + ) + + # ── Prediction Markets (Powered by Predexon) ──────────────────────────── + + async def pm(self, path: str, **params: Any) -> dict[str, Any]: + """Query Predexon prediction market data (GET, Solana payment). Powered by Predexon.""" + return await self._get_with_payment_raw(f"/v1/pm/{path}", params or None) + + async def pm_query(self, path: str, query: dict[str, Any]) -> dict[str, Any]: + """Structured query for Predexon data (POST, Solana payment). Powered by Predexon.""" + return await self._request_with_payment_raw(f"/v1/pm/{path}", query) + + async def pm_markets(self, **params: Any) -> dict[str, Any]: + """RETIRED — ``/v1/pm/markets`` no longer exists. + + Predexon sunset market matching on 2026-07-20 and the whole + canonical layer went with it, so this path returns 410 upstream. + Use ``pm("markets/search", q=...)`` for cross-venue lookups. + + Kept as a raising stub rather than deleted so upgrading does not + break imports or attribute access; it raises before any network + I/O, so you never pay a round trip to learn it is gone. + + :raises RetiredEndpointError: always. + """ + raise RetiredEndpointError( + "/v1/pm/markets was sunset by Predexon on 2026-07-20 (upstream 410). Use pm('markets/search', q=...) for cross-venue lookups." + ) + + async def pm_listings(self, **params: Any) -> dict[str, Any]: + """RETIRED — ``/v1/pm/markets/listings`` no longer exists. + + Predexon sunset market matching on 2026-07-20 and the whole + canonical layer went with it, so this path returns 410 upstream. + Use ``pm("markets/search", q=...)`` for cross-venue lookups. + + Kept as a raising stub rather than deleted so upgrading does not + break imports or attribute access; it raises before any network + I/O, so you never pay a round trip to learn it is gone. + + :raises RetiredEndpointError: always. + """ + raise RetiredEndpointError( + "/v1/pm/markets/listings was sunset by Predexon on 2026-07-20 (upstream 410). Use pm('markets/search', q=...) for cross-venue lookups." + ) + + async def pm_outcome(self, predexon_id: str) -> dict[str, Any]: + """RETIRED — ``/v1/pm/outcomes/{predexon_id}`` no longer exists. + + Predexon sunset market matching on 2026-07-20 and the whole + canonical layer went with it, so this path returns 410 upstream. + Use ``pm("markets/search", q=...)`` for cross-venue lookups. + + Kept as a raising stub rather than deleted so upgrading does not + break imports or attribute access; it raises before any network + I/O, so you never pay a round trip to learn it is gone. + + :raises RetiredEndpointError: always. + """ + raise RetiredEndpointError( + "/v1/pm/outcomes/{predexon_id} was sunset by Predexon on 2026-07-20 (upstream 410). Use pm('markets/search', q=...) for cross-venue lookups." + ) + + async def pm_polymarket_markets(self, **params: Any) -> dict[str, Any]: + """List Polymarket markets (Predexon v2). Tier 1 ($0.001/call).""" + return await self.pm("polymarket/markets", **params) + + async def pm_polymarket_events(self, **params: Any) -> dict[str, Any]: + """List Polymarket events (Predexon v2). Tier 1 ($0.001/call).""" + return await self.pm("polymarket/events", **params) + + async def pm_polymarket_markets_keyset(self, **params: Any) -> dict[str, Any]: + """Polymarket markets with cursor-based keyset pagination. Tier 1 ($0.001/call).""" + return await self.pm("polymarket/markets/keyset", **params) + + async def pm_polymarket_events_keyset(self, **params: Any) -> dict[str, Any]: + """Polymarket events with cursor-based keyset pagination. Tier 1 ($0.001/call).""" + return await self.pm("polymarket/events/keyset", **params) + + async def pm_polymarket_positions(self, **params: Any) -> dict[str, Any]: + """Polymarket open positions (per-wallet, market-level PnL). Tier 1 ($0.001/call).""" + return await self.pm("polymarket/positions", **params) + + async def pm_polymarket_trades(self, **params: Any) -> dict[str, Any]: + """Recent Polymarket trades. Tier 1 ($0.001/call).""" + return await self.pm("polymarket/trades", **params) + + async def pm_polymarket_leaderboard(self, **params: Any) -> dict[str, Any]: + """Polymarket trader leaderboard. Tier 1 ($0.001/call).""" + return await self.pm("polymarket/leaderboard", **params) + + async def pm_kalshi_markets(self, **params: Any) -> dict[str, Any]: + """List Kalshi markets. Tier 1 ($0.001/call).""" + return await self.pm("kalshi/markets", **params) + + async def pm_limitless_markets(self, **params: Any) -> dict[str, Any]: + """List Limitless markets. Tier 1 ($0.001/call).""" + return await self.pm("limitless/markets", **params) + + async def pm_sports_categories(self) -> dict[str, Any]: + """List available sports categories. Tier 1 ($0.001/call). + + .. warning:: + Upstream is returning 500 for every ``sports/*`` path as of + 2026-08-04. The route still resolves, so this keeps working the + moment Predexon restores it, but do not build on it yet. + """ + return await self.pm("sports/categories") + + async def pm_sports_markets(self, **params: Any) -> dict[str, Any]: + """List sports markets grouped by game. Tier 1 ($0.001/call). + + .. warning:: + Upstream is returning 500 for every ``sports/*`` path as of + 2026-08-04. The route still resolves, so this keeps working the + moment Predexon restores it, but do not build on it yet. + """ + return await self.pm("sports/markets", **params) + + async def pm_wallet_identity(self, wallet: str) -> dict[str, Any]: + """Identity + profile for one wallet. Tier 2 ($0.005/call).""" + return await self.pm(f"polymarket/wallet/identity/{wallet}") + + async def pm_wallet_identities(self, addresses: list[str]) -> dict[str, Any]: + """Bulk identity for up to 200 wallet addresses. Tier 2 ($0.005/call).""" + return await self.pm_query("polymarket/wallet/identities", {"addresses": addresses}) + + async def pm_wallet_cluster(self, address: str) -> dict[str, Any]: + """Wallet-cluster discovery (on-chain transfers + identity proofs). Tier 2 ($0.005/call).""" + return await self.pm(f"polymarket/wallet/{address}/cluster") + + # ── Exa Web Search (Powered by Exa) ───────────────────────────────────── + + async def exa(self, path: str, body: dict[str, Any]) -> dict[str, Any]: + """Generic Exa endpoint proxy (POST, Solana payment). Powered by Exa. + + Args: + path: Exa endpoint — one of: "search", "find-similar", "contents", "answer" + body: Request body (see Exa API docs) + """ + return await self._request_with_payment_raw( + f"/v1/exa/{path}", body, timeout=self._search_timeout + ) + + async def exa_search(self, query: str, **kwargs: Any) -> dict[str, Any]: + """Neural and keyword web search via Exa (Solana payment, $0.01/request).""" + return await self._request_with_payment_raw( + "/v1/exa/search", {"query": query, **kwargs}, timeout=self._search_timeout + ) + + async def exa_find_similar(self, url: str, **kwargs: Any) -> dict[str, Any]: + """Find pages semantically similar to a given URL via Exa (Solana payment, $0.01/request).""" + return await self._request_with_payment_raw( + "/v1/exa/find-similar", {"url": url, **kwargs}, timeout=self._search_timeout + ) + + async def exa_contents(self, urls: list[str], **kwargs: Any) -> dict[str, Any]: + """Extract full text content from URLs via Exa (Solana payment, $0.002/URL).""" + return await self._request_with_payment_raw( + "/v1/exa/contents", {"urls": urls, **kwargs}, timeout=self._search_timeout + ) + + async def exa_answer(self, query: str, **kwargs: Any) -> dict[str, Any]: + """AI-generated answer grounded in live web search via Exa (Solana payment, $0.01/request).""" + return await self._request_with_payment_raw( + "/v1/exa/answer", {"query": query, **kwargs}, timeout=self._search_timeout + ) + + # ── DefiLlama (DeFi protocols / TVL / yields / prices) ────────────────── + + async def defi(self, path: str, **params: Any) -> dict[str, Any]: + """Query DefiLlama DeFi data (GET, Solana payment). $0.005/call + ($0.001 for prices/{coins}).""" + return await self._get_with_payment_raw(f"/v1/defillama/{path}", params or None) + + async def defi_protocols(self) -> dict[str, Any]: + """All DeFi protocols with TVL ($0.005/call).""" + return await self.defi("protocols") + + async def defi_protocol(self, slug: str) -> dict[str, Any]: + """Single protocol details + historical TVL ($0.005/call).""" + return await self.defi(f"protocol/{slug}") + + async def defi_chains(self) -> dict[str, Any]: + """Current TVL of every chain ($0.005/call).""" + return await self.defi("chains") + + async def defi_yields(self, **params: Any) -> dict[str, Any]: + """Yield pools with APY/TVL ($0.005/call).""" + return await self.defi("yields", **params) + + async def defi_prices(self, coins: list[str] | str) -> dict[str, Any]: + """Token price lookup ($0.001/call).""" + joined = ",".join(coins) if isinstance(coins, list) else coins + return await self.defi(f"prices/{joined}") + + # ── 0x DEX (swap quotes + gasless) — free passthrough ─────────────────── + + async def dex( + self, + path: str, + *, + method: str = "GET", + body: dict[str, Any] | None = None, + **params: Any, + ) -> dict[str, Any]: + """Query the 0x Swap / Gasless APIs (free — no x402 payment).""" + endpoint = f"/v1/zerox/{path}" + if method.upper() == "POST": + return await self._request_with_payment_raw(endpoint, body or {}) + return await self._get_with_payment_raw(endpoint, params or None) + + async def dex_price(self, **params: Any) -> dict[str, Any]: + """Indicative Permit2 swap price — no commitment (free).""" + return await self.dex("price", **params) + + async def dex_quote(self, **params: Any) -> dict[str, Any]: + """Firm Permit2 swap quote with permit2.eip712 + tx data (free).""" + return await self.dex("quote", **params) + + async def dex_gasless_price(self, **params: Any) -> dict[str, Any]: + """Gasless indicative price quote (free).""" + return await self.dex("gasless/price", **params) + + async def dex_gasless_quote(self, **params: Any) -> dict[str, Any]: + """Gasless firm quote — returns trade.eip712 to sign (free).""" + return await self.dex("gasless/quote", **params) + + async def dex_gasless_submit(self, body: dict[str, Any]) -> dict[str, Any]: + """Submit a signed gasless trade; the 0x relayer pays gas (free).""" + return await self.dex("gasless/submit", method="POST", body=body) + + async def dex_gasless_status(self, trade_hash: str) -> dict[str, Any]: + """Poll a gasless trade's status by tradeHash (free).""" + return await self.dex(f"gasless/status/{trade_hash}") + + async def dex_chains(self) -> dict[str, Any]: + """Chains where the Swap API is supported (free).""" + return await self.dex("swap/chains") + + async def dex_gasless_chains(self) -> dict[str, Any]: + """Chains where the Gasless API is supported (free).""" + return await self.dex("gasless/chains") + + # ── Modal Sandbox (pay-per-call cloud compute) ─────────────────────────── + + async def modal(self, path: str, body: dict[str, Any] | None = None) -> dict[str, Any]: + """Call the Modal sandbox compute API (POST, Solana payment).""" + return await self._request_with_payment_raw(f"/v1/modal/{path}", body or {}) + + async def modal_sandbox_create(self, **body: Any) -> dict[str, Any]: + """Create a sandboxed compute environment ($0.01 CPU / $0.05 GPU).""" + return await self.modal("sandbox/create", body) + + async def modal_sandbox_exec( + self, sandbox_id: str, command: list[str], **body: Any + ) -> dict[str, Any]: + """Execute a command in a sandbox; returns stdout/stderr ($0.001).""" + return await self.modal( + "sandbox/exec", {"sandbox_id": sandbox_id, "command": command, **body} + ) + + async def modal_sandbox_status(self, sandbox_id: str) -> dict[str, Any]: + """Check a sandbox's status ($0.001).""" + return await self.modal("sandbox/status", {"sandbox_id": sandbox_id}) + + async def modal_sandbox_terminate(self, sandbox_id: str) -> dict[str, Any]: + """Terminate a sandbox ($0.001).""" + return await self.modal("sandbox/terminate", {"sandbox_id": sandbox_id}) + + +# A typing placeholder so the chat_completion_stream return type docs above +# don't reference a name pyright can't resolve. +AsyncSolanaIterator = Any diff --git a/blockrun_llm/solana_wallet.py b/blockrun_llm/solana_wallet.py new file mode 100644 index 0000000..636aa3d --- /dev/null +++ b/blockrun_llm/solana_wallet.py @@ -0,0 +1,628 @@ +""" +BlockRun Solana Wallet Management. + +Stores keys as bs58-encoded strings at ~/.blockrun/.solana-session. +Requires: solders>=0.21.0, base58>=2.1.0 +""" + +from __future__ import annotations + +import json +import os +import time +from pathlib import Path +from typing import TYPE_CHECKING + +if TYPE_CHECKING: + from .solana_client import SolanaLLMClient + +WALLET_DIR = Path.home() / ".blockrun" +SOLANA_WALLET_FILE = WALLET_DIR / ".solana-session" + + +def _require_solders() -> None: + try: + import solders # noqa: F401 + except ImportError: + raise ImportError( + "Solana support requires 'solders' and 'base58' packages. " + "Install with: pip install blockrun-llm[solana]" + ) + + +def create_solana_wallet() -> dict[str, str]: + """ + Create a new Solana wallet. + + Returns: + Dict with 'address' (base58 pubkey) and 'private_key' (bs58 secret key) + """ + _require_solders() + from solders.keypair import Keypair # type: ignore + + kp = Keypair() + return { + "address": str(kp.pubkey()), + "private_key": str(kp), # bs58-encoded 64-byte keypair + } + + +def solana_key_to_bytes(private_key: str) -> bytes: + """ + Convert a Solana private key string to bytes (64 bytes). + + Accepts a bs58-encoded 64-byte keypair (standard Solana format), a + bs58-encoded 32-byte seed from other providers (automatically expanded), + the Solana CLI JSON byte-array format (``~/.config/solana/id.json``), or + a 64-byte hex string with or without ``0x``. A 32-byte hex key is + rejected with an explicit hint that it is the EVM (Base) wallet format. + + Args: + private_key: Solana secret key in any accepted encoding + + Returns: + 64-byte secret key as bytes + + Raises: + ValueError: If key is invalid + """ + key = private_key.strip() + + # Solana CLI JSON array format: [12,34,...] with 64 (or 32) byte values + if key.startswith("["): + try: + parsed = json.loads(key) + except json.JSONDecodeError as e: + raise ValueError( + "Invalid Solana private key: looks like a JSON byte array " "but is not valid JSON" + ) from e + if not isinstance(parsed, list) or not all( + isinstance(n, int) and 0 <= n <= 255 for n in parsed + ): + raise ValueError( + "Invalid Solana private key: JSON array must contain only " "byte values (0-255)" + ) + if len(parsed) not in (32, 64): + raise ValueError( + f"Invalid Solana key length: expected 32 or 64 bytes, got {len(parsed)}" + ) + return _expand_key_bytes(bytes(parsed)) + + # Hex forms. bs58 keys are 86-88 chars, so 64/128 hex chars are unambiguous. + hex_body = key[2:] if key[:2] in ("0x", "0X") else key + if len(hex_body) in (64, 128) and all(c in "0123456789abcdefABCDEF" for c in hex_body): + if len(hex_body) == 64: + raise ValueError( + "Invalid Solana private key: this is a 32-byte hex key — the " + "EVM (Base) wallet format, not a Solana key. Solana keys are " + "64 bytes, usually base58-encoded (86-88 characters)." + ) + return _expand_key_bytes(bytes.fromhex(hex_body)) + + try: + from solders.keypair import Keypair # type: ignore + + try: + kp = Keypair.from_base58_string(key) + decoded = bytes(kp) + if len(decoded) == 64: + return decoded + except Exception: + pass + + # Fallback: try as 32-byte seed + import base58 as b58 + + decoded = b58.b58decode(key) + if len(decoded) in (32, 64): + return _expand_key_bytes(decoded) + + raise ValueError(f"Expected 32 or 64 bytes, got {len(decoded)}") + except Exception as e: + # Wrap every failure — including the ``ValueError`` modern ``base58`` + # raises on invalid characters — in the documented message. A bare + # ``except ValueError: raise`` here used to leak base58's raw + # "Invalid character" error past the wrapper, breaking callers (and + # the test) that match on "Invalid Solana private key". + raise ValueError( + f"Invalid Solana private key: {e}. Expected a base58-encoded " + "64-byte key (standard Solana format), a 64-byte hex string, or " + "a Solana CLI JSON byte array." + ) from e + + +def _expand_key_bytes(decoded: bytes) -> bytes: + """Expand a 32-byte seed (or normalize a 64-byte keypair) to 64 bytes.""" + _require_solders() + from solders.keypair import Keypair # type: ignore + + if len(decoded) == 32: + return bytes(Keypair.from_seed(decoded)) + return bytes(Keypair.from_seed(decoded[:32])) + + +def get_solana_public_key(private_key: str) -> str: + """ + Get the Solana public key (address) from a bs58 private key. + + Accepts both 64-byte full keypairs and 32-byte seeds. + + Args: + private_key: bs58-encoded Solana secret key (32 or 64 bytes) + + Returns: + Base58 public key string + """ + _require_solders() + from solders.keypair import Keypair # type: ignore + + # solana_key_to_bytes handles every accepted encoding, including 32-byte + # seeds, so a failure here is final — no fallback decode. + secret = solana_key_to_bytes(private_key) + kp = Keypair.from_seed(secret[:32]) + return str(kp.pubkey()) + + +def save_solana_wallet(private_key: str) -> Path: + WALLET_DIR.mkdir(exist_ok=True) + SOLANA_WALLET_FILE.write_text(private_key) + SOLANA_WALLET_FILE.chmod(0o600) + return SOLANA_WALLET_FILE + + +def _expand_solana_seed(private_key: str) -> str: + """If private_key is a 32-byte seed, expand to 64-byte keypair bs58 string.""" + import base58 as b58 + from solders.keypair import Keypair # type: ignore + + decoded = b58.b58decode(private_key) + if len(decoded) == 32: + kp = Keypair.from_seed(decoded) + return b58.b58encode(bytes(kp)).decode() + return private_key + + +def scan_solana_wallets() -> list[dict[str, str]]: + """ + Discover ~/./solana-wallet.json files from other providers. + + Each file should contain JSON with "privateKey" and "address" fields. + Results are sorted by modification time (most recent first). Discovery is + opt-in and must never replace the canonical BlockRun wallet automatically. + 32-byte seeds are automatically converted to 64-byte keypairs. + + Returns: + List of dicts with 'private_key', 'address' and 'source', most recent + first. 'address' is the file's own claim — use + list_discovered_solana_wallets() for an address derived from the key. + """ + home = Path.home() + results: list[tuple] = [] # (mtime, private_key, address, source) + + try: + for entry in home.iterdir(): + if not entry.name.startswith(".") or not entry.is_dir(): + continue + wallet_file = entry / "solana-wallet.json" + if not wallet_file.is_file(): + continue + try: + data = json.loads(wallet_file.read_text()) + pk = data.get("privateKey", "") + addr = data.get("address", "") + if pk and addr: + # Expand 32-byte seeds to full keypairs + try: + pk = _expand_solana_seed(pk) + except Exception: + pass + mtime = wallet_file.stat().st_mtime + results.append((mtime, pk, addr, str(wallet_file))) + except (json.JSONDecodeError, OSError): + continue + except OSError: + pass + + # Sort by modification time, most recent first + results.sort(key=lambda x: x[0], reverse=True) + return [{"private_key": pk, "address": addr, "source": src} for _, pk, addr, src in results] + + +def list_discovered_solana_wallets() -> list[dict[str, str]]: + """ + List Solana wallets from other applications, safe to show to a user. + + Solana counterpart of ``wallet.list_discovered_wallets``: no secret key is + returned and the address is derived from the key rather than trusted from + the file. Nothing here is active — adopt one with import_solana_wallet(). + + Returns: + List of dicts with 'address' and 'source', most recent first + """ + listed = [] + for entry in scan_solana_wallets(): + try: + address = get_solana_public_key(entry["private_key"]) + except Exception: + continue + listed.append({"address": address, "source": entry.get("source", "")}) + return listed + + +def import_solana_wallet(address: str) -> str: + """ + Adopt a discovered Solana wallet, making it the active BlockRun wallet. + + Solana counterpart of ``wallet.import_wallet``. Matching is done against the + address derived from each discovered key, and the current + ~/.blockrun/.solana-session is backed up before being overwritten. + + Args: + address: Address to adopt, as shown by list_discovered_solana_wallets() + + Returns: + The adopted address + + Raises: + ValueError: If no discovered wallet derives to that address + """ + wanted = address.strip() + + for entry in scan_solana_wallets(): + try: + derived = get_solana_public_key(entry["private_key"]) + except Exception: + continue + + # Base58 is case-sensitive — compare exactly, unlike EVM hex. + if derived != wanted: + continue + + if SOLANA_WALLET_FILE.exists(): + current = SOLANA_WALLET_FILE.read_text().strip() + if current and current != entry["private_key"]: + backup = SOLANA_WALLET_FILE.with_name(f".solana-session.backup-{int(time.time())}") + backup.write_text(current) + backup.chmod(0o600) + + save_solana_wallet(entry["private_key"]) + return derived + + available = [w["address"] for w in list_discovered_solana_wallets()] + raise ValueError( + f"No discovered wallet controls {address}. " + f"Available: {', '.join(available) if available else 'none'}" + ) + + +def load_solana_wallet() -> str | None: + """ + Load Solana wallet private key. + + Priority: + 1. ~/.blockrun/.solana-session + """ + # The canonical BlockRun wallet always wins over a discovered provider key. + if SOLANA_WALLET_FILE.exists(): + try: + key = SOLANA_WALLET_FILE.read_text().strip() + except OSError: + return None # unreadable (bad perms/ownership) → treat as "no wallet" + if key: + return key + return None + + +def _public_key_from(private_key: str, source: str) -> str: + """Derive the public key, attributing failures to where the key was loaded from.""" + try: + return get_solana_public_key(private_key) + except ValueError as e: + raise ValueError(f"{e} (key loaded from {source})") from e + + +def get_or_create_solana_wallet() -> dict[str, object]: + """ + Get existing Solana wallet or create new one. + + Priority: + 1. SOLANA_WALLET_KEY env var + 2. ~/.blockrun/.solana-session + 3. Create new + + Returns: + Dict with 'address', 'private_key', 'is_new' + """ + # 1. Environment variable + env_key = os.environ.get("SOLANA_WALLET_KEY") + if env_key: + return { + "private_key": env_key, + "address": _public_key_from(env_key, "the SOLANA_WALLET_KEY environment variable"), + "is_new": False, + } + + # 2. Canonical BlockRun session file. scan_solana_wallets() is exposed + # only for an explicit migration flow. + if SOLANA_WALLET_FILE.exists(): + file_key = SOLANA_WALLET_FILE.read_text().strip() + if file_key: + return { + "private_key": file_key, + "address": _public_key_from(file_key, str(SOLANA_WALLET_FILE)), + "is_new": False, + } + + # 3. Create new + wallet = create_solana_wallet() + save_solana_wallet(wallet["private_key"]) + return {**wallet, "is_new": True} + + +def format_solana_wallet_migration_notice(new_address: str) -> str | None: + """ + Warn when a new Solana wallet was created while provider wallets exist. + + Solana counterpart of ``wallet.format_wallet_migration_notice``. Addresses + are derived from the discovered secret key rather than trusted from the + file's "address" field. + + Args: + new_address: Address of the wallet that was just created + + Returns: + Formatted notice, or None if nothing was discovered + """ + try: + discovered = scan_solana_wallets() + except Exception: + return None + + addresses = [] + for entry in discovered: + try: + addresses.append(get_solana_public_key(entry["private_key"])) + except Exception: + continue + + if not addresses: + return None + + found = "\n".join(f" {addr}" for addr in addresses) + return f""" +NOTICE: BlockRun created a new Solana wallet, but also found existing +wallet(s) belonging to other applications on this system: + +{found} + +BlockRun now uses only its own wallet: + + {new_address} + +Discovered wallets are never adopted automatically — one may belong to a +different application, or have been planted to make you fund an address you +do not control. + +If an address above is yours and holds your USDC, adopt it deliberately: + + from blockrun_llm import import_solana_wallet + import_solana_wallet("") + +Your current wallet is backed up first. You can also set +SOLANA_WALLET_KEY= for a single run without changing anything. +""" + + +def setup_agent_solana_wallet(silent: bool = False) -> SolanaLLMClient: + """ + Set up Solana wallet for agent use and return a SolanaLLMClient. + + This is the entry point for Claude Code skills and other agent runtimes. + It auto-creates a Solana wallet if needed and prints address if new. + + Args: + silent: If True, don't print welcome message (default: False) + + Returns: + Configured SolanaLLMClient ready for use + + Example: + from blockrun_llm import setup_agent_solana_wallet + + client = setup_agent_solana_wallet() + response = client.chat("openai/gpt-5.2", "Hello!") + """ + import sys + + result = get_or_create_solana_wallet() + + if result["is_new"]: + # Printed even when silent: `silent` suppresses the welcome message, + # and losing sight of a funded wallet is not something to stay quiet + # about. + notice = format_solana_wallet_migration_notice(str(result["address"])) + if notice: + print(notice, file=sys.stderr) + + if not silent: + print(f"New Solana wallet created: {result['address']}", file=sys.stderr) + + from .solana_client import SolanaLLMClient + + return SolanaLLMClient(private_key=result["private_key"]) + + +def get_solana_usdc_balance(address: str, rpc_url: str | None = None) -> float: + """ + Get USDC-SPL balance for a Solana address. + + Args: + address: Solana wallet address (base58) + rpc_url: Solana RPC endpoint (default: mainnet-beta) + + Returns: + USDC balance as float (6 decimals) + """ + import httpx + + rpc = rpc_url or "https://api.mainnet-beta.solana.com" + # USDC mint on Solana mainnet + usdc_mint = "EPjFWdd5AufqSSqeM2qN1xzybapC8G4wEGGkZwyTDt1v" + + try: + resp = httpx.post( + rpc, + json={ + "jsonrpc": "2.0", + "id": 1, + "method": "getTokenAccountsByOwner", + "params": [ + address, + {"mint": usdc_mint}, + {"encoding": "jsonParsed"}, + ], + }, + timeout=10, + ) + resp.raise_for_status() + data = resp.json() + + accounts = data.get("result", {}).get("value", []) + if not accounts: + return 0.0 + + # Sum all USDC token accounts (usually just one) + total = 0.0 + for acct in accounts: + info = acct.get("account", {}).get("data", {}).get("parsed", {}).get("info", {}) + token_amount = info.get("tokenAmount", {}) + total += float(token_amount.get("uiAmount", 0)) + return total + + except Exception: + return 0.0 + + +# QR code file paths for Solana +SOLANA_QR_FILE = WALLET_DIR / "solana_qr.png" +SOLANA_QR_ASCII_FILE = WALLET_DIR / "solana_qr.txt" + + +def generate_solana_qr_ascii(address: str) -> str: + """ + Generate ASCII QR code for Solana wallet funding. + Uses solana: URI scheme. Caches to ~/.blockrun/solana_qr.txt. + + Args: + address: Solana wallet address (base58) + + Returns: + ASCII art QR code string + """ + solana_uri = f"solana:{address}" + cache_key = f"v1:{solana_uri}" + + # Try cache + if SOLANA_QR_ASCII_FILE.exists(): + try: + cached = SOLANA_QR_ASCII_FILE.read_text() + lines = cached.split("\n", 1) + if len(lines) == 2 and lines[0] == cache_key: + return lines[1] + except Exception: + pass + + # Generate new QR + try: + from io import StringIO + + import qrcode + + qr = qrcode.QRCode( + version=1, + error_correction=qrcode.constants.ERROR_CORRECT_L, + box_size=1, + border=1, + ) + qr.add_data(solana_uri) + qr.make(fit=True) + + f = StringIO() + qr.print_ascii(out=f, invert=True) + qr_ascii = f.getvalue() + + # Cache + try: + WALLET_DIR.mkdir(exist_ok=True) + SOLANA_QR_ASCII_FILE.write_text(f"{cache_key}\n{qr_ascii}") + except Exception: + pass + + return qr_ascii + + except ImportError: + return f"[QR code requires 'qrcode' package: pip install qrcode[pil]]\nAddress: {address}" + + +def save_solana_wallet_qr(address: str, path: str | None = None) -> str: + """ + Save Solana QR code as PNG image. + + Args: + address: Solana wallet address (base58) + path: Optional custom path (default: ~/.blockrun/solana_qr.png) + + Returns: + Path to saved QR image, or empty string on failure + """ + try: + import qrcode + + solana_uri = f"solana:{address}" + + qr = qrcode.QRCode( + version=4, + error_correction=qrcode.constants.ERROR_CORRECT_L, + box_size=10, + border=2, + ) + qr.add_data(solana_uri) + qr.make(fit=True) + + img = qr.make_image(fill_color="black", back_color="white").convert("RGB") + + save_path = Path(path) if path else SOLANA_QR_FILE + save_path.parent.mkdir(exist_ok=True) + img.save(str(save_path)) + + return str(save_path) + + except ImportError: + return "" + + +def open_solana_wallet_qr(address: str) -> str: + """ + Generate Solana QR code and open it in the default image viewer. + + Args: + address: Solana wallet address (base58) + + Returns: + Path to saved QR image + """ + import platform + import subprocess + + qr_path = save_solana_wallet_qr(address) + if qr_path: + try: + if platform.system() == "Darwin": + subprocess.run(["open", qr_path], check=True) + elif platform.system() == "Windows": + subprocess.run(["start", qr_path], shell=True, check=True) + else: + subprocess.run(["xdg-open", qr_path], check=True) + except Exception: + pass + return qr_path diff --git a/blockrun_llm/speech.py b/blockrun_llm/speech.py new file mode 100644 index 0000000..04bfd1b --- /dev/null +++ b/blockrun_llm/speech.py @@ -0,0 +1,405 @@ +""" +BlockRun Speech Client - Text-to-speech and sound effects (ElevenLabs) via x402 micropayments. + +SECURITY NOTE - Private Key Handling: +===================================== +Your private key NEVER leaves your machine. Here's what happens: + +1. Key stays local - only used to sign an EIP-712 typed data message +2. Only the SIGNATURE is sent in the PAYMENT-SIGNATURE header +3. BlockRun verifies the signature on-chain via Coinbase CDP facilitator + +Usage: + from blockrun_llm import SpeechClient + + client = SpeechClient() # Uses BLOCKRUN_WALLET_KEY from env + + # Text-to-speech (paid, price scales with character count) + result = client.generate("Hello from BlockRun!", voice="sarah") + print(result.data[0].url) # audio URL + + # Sound effects (paid, flat $0.05/generation) + result = client.sound_effect("rain on a tin roof, distant thunder") + print(result.data[0].url) + + # List available voices (free, rate-limited) + voices = client.list_voices() + +Models & pricing: + elevenlabs/flash-v2.5 $0.05/1k chars ~75ms latency, 32 languages (default) + elevenlabs/turbo-v2.5 $0.05/1k chars ~250ms latency, 32 languages + elevenlabs/multilingual-v2 $0.10/1k chars long-form narration, 29 languages + elevenlabs/v3 $0.10/1k chars max expressiveness, 70+ languages + elevenlabs/sound-effects $0.05/generation (up to 22s) + +Price = (characters / 1000) x model rate, minimum $0.001/request. +""" + +from __future__ import annotations + +import os +from typing import Any + +import httpx +from dotenv import load_dotenv +from eth_account import Account + +from .apikey import ( + api_key_base_url, + auth_headers, + missing_credential_error, + payment_mode, + raise_for_api_key_402, + resolve_api_key, +) +from .tx_log import paid_request_error_prefix +from .types import APIError, PaymentError, SpeechResponse, retry_after_of +from .validation import ( + sanitize_error_response, + validate_api_url, + validate_private_key, +) +from .x402 import create_payment_payload, extract_payment_details, parse_payment_required + +load_dotenv() + +# Friendly voice aliases accepted by /v1/audio/speech (raw ElevenLabs +# voice_ids pass through unchanged). Mirrors backend VOICE_ALIASES. +VOICE_ALIASES = [ + "sarah", # Mature, reassuring, confident (default) + "george", # Warm, captivating storyteller + "laura", # Enthusiast, quirky + "charlie", # Deep, confident, energetic + "river", # Relaxed, neutral, informative + "roger", # Laid-back, casual, resonant + "callum", # Husky trickster + "harry", # Fierce warrior +] + + +class SpeechClient: + """ + BlockRun Speech Client (BlockRun Voice). + + Text-to-speech and sound-effect generation using ElevenLabs models + with automatic x402 micropayments on Base chain. + + TTS pricing scales with input characters; sound effects are flat + $0.05/generation. + """ + + DEFAULT_API_URL = "https://blockrun.ai/api" + DEFAULT_MODEL = "elevenlabs/flash-v2.5" + DEFAULT_SOUNDFX_MODEL = "elevenlabs/sound-effects" + DEFAULT_VOICE = "sarah" + DEFAULT_TIMEOUT = 120.0 # synthesis is synchronous (<1s for Flash) + + def __init__( + self, + private_key: str | None = None, + api_url: str | None = None, + timeout: float = 120.0, + ): + """ + Initialize the BlockRun Speech client. + + Args: + private_key: EVM wallet private key (or set BLOCKRUN_WALLET_KEY env var) + api_url: API endpoint URL (default: https://blockrun.ai/api) + timeout: Request timeout in seconds (default: 120) + """ + from .wallet import load_wallet + + # Account rail first, and before any wallet variable is read: an API + # key is a complete credential on its own, so demanding a private key + # alongside it would make every API-key user invent a wallet they + # never use. See apikey.py for the precedence rule. + api_key = resolve_api_key(private_key) + key = ( + None + if api_key + else ( + private_key + or os.environ.get("BLOCKRUN_WALLET_KEY") + or os.environ.get("BASE_CHAIN_WALLET_KEY") + or load_wallet() + ) + ) + if not api_key and not key: + raise missing_credential_error() + + if key: + validate_private_key(key) + self.api_key = api_key + # No wallet on the account rail: nothing is signed locally. + self.account = Account.from_key(key) if key else None + + # BLOCKRUN_API_URL names an x402 gateway; an API-key client must not + # follow it and hand the key to a host set up for another rail. + api_url_raw = ( + api_key_base_url(api_url) + if api_key + else (api_url or os.environ.get("BLOCKRUN_API_URL") or self.DEFAULT_API_URL) + ) + validate_api_url(api_url_raw) + self.api_url = api_url_raw.rstrip("/") + + self.timeout = timeout + self._client = httpx.Client(headers=auth_headers(api_key), timeout=timeout) + + def generate( + self, + input: str, + *, + model: str | None = None, + voice: str | None = None, + response_format: str | None = None, + speed: float | None = None, + ) -> SpeechResponse: + """ + Synthesize speech from text (OpenAI-compatible TTS). + + Price scales with character count: (chars / 1000) x model rate, + minimum $0.001/request. Synthesis is synchronous. + + Args: + input: Text to synthesize. Per-model character caps apply + (flash/turbo 40k, multilingual-v2 10k, v3 5k). + model: Speech model ID (default: "elevenlabs/flash-v2.5") + Options: "elevenlabs/flash-v2.5", "elevenlabs/turbo-v2.5", + "elevenlabs/multilingual-v2", "elevenlabs/v3" + voice: Voice alias (sarah, george, laura, charlie, river, roger, + callum, harry) or a raw ElevenLabs voice_id + (default: "sarah") + response_format: "mp3" (default), "opus", "pcm", or "wav" + speed: Playback speed 0.7-1.2 (optional) + + Returns: + SpeechResponse with audio URL, format, and character count + + Raises: + PaymentError: If wallet has insufficient balance + APIError: If the API returns an error + + Example: + result = client.generate("Welcome to BlockRun.", voice="george") + print(result.data[0].url) + """ + body: dict[str, Any] = { + "model": model or self.DEFAULT_MODEL, + "input": input, + } + if voice: + body["voice"] = voice + if response_format: + body["response_format"] = response_format + if speed is not None: + body["speed"] = speed + + return self._request_with_payment("/v1/audio/speech", body) + + # OpenAI-style alias + speak = generate + + def sound_effect( + self, + text: str, + *, + model: str | None = None, + duration_seconds: float | None = None, + prompt_influence: float | None = None, + response_format: str | None = None, + ) -> SpeechResponse: + """ + Generate a cinematic sound effect from a text prompt. + + Flat $0.05/generation, up to 22 seconds of audio. + + Args: + text: Sound effect description (max 1000 chars). + E.g. "rain on a tin roof", "sci-fi door whoosh" + model: Model ID (default: "elevenlabs/sound-effects") + duration_seconds: Target duration 0.5-22s (optional; auto if unset) + prompt_influence: 0-1, higher follows the prompt more literally + response_format: "mp3" (default), "opus", "pcm", or "wav" + + Returns: + SpeechResponse with audio URL and format + + Example: + result = client.sound_effect("crackling campfire at night") + print(result.data[0].url) + """ + body: dict[str, Any] = { + "model": model or self.DEFAULT_SOUNDFX_MODEL, + "text": text, + } + if duration_seconds is not None: + body["duration_seconds"] = duration_seconds + if prompt_influence is not None: + body["prompt_influence"] = prompt_influence + if response_format: + body["response_format"] = response_format + + return self._request_with_payment("/v1/audio/sound-effects", body) + + def list_voices(self) -> list[dict[str, Any]]: + """ + List available voices for TTS (free, rate-limited 60 req/min/IP). + + Returns: + List of voice dicts. Pass a voice's `alias` (if present) or + `voice_id` as the `voice` argument to generate(). + """ + response = self._client.get(f"{self.api_url}/v1/audio/voices") + raise_for_api_key_402(response, self.api_key) + + if response.status_code != 200: + try: + error_body = response.json() + except Exception: + error_body = {"error": "Request failed"} + raise APIError( + f"API error: {response.status_code}", + response.status_code, + sanitize_error_response(error_body), + retry_after=retry_after_of(response), + ) + + return response.json().get("data", []) + + def _request_with_payment(self, endpoint: str, body: dict[str, Any]) -> SpeechResponse: + """Make a request with automatic x402 payment handling.""" + url = f"{self.api_url}{endpoint}" + + response = self._client.post( + url, + json=body, + headers={"Content-Type": "application/json"}, + ) + + if response.status_code == 402: + # Account rail: a 402 is the account being out of credit, not a + # challenge to sign. Nothing here can sign, so say so plainly. + raise_for_api_key_402(response, self.api_key) + return self._handle_payment_and_retry(url, endpoint, body, response) + + if response.status_code != 200: + try: + error_body = response.json() + except Exception: + error_body = {"error": "Request failed"} + raise APIError( + f"API error: {response.status_code}", + response.status_code, + sanitize_error_response(error_body), + retry_after=retry_after_of(response), + ) + + return SpeechResponse(**response.json()) + + def _handle_payment_and_retry( + self, + url: str, + endpoint: str, + body: dict[str, Any], + response: httpx.Response, + ) -> SpeechResponse: + """Handle 402 response: parse requirements, sign payment, retry.""" + payment_header = response.headers.get("payment-required") + if not payment_header: + try: + resp_body = response.json() + if "x402" in resp_body: + payment_header = resp_body + except Exception: + pass + + if not payment_header: + raise PaymentError("402 response but no payment requirements found") + + if isinstance(payment_header, str): + payment_required = parse_payment_required(payment_header) + else: + payment_required = payment_header + + details = extract_payment_details(payment_required) + resource = details.get("resource") or {} + extensions = payment_required.get("extensions", {}) + + payment_payload = create_payment_payload( + account=self.account, + recipient=details["recipient"], + amount=details["amount"], + network=details.get("network", "eip155:8453"), + resource_url=resource.get("url", f"{self.api_url}{endpoint}"), + resource_description=resource.get("description", "BlockRun Voice"), + max_timeout_seconds=details.get("maxTimeoutSeconds", 300), + extra=details.get("extra"), + asset=details.get("asset"), + extensions=extensions, + ) + + retry_response = self._client.post( + url, + json=body, + headers={ + "Content-Type": "application/json", + "PAYMENT-SIGNATURE": payment_payload, + }, + ) + + if retry_response.status_code == 402: + # Account rail: a 402 is the account being out of credit, not a + # challenge to sign. Nothing here can sign, so say so plainly. + raise_for_api_key_402(retry_response, self.api_key) + raise PaymentError("Payment was rejected. Check your wallet balance.") + + if retry_response.status_code != 200: + try: + error_body = retry_response.json() + except Exception: + error_body = {"error": "Request failed"} + raise APIError( + f"{paid_request_error_prefix(retry_response.headers)}: {retry_response.status_code}", + retry_response.status_code, + sanitize_error_response(error_body), + retry_after=retry_after_of(retry_response), + ) + + data = retry_response.json() + # Attach tx hash from response header + tx_hash = retry_response.headers.get("x-payment-receipt") or retry_response.headers.get( + "X-Payment-Receipt" + ) + if tx_hash: + data["txHash"] = tx_hash + + return SpeechResponse(**data) + + @property + def payment_mode(self) -> str: + """Which rail this client pays on: ``"apikey"`` or ``"wallet"``. + + Worth checking once at startup when both a key and a wallet are + configured in the environment: it is the difference between + spending credit and spending USDC.""" + return payment_mode(self) + + def get_wallet_address(self) -> str: + # No address on the account rail: payment comes from prepaid + # credit, so there is nothing to return but the empty string. + if self.api_key: + return "" + """Get the wallet address being used for payments.""" + return self.account.address + + def close(self): + """Close the HTTP client.""" + self._client.close() + + def __enter__(self): + return self + + def __exit__(self, exc_type, exc_val, exc_tb): + self.close() diff --git a/blockrun_llm/surf.py b/blockrun_llm/surf.py new file mode 100644 index 0000000..c0b5dc5 --- /dev/null +++ b/blockrun_llm/surf.py @@ -0,0 +1,461 @@ +""" +BlockRun Surf Client - asksurf.ai crypto-data gateway via x402 micropayments. + +Surf is a single backend partner exposing ~83 crypto-intelligence endpoints +(exchange data, on-chain SQL, prediction markets, wallet/social analytics, …). + +Pricing is tiered: + Tier 1 $0.001 market data, lists, single-token reads + Tier 2 $0.005 AI-derived intelligence (rankings, trends, search) + Tier 3 $0.020 heavy LLM reports + on-chain SQL/structured queries + +Usage: + from blockrun_llm import SurfClient + + client = SurfClient() + + # Discovery + print(SurfClient.endpoints()) # full catalog (list of dicts) + print(client.price("market/ranking")) # 0.001 + print(client.endpoint_info("onchain/sql")) # {'method': 'POST', 'tier': 3, ...} + + # GET endpoints — pass query params + data = client.get("market/ranking", {"limit": 20}) + price = client.get("exchange/price", {"pair": "BTC/USDT"}) + + # POST endpoints — JSON body + result = client.post("onchain/sql", {"query": "SELECT count() FROM ethereum.blocks"}) + + # Generic helper (auto-routes GET/POST from the catalog) + out = client.call("token/holders", params={"address": "0x...", "chain": "ethereum"}) + +SECURITY NOTE: your private key never leaves your machine. Only EIP-712 +signatures are sent in the PAYMENT-SIGNATURE header. +""" + +from __future__ import annotations + +import os +from typing import Any + +import httpx +from dotenv import load_dotenv +from eth_account import Account +from typing_extensions import Self + +from .apikey import ( + api_key_base_url, + auth_headers, + missing_credential_error, + payment_mode, + raise_for_api_key_402, + resolve_api_key, +) +from .tx_log import paid_request_error_prefix +from .types import APIError, PaymentError, retry_after_of +from .validation import ( + sanitize_error_response, + validate_api_url, + validate_private_key, +) +from .x402 import ( + create_payment_payload, + extract_payment_details, + parse_payment_required, +) + +load_dotenv() + + +# Mirrors src/lib/surf.ts SURF_TIER_*_PRICE on the backend. +SURF_TIER_PRICES: dict[int, float] = { + 1: 0.001, + 2: 0.005, + 3: 0.020, +} + + +# Mirrors src/lib/surf.ts SURF_ENDPOINTS. Each tuple is (path, method, tier, required_params). +# Keep this list in sync when backend endpoints change — used for discovery, parameter +# validation, and auto GET/POST routing in SurfClient.call(). +_SURF_CATALOG: list[tuple[str, str, int, tuple[str, ...]]] = [ + # exchange + ("exchange/markets", "GET", 1, ()), + ("exchange/price", "GET", 1, ("pair",)), + ("exchange/perp", "GET", 1, ("pair",)), + ("exchange/depth", "GET", 2, ("pair",)), + ("exchange/klines", "GET", 2, ("pair",)), + ("exchange/funding-history", "GET", 2, ("pair",)), + ("exchange/long-short-ratio", "GET", 2, ("pair",)), + # fund + ("fund/detail", "GET", 1, ()), + ("fund/portfolio", "GET", 1, ()), + ("fund/ranking", "GET", 1, ("metric",)), + # market + ("market/ranking", "GET", 1, ()), + ("market/fear-greed", "GET", 1, ()), + ("market/futures", "GET", 1, ()), + ("market/price", "GET", 1, ("symbol",)), + ("market/etf", "GET", 1, ("symbol",)), + ("market/options", "GET", 1, ("symbol",)), + ("market/liquidation/exchange-list", "GET", 2, ()), + ("market/liquidation/order", "GET", 2, ()), + ("market/liquidation/chart", "GET", 2, ("symbol",)), + ("market/onchain-indicator", "GET", 2, ("symbol", "metric")), + ("market/price-indicator", "GET", 2, ("indicator", "symbol")), + # news + ("news/feed", "GET", 1, ()), + ("news/detail", "GET", 1, ("id",)), + # onchain + ("onchain/bridge/ranking", "GET", 1, ()), + ("onchain/yield/ranking", "GET", 1, ()), + ("onchain/gas-price", "GET", 1, ("chain",)), + ("onchain/tx", "GET", 1, ("hash", "chain")), + ("onchain/schema", "GET", 3, ()), + ("onchain/query", "POST", 3, ()), + ("onchain/sql", "POST", 3, ()), + # prediction-market + ("prediction-market/category-metrics", "GET", 1, ()), + ("prediction-market/polymarket/ranking", "GET", 1, ()), + ("prediction-market/polymarket/trades", "GET", 1, ()), + ("prediction-market/polymarket/markets", "GET", 1, ("market_slug",)), + ("prediction-market/polymarket/events", "GET", 1, ("event_slug",)), + ("prediction-market/polymarket/prices", "GET", 1, ("condition_id",)), + ("prediction-market/polymarket/volumes", "GET", 1, ("condition_id",)), + ("prediction-market/polymarket/open-interest", "GET", 1, ("condition_id",)), + ("prediction-market/polymarket/positions", "GET", 2, ("address",)), + ("prediction-market/polymarket/activity", "GET", 2, ("address",)), + ("prediction-market/kalshi/ranking", "GET", 1, ()), + ("prediction-market/kalshi/markets", "GET", 1, ("market_ticker",)), + ("prediction-market/kalshi/events", "GET", 1, ("event_ticker",)), + ("prediction-market/kalshi/prices", "GET", 1, ("ticker",)), + ("prediction-market/kalshi/trades", "GET", 1, ("ticker",)), + ("prediction-market/kalshi/volumes", "GET", 1, ("ticker",)), + ("prediction-market/kalshi/open-interest", "GET", 1, ("ticker",)), + # project + ("project/detail", "GET", 1, ()), + ("project/defi/metrics", "GET", 1, ("metric",)), + ("project/defi/ranking", "GET", 1, ("metric",)), + # search + ("search/airdrop", "GET", 2, ()), + ("search/events", "GET", 2, ()), + ("search/kalshi", "GET", 2, ()), + ("search/polymarket", "GET", 2, ()), + ("search/web", "GET", 2, ("q",)), + ("search/project", "GET", 2, ("q",)), + ("search/news", "GET", 2, ("q",)), + ("search/wallet", "GET", 2, ("q",)), + ("search/fund", "GET", 2, ("q",)), + ("search/social/people", "GET", 2, ("q",)), + ("search/social/posts", "GET", 2, ("q",)), + # social + ("social/detail", "GET", 2, ()), + ("social/ranking", "GET", 2, ()), + ("social/smart-followers/history", "GET", 2, ()), + ("social/mindshare", "GET", 2, ("q", "interval")), + ("social/tweets", "GET", 1, ("ids",)), + ("social/tweet/replies", "GET", 1, ("tweet_id",)), + ("social/user", "GET", 1, ("handle",)), + ("social/user/followers", "GET", 1, ("handle",)), + ("social/user/following", "GET", 1, ("handle",)), + ("social/user/posts", "GET", 1, ("handle",)), + ("social/user/replies", "GET", 1, ("handle",)), + # token + ("token/tokenomics", "GET", 1, ()), + ("token/dex-trades", "GET", 2, ("address",)), + ("token/holders", "GET", 2, ("address", "chain")), + ("token/transfers", "GET", 2, ("address", "chain")), + # wallet + ("wallet/detail", "GET", 2, ("address",)), + ("wallet/history", "GET", 2, ("address",)), + ("wallet/net-worth", "GET", 2, ("address",)), + ("wallet/transfers", "GET", 2, ("address",)), + ("wallet/protocols", "GET", 2, ("address",)), + ("wallet/labels/batch", "GET", 2, ("addresses",)), + # web + ("web/fetch", "GET", 2, ("url",)), +] + +_CATALOG_BY_PATH: dict[str, tuple[str, int, tuple[str, ...]]] = { + path: (method, tier, required) for path, method, tier, required in _SURF_CATALOG +} + + +class SurfClient: + """ + BlockRun Surf Client. + + Wraps the `/v1/surf/*` partner proxy. Use SurfClient.endpoints() for discovery + and `get()` / `post()` / `call()` to fetch data. Payment is automatic on every + request via x402 (tier 1 / 2 / 3 → $0.001 / $0.005 / $0.020). + """ + + DEFAULT_API_URL = "https://blockrun.ai/api" + DEFAULT_TIMEOUT = 60.0 + + def __init__( + self, + private_key: str | None = None, + api_url: str | None = None, + timeout: float = DEFAULT_TIMEOUT, + ): + from .wallet import load_wallet + + # Account rail first, and before any wallet variable is read: an API + # key is a complete credential on its own, so demanding a private key + # alongside it would make every API-key user invent a wallet they + # never use. See apikey.py for the precedence rule. + api_key = resolve_api_key(private_key) + key = ( + None + if api_key + else ( + private_key + or os.environ.get("BLOCKRUN_WALLET_KEY") + or os.environ.get("BASE_CHAIN_WALLET_KEY") + or load_wallet() + ) + ) + if not api_key and not key: + raise missing_credential_error() + + if key: + validate_private_key(key) + self.api_key = api_key + # No wallet on the account rail: nothing is signed locally. + self.account = Account.from_key(key) if key else None + + # BLOCKRUN_API_URL names an x402 gateway; an API-key client must not + # follow it and hand the key to a host set up for another rail. + api_url_raw = ( + api_key_base_url(api_url) + if api_key + else (api_url or os.environ.get("BLOCKRUN_API_URL") or self.DEFAULT_API_URL) + ) + validate_api_url(api_url_raw) + self.api_url = api_url_raw.rstrip("/") + + self.timeout = timeout + self._client = httpx.Client(headers=auth_headers(api_key), timeout=timeout) + + # ------------------------------------------------------------ Discovery API + + @staticmethod + def endpoints() -> list[dict[str, Any]]: + """Return the full Surf endpoint catalog with method, tier, and price.""" + return [ + { + "path": path, + "method": method, + "tier": tier, + "price_usd": SURF_TIER_PRICES[tier], + "required_params": list(required), + } + for path, method, tier, required in _SURF_CATALOG + ] + + @staticmethod + def endpoint_info(path: str) -> dict[str, Any] | None: + """Return catalog info for one path, or None if unknown.""" + entry = _CATALOG_BY_PATH.get(path) + if not entry: + return None + method, tier, required = entry + return { + "path": path, + "method": method, + "tier": tier, + "price_usd": SURF_TIER_PRICES[tier], + "required_params": list(required), + } + + @staticmethod + def price(path: str) -> float: + """Return the settled USDC price for a Surf endpoint.""" + info = SurfClient.endpoint_info(path) + if info is None: + raise ValueError(f"Unknown Surf endpoint: {path!r}") + return info["price_usd"] + + # ---------------------------------------------------------------- Core HTTP + + def get(self, path: str, params: dict[str, Any] | None = None) -> dict[str, Any]: + """GET an `/v1/surf/{path}` endpoint with optional query params.""" + self._validate_path(path, "GET", params or {}) + return self._request("GET", path, params=params, json_body=None) + + def post(self, path: str, body: dict[str, Any] | None = None) -> dict[str, Any]: + """POST an `/v1/surf/{path}` endpoint with an optional JSON body.""" + self._validate_path(path, "POST", body or {}) + return self._request("POST", path, params=None, json_body=body) + + def call( + self, + path: str, + *, + params: dict[str, Any] | None = None, + body: dict[str, Any] | None = None, + ) -> dict[str, Any]: + """ + Generic helper that auto-routes to GET or POST based on the catalog. + + For GET endpoints the supplied `params` become query string entries; for + POST endpoints both `params` and `body` get merged into the JSON body + (body wins on conflict). + """ + info = self.endpoint_info(path) + if info is None: + raise ValueError( + f"Unknown Surf endpoint: {path!r}. " + f"Try SurfClient.endpoints() to list available paths." + ) + if info["method"] == "GET": + merged_params = {**(params or {}), **(body or {})} + return self.get(path, merged_params or None) + merged_body = {**(params or {}), **(body or {})} + return self.post(path, merged_body or None) + + # ---------------------------------------------------------------- Internals + + def _validate_path(self, path: str, method: str, supplied: dict[str, Any]) -> None: + info = self.endpoint_info(path) + if info is None: + return # allow forward-compat with newer backend endpoints + if info["method"] != method: + raise ValueError( + f"Surf endpoint {path!r} requires method {info['method']}, got {method}" + ) + missing = [p for p in info["required_params"] if p not in supplied] + if missing: + raise ValueError(f"Surf endpoint {path!r} is missing required params: {missing}") + + def _request( + self, + method: str, + path: str, + *, + params: dict[str, Any] | None, + json_body: dict[str, Any] | None, + ) -> dict[str, Any]: + url = f"{self.api_url}/v1/surf/{path}" + headers = {"Content-Type": "application/json"} if json_body is not None else {} + response = self._client.request( + method, + url, + params=params, + json=json_body, + headers=headers, + ) + if response.status_code == 402: + # Account rail: a 402 is the account being out of credit, not a + # challenge to sign. Nothing here can sign, so say so plainly. + raise_for_api_key_402(response, self.api_key) + return self._handle_payment_and_retry(method, url, params, json_body, response) + return self._unwrap(response) + + def _handle_payment_and_retry( + self, + method: str, + url: str, + params: dict[str, Any] | None, + json_body: dict[str, Any] | None, + response: httpx.Response, + ) -> dict[str, Any]: + payment_header: Any = response.headers.get("payment-required") + if not payment_header: + try: + resp_body = response.json() + if "x402" in resp_body or "accepts" in resp_body: + payment_header = resp_body + except Exception: + pass + if not payment_header: + raise PaymentError("402 response but no payment requirements found") + + if isinstance(payment_header, str): + payment_required = parse_payment_required(payment_header) + else: + payment_required = payment_header + + details = extract_payment_details(payment_required) + resource = details.get("resource") or {} + extensions = payment_required.get("extensions", {}) + + payment_payload = create_payment_payload( + account=self.account, + recipient=details["recipient"], + amount=details["amount"], + network=details.get("network", "eip155:8453"), + resource_url=resource.get("url", url), + resource_description=resource.get("description", "BlockRun Surf"), + max_timeout_seconds=details.get("maxTimeoutSeconds", 300), + extra=details.get("extra"), + asset=details.get("asset"), + extensions=extensions, + ) + + retry_headers: dict[str, str] = {"PAYMENT-SIGNATURE": payment_payload} + if json_body is not None: + retry_headers["Content-Type"] = "application/json" + + retry = self._client.request( + method, + url, + params=params, + json=json_body, + headers=retry_headers, + ) + if retry.status_code == 402: + # Account rail: a 402 is the account being out of credit, not a + # challenge to sign. Nothing here can sign, so say so plainly. + raise_for_api_key_402(retry, self.api_key) + raise PaymentError("Payment was rejected. Check your wallet balance.") + data = self._unwrap(retry, after_payment=True) + tx_hash = retry.headers.get("x-payment-receipt") or retry.headers.get("X-Payment-Receipt") + if tx_hash and isinstance(data, dict): + data.setdefault("txHash", tx_hash) + return data + + @staticmethod + def _unwrap(response: httpx.Response, *, after_payment: bool = False) -> dict[str, Any]: + if response.status_code == 200: + return response.json() + try: + error_body = response.json() + except Exception: + error_body = {"error": "Request failed"} + prefix = paid_request_error_prefix(response.headers) if after_payment else "API error" + raise APIError( + f"{prefix}: {response.status_code}", + response.status_code, + sanitize_error_response(error_body), + retry_after=retry_after_of(response), + ) + + # ------------------------------------------------------------------ Helpers + + @property + def payment_mode(self) -> str: + """Which rail this client pays on: ``"apikey"`` or ``"wallet"``. + + Worth checking once at startup when both a key and a wallet are + configured in the environment: it is the difference between + spending credit and spending USDC.""" + return payment_mode(self) + + def get_wallet_address(self) -> str: + # No address on the account rail: payment comes from prepaid + # credit, so there is nothing to return but the empty string. + if self.api_key: + return "" + """Return the EVM wallet address used for payments.""" + return self.account.address + + def close(self) -> None: + self._client.close() + + def __enter__(self) -> Self: + return self + + def __exit__(self, exc_type, exc_val, exc_tb) -> None: + self.close() diff --git a/blockrun_llm/tx_log.py b/blockrun_llm/tx_log.py new file mode 100644 index 0000000..44ead88 --- /dev/null +++ b/blockrun_llm/tx_log.py @@ -0,0 +1,388 @@ +""" +Opt-in per-transaction log for paid BlockRun API calls. + +When a client is constructed with ``transaction_log=True`` (or with the +``BLOCKRUN_TX_LOG`` env var set), every paid call appends ONE plain-text +line to a project-local file — default ``./log/transactions.log``. The +format is designed to be eyeballable in a terminal and ``grep``-friendly:: + + 2026-05-21 15:44:46 chat anthropic/claude-sonnet-4.6 in= 3 out=4 $0.034137 0x6513d128... + +Columns (single space between blocks, two spaces between fields): + +* ``ts`` local timestamp ``YYYY-MM-DD HH:MM:SS`` +* ``endpoint`` short tag (``chat``, ``image``, ``video``, ``search``, …) +* ``model`` left-padded to 30 chars +* ``in=N`` prompt tokens (right-aligned width 5) +* ``out=N`` completion tokens +* ``$cost`` six-decimal USD ``$0.034137`` +* ``tx…`` first 10 chars of the on-chain settlement hash + ``…`` + +This log is **independent of the ``~/.blockrun/cache`` layer**: enabling +``transaction_log`` does not change the cache, the response archive, or +``cost_log.jsonl``. It just adds a clean, human-readable ledger next to +your code, with the on-chain tx hash so each row is verifiable against +the chain explorer. + +All writes are best-effort and swallow OSErrors so a read-only filesystem +can never break a paid call. +""" + +from __future__ import annotations + +import base64 +import json +import os +import time +from datetime import datetime +from pathlib import Path +from typing import Any + +DEFAULT_LOG_DIR = Path("./log") +LOG_NAME = "transactions.log" + + +# --------------------------------------------------------------------------- +# Settlement header decoding (PAYMENT-RESPONSE → on-chain dict) +# --------------------------------------------------------------------------- + +# The settlement header has two names on the wire, and the one our gateways +# actually send is NOT the one this SDK was written against. Both BlockRun +# gateways emit ``PAYMENT-RESPONSE`` (the x402 v2 spec name) and neither ever +# emits ``X-PAYMENT-RESPONSE``; reading only the legacy name decodes nothing at +# all against production. The sidecar hit exactly this and fixed it in +# blockrun-litellm 0.6.0, live-verified against a real paid call. +# +# The legacy name stays accepted: other x402 facilitators still send it, and an +# unknown header costs nothing to check. Order matters only if both are present, +# in which case the spec name wins. +_SETTLEMENT_HEADER_NAMES = ("PAYMENT-RESPONSE", "X-PAYMENT-RESPONSE") + + +def read_settlement_header(headers: Any) -> str | None: + """Pull the raw settlement header out of a response, under either name. + + Single source of truth for the header name — call sites must not hand-roll + the fallback, which is how the SDK ended up reading only the legacy name in + four separate places. Never raises: a header mapping that doesn't behave + like one yields ``None`` rather than exploding on an error path. + """ + try: + for name in _SETTLEMENT_HEADER_NAMES: + value = headers.get(name) + if value: + return value + except Exception: + return None + return None + + +def decode_settlement_header(header_value: str | None) -> dict[str, Any] | None: + """Decode a ``PAYMENT-RESPONSE`` header into a settlement dict. + + The x402 facilitator returns a base64-encoded JSON describing what + landed on chain. Field names vary by chain — EVM uses ``transaction``, + Solana uses ``signature`` — so both are normalised to ``tx_hash``. + + Returns ``None`` when the header is missing or unparseable; settlement + is informational, never load-bearing. + """ + if not header_value: + return None + try: + data = json.loads(base64.b64decode(header_value)) + except Exception: + return None + if not isinstance(data, dict): + return None + tx_hash = ( + data.get("transaction") + or data.get("txHash") + or data.get("transactionHash") + or data.get("signature") + ) + amount = data.get("amount") or data.get("value") + return { + "tx_hash": tx_hash, + "amount_micro_usdc": str(amount) if amount is not None else None, + "network": data.get("network"), + "payer": data.get("payer") or data.get("from"), + "payee": data.get("payee") or data.get("to") or data.get("recipient"), + "success": data.get("success"), + "raw": data, + } + + +def paid_request_error_prefix(headers: Any) -> str: + """Error prefix for a failed request that carried a payment header. + + This used to be the flat string "API error after payment", which reads as + *your money is gone* — usually false, and it cost real time: a 500 from an + image edit was read as a lost payment by two separate readers and reported + as real spend before anyone checked the gateway. The wording manufactured + the false alarm. + + The fix is to report only what is known, which is less than it looks: + + * settlement present → funds **did** move; say so, and name the tx. + * settlement absent → **unknown**, and it must not be read as "free". + + Absence is genuinely uninformative, in two ways that bite: + + 1. Base settles synchronously after the upstream call, so absence there + usually does mean nothing moved. Solana's paid chat path settles + *in parallel* with the upstream call and re-raises immediately + (``logChargedButFailed(...); throw primaryError``) — the response is on + the wire before settlement lands. So on the one path where the caller is + charged for a 5xx and the gateway logs ``CHARGED BUT REQUEST FAILED — + refund manually``, the error carries **no header at all**. Absence and + "you were charged" co-occur *systematically*, not by chance. + 2. A gateway could always settle and omit the header. + + Hence the hedge names the usual case without asserting it. Claiming "payment + likely not taken" would replace a false alarm with a false all-clear, on + exactly the requests that need a manual refund — the worse of the two errors + for anyone reconciling spend. + + Gated on ``tx_hash``, never on the header's ``success`` field: our gateways + hard-code ``success: true`` even when settle didn't land, so that clients + parsing the header don't surface a spurious error. A tx hash is the only + thing in there that means money moved — the gateways gate their own revenue + accounting on exactly the same field. + """ + settlement = None + try: + settlement = decode_settlement_header(read_settlement_header(headers)) + except Exception: + settlement = None + if settlement and settlement.get("tx_hash"): + return f"API error after settlement (payment SETTLED, tx {settlement['tx_hash']})" + return ( + "API error on the paid request (no settlement reported — a failed call " + "usually moves no funds, but settlement can land after the error; check " + "your wallet history before assuming nothing was charged)" + ) + + +# --------------------------------------------------------------------------- +# Path resolution +# --------------------------------------------------------------------------- + + +def _resolve_log_dir(option: bool | str | os.PathLike[str] | Path | None) -> Path | None: + """Translate the ``transaction_log=...`` constructor argument into a Path. + + ``True`` → default ``./log`` + string / Path → that path (``~`` expanded) + ``None`` → honor ``BLOCKRUN_TX_LOG`` env var (``1``/``true`` → default, + anything else → that path); env unset → disabled + ``False`` → disabled + """ + if option is None: + env = os.environ.get("BLOCKRUN_TX_LOG") + if not env: + return None + if env.strip().lower() in {"1", "true", "yes", "on"}: + return DEFAULT_LOG_DIR + return Path(env).expanduser() + if option is False: + return None + if option is True: + return DEFAULT_LOG_DIR + return Path(option).expanduser() + + +# --------------------------------------------------------------------------- +# Endpoint → short tag mapping (matches the example in the README) +# --------------------------------------------------------------------------- + + +def _endpoint_tag(endpoint: str) -> str: + """Compress an API path into the 4–6 char tag used in the log.""" + if "/v1/chat/" in endpoint: + return "chat" + if "/v1/image" in endpoint: + return "image" + if "/v1/video" in endpoint: + return "video" + if "/v1/music" in endpoint or "/v1/audio" in endpoint: + return "music" + if "/v1/search" in endpoint: + return "search" + if "/v1/voice" in endpoint: + return "voice" + if "/v1/phone" in endpoint: + return "phone" + if "/v1/surf" in endpoint: + return "surf" + if "/v1/pm/" in endpoint: + return "pm" + if "/v1/price" in endpoint: + return "price" + # Fallback: last path segment + tail = endpoint.rstrip("/").rsplit("/", 1)[-1] + return tail[:6] or "call" + + +# --------------------------------------------------------------------------- +# Token-count extraction +# --------------------------------------------------------------------------- + + +def _extract_tokens(response: Any) -> tuple[int, int]: + """Best-effort ``(prompt_tokens, completion_tokens)`` from a chat response. + + Handles both the OpenAI-shaped dict (``usage.prompt_tokens`` / + ``usage.completion_tokens``) and the pydantic ``ChatResponse`` model + used by the SDK. Returns ``(0, 0)`` when no usage is reported, which + is the right answer for image / video / search calls. + """ + if response is None: + return 0, 0 + usage: Any = None + if isinstance(response, dict): + usage = response.get("usage") + else: + usage = getattr(response, "usage", None) + if usage is None: + return 0, 0 + if isinstance(usage, dict): + prompt = usage.get("prompt_tokens") or usage.get("input_tokens") or 0 + completion = usage.get("completion_tokens") or usage.get("output_tokens") or 0 + else: + prompt = getattr(usage, "prompt_tokens", 0) or getattr(usage, "input_tokens", 0) or 0 + completion = ( + getattr(usage, "completion_tokens", 0) or getattr(usage, "output_tokens", 0) or 0 + ) + try: + return int(prompt), int(completion) + except (TypeError, ValueError): + return 0, 0 + + +# --------------------------------------------------------------------------- +# Row formatter +# --------------------------------------------------------------------------- + + +def format_row( + *, + ts: float | None = None, + endpoint: str, + model: str | None, + in_tokens: int, + out_tokens: int, + cost_usd: float, + tx_hash: str | None, +) -> str: + """Format one log row exactly like the example in the module docstring.""" + if ts is None: + ts = time.time() + when = datetime.fromtimestamp(ts).strftime("%Y-%m-%d %H:%M:%S") + tag = _endpoint_tag(endpoint) + model_str = (model or "-")[:30].ljust(30) + tx_str = f"{tx_hash[:10]}…" if tx_hash else "(no-tx)" + return ( + f"{when} {tag:<5} {model_str} " + f"in={in_tokens:>5} out={out_tokens:<3} ${cost_usd:.6f} {tx_str}" + ) + + +# --------------------------------------------------------------------------- +# Logger +# --------------------------------------------------------------------------- + + +class TransactionLogger: + """Appends one plain-text row per paid call to a project-local file. + + Construct directly to bypass the client wiring:: + + logger = TransactionLogger("./log") + logger.log( + endpoint="/v1/chat/completions", + request=body, + response=chat_response, + cost_usd=0.034137, + model="anthropic/claude-sonnet-4.6", + settlement={"tx_hash": "0x6513d128…"}, + ) + + Most callers will let ``LLMClient`` / ``SolanaLLMClient`` build one + automatically via the ``transaction_log=`` constructor argument. + """ + + def __init__(self, directory: str | os.PathLike[str] | Path = DEFAULT_LOG_DIR): + self.directory = Path(directory).expanduser() + self.path = self.directory / LOG_NAME + + def log( + self, + *, + endpoint: str, + request: dict[str, Any], + response: Any, + cost_usd: float, + model: str | None = None, + wallet: str | None = None, + network: str | None = None, + client_kind: str | None = None, + settlement: dict[str, Any] | None = None, + ) -> Path | None: + """Append one formatted row to ``./log/transactions.log``. + + Returns the log path on success, or ``None`` if the file could not + be created — logging is best-effort and never raises. + """ + try: + self.directory.mkdir(parents=True, exist_ok=True) + except OSError: + return None + + in_tokens, out_tokens = _extract_tokens(response) + tx_hash = (settlement or {}).get("tx_hash") if settlement else None + row = format_row( + ts=time.time(), + endpoint=endpoint, + model=model or (request.get("model") if isinstance(request, dict) else None), + in_tokens=in_tokens, + out_tokens=out_tokens, + cost_usd=float(cost_usd or 0.0), + tx_hash=tx_hash, + ) + + # Silence unused-arg warnings without changing the public API — the + # extra metadata is intentionally accepted for future structured + # outputs (e.g. a `.jsonl` companion behind a flag) but the text + # log keeps just what fits on one line. + del wallet, network, client_kind + + try: + with open(self.path, "a") as f: + f.write(row + "\n") + except OSError: + return None + return self.path + + def entries(self) -> list[str]: + """Return every log line as a list of strings (oldest first). + + Useful for tests and for users who want to reconcile the log + against an on-chain explorer programmatically. + """ + if not self.path.exists(): + return [] + try: + return [ + line.rstrip("\n") for line in self.path.read_text().splitlines() if line.strip() + ] + except OSError: + return [] + + +__all__ = [ + "DEFAULT_LOG_DIR", + "TransactionLogger", + "decode_settlement_header", + "format_row", +] diff --git a/blockrun_llm/types.py b/blockrun_llm/types.py index 63e5812..9b4a2c7 100644 --- a/blockrun_llm/types.py +++ b/blockrun_llm/types.py @@ -1,91 +1,1089 @@ -"""Type definitions for BlockRun LLM SDK.""" - -from typing import List, Optional, Literal -from pydantic import BaseModel - - -class ChatMessage(BaseModel): - """A single chat message.""" - - role: Literal["system", "user", "assistant"] - content: str - - -class ChatChoice(BaseModel): - """A single completion choice.""" - - index: int - message: ChatMessage - finish_reason: Optional[str] = None - - -class ChatUsage(BaseModel): - """Token usage information.""" - - prompt_tokens: int - completion_tokens: int - total_tokens: int - - -class ChatResponse(BaseModel): - """Response from chat completion.""" - - id: str - object: str = "chat.completion" - created: int - model: str - choices: List[ChatChoice] - usage: Optional[ChatUsage] = None - - -class Model(BaseModel): - """Available model information.""" - - id: str - name: str - provider: str - description: str - input_price: float # Per 1M tokens - output_price: float # Per 1M tokens - context_window: int - max_output: int - available: bool = True - - -class PaymentRequirement(BaseModel): - """x402 payment requirement.""" - - scheme: str - network: str - asset: str - amount: str - pay_to: str - max_timeout_seconds: int = 300 - - -class PaymentRequired(BaseModel): - """x402 payment required response.""" - - x402_version: int = 1 - accepts: List[PaymentRequirement] - - -class BlockrunError(Exception): - """Base exception for BlockRun SDK.""" - - pass - - -class PaymentError(BlockrunError): - """Payment-related error.""" - - pass - - -class APIError(BlockrunError): - """API-related error.""" - - def __init__(self, message: str, status_code: int, response: Optional[dict] = None): - super().__init__(message) - self.status_code = status_code - self.response = response +"""Type definitions for BlockRun LLM SDK.""" + +from typing import Any, Dict, List, Literal, Optional, Union + +from pydantic import BaseModel + + +# Tool calling types (OpenAI compatible) +class FunctionDefinition(BaseModel): + """Function definition for tool calling.""" + + name: str + description: Optional[str] = None + parameters: Optional[Dict[str, Any]] = None + strict: Optional[bool] = None + + +class Tool(BaseModel): + """Tool definition for chat completions.""" + + type: Literal["function"] = "function" + function: FunctionDefinition + + +class FunctionCall(BaseModel): + """Function call details within a tool call.""" + + name: str + arguments: str + + +class ToolCall(BaseModel): + """Tool call made by the assistant.""" + + id: str + type: Literal["function"] = "function" + function: FunctionCall + + +# Tool choice can be a string or object specifying which tool to use +ToolChoiceFunction = Dict[str, Any] # {"type": "function", "function": {"name": "..."}} +ToolChoice = Union[Literal["none", "auto", "required"], ToolChoiceFunction] + + +class ChatMessage(BaseModel): + """A single chat message. + + Passthrough: the named fields below are conveniences; any other field the + gateway forwards (e.g. ``annotations``, ``audio``, future OpenAI additions) + is preserved via ``extra = "allow"`` rather than silently dropped. + """ + + role: Literal["system", "user", "assistant", "tool"] + content: Optional[str] = None + name: Optional[str] = None # For tool messages + tool_call_id: Optional[str] = None # For tool result messages + tool_calls: Optional[List[ToolCall]] = None # For assistant messages with tool calls + # Extended fields returned by reasoning-capable upstream providers + # (DeepSeek Reasoner, Grok 4 reasoning, xAI multi-agent, etc.). + # Backend strips these from inbound requests but may forward them on the + # response side, so we accept them as optional. + reasoning_content: Optional[str] = None + thinking: Optional[str] = None + + class Config: + extra = "allow" + + +class ChatChoice(BaseModel): + """A single completion choice.""" + + index: int + message: ChatMessage + finish_reason: Optional[str] = None # OpenAI-compatible; upstreams may add new values + + class Config: + extra = "allow" + + +class ChatUsage(BaseModel): + """Token usage information.""" + + prompt_tokens: int + completion_tokens: int + total_tokens: int + num_sources_used: Optional[int] = None # xAI Live Search sources used + # Anthropic prompt caching — populated on anthropic/* models when cache + # headers are sent. Reads are cheaper; writes incur a one-time surcharge. + cache_read_input_tokens: Optional[int] = None + cache_creation_input_tokens: Optional[int] = None + # Provider-native token detail. reasoning_tokens is a subset of + # completion_tokens and must not be added again when calculating spend. + prompt_tokens_details: Optional[Dict[str, Any]] = None + completion_tokens_details: Optional[Dict[str, Any]] = None + + @property + def reasoning_tokens(self) -> Optional[int]: + """Reasoning tokens the model spent, when upstream reports them. + + Nested under ``completion_tokens_details`` in the OpenAI shape the + gateway forwards. The flat fallback matters because this class allows + extras: a payload carrying a top-level ``reasoning_tokens`` used to + reach callers through ``__getattr__``, and a property of the same name + takes precedence over that, so without the fallback this would answer + None for a number the payload demonstrably carried. + + ``bool`` is excluded deliberately — it is an ``int`` subclass, and + ``True`` is not a token count. + """ + detail = self.completion_tokens_details or {} + value = detail.get("reasoning_tokens") + if value is None: + value = (self.model_extra or {}).get("reasoning_tokens") + if isinstance(value, bool) or not isinstance(value, int) or value < 0: + return None + return value + + class Config: + extra = "allow" + + +class ChatResponse(BaseModel): + """Response from chat completion. + + Passthrough: unknown top-level fields the gateway returns (e.g. + ``system_fingerprint``, ``service_tier``, ``prompt_logprobs``) are kept via + ``extra = "allow"`` so the SDK never strips what the API sends. + """ + + id: str + object: str = "chat.completion" + created: int + model: str + choices: List[ChatChoice] + usage: Optional[ChatUsage] = None + citations: Optional[List[str]] = None # xAI Live Search citation URLs + + # Real x402 charge for THIS call, in USD — the exact amount debited from the + # wallet (0.0 for free / cached calls). This is the authoritative number to + # bill/track against; token-count × list-price estimates do NOT match it + # because the gateway price carries a per-call floor + margin. Populated by + # the client on every chat completion. ``settlement`` carries the decoded + # on-chain receipt (tx hash / micro-USDC / network) when the facilitator + # returned an X-PAYMENT-RESPONSE header. + cost_usd: Optional[float] = None + settlement: Optional[Dict[str, Any]] = None + + class Config: + extra = "allow" + + +# --------------------------------------------------------------------------- +# Streaming (SSE) chunk types — OpenAI Chat Completions chunk schema. +# +# Backend emits ``data: \n\n`` lines terminated by ``data: [DONE]\n\n``. +# First chunk's delta has ``role="assistant"``; subsequent chunks fill +# ``content``; final chunk carries ``finish_reason`` and optionally ``usage``. +# --------------------------------------------------------------------------- + + +class ChatChunkFunctionCall(BaseModel): + """Streaming function-call delta. The model sends ``name`` on the first + frame and ``arguments`` in fragments afterwards, so both are optional here — + unlike the non-stream :class:`FunctionCall` where both are required.""" + + name: Optional[str] = None + arguments: Optional[str] = None + + class Config: + extra = "allow" + + +class ChatChunkToolCall(BaseModel): + """One streaming tool-call delta. + + OpenAI streams tool calls incrementally: the first frame carries + ``index`` + ``id`` + ``function.name`` (+ empty args), later frames carry + only ``index`` + ``function.arguments`` fragments. Every field is therefore + optional. The strict non-stream :class:`ToolCall` (``id`` / ``function.name`` + / ``arguments`` all required) rejected the argument-fragment frames, which + made ``ChatCompletionChunk(**chunk)`` raise and fall back to + ``model_construct`` — leaving ``choices`` as raw dicts and crashing the + archive loop with ``'dict' object has no attribute 'delta'``. Using this + lenient type keeps streamed tool calls parsing into real objects. + """ + + index: Optional[int] = None + id: Optional[str] = None + # Kept as a free-form ``str`` (not ``Literal["function"]``) so an upstream + # that streams a non-"function" tool type can't fail validation and re-trigger + # the very ``model_construct`` fallback this lenient type exists to avoid. + type: Optional[str] = None + function: Optional[ChatChunkFunctionCall] = None + + class Config: + extra = "allow" + + +class ChatChunkDelta(BaseModel): + """Incremental ``message`` delta sent over SSE. + + Any field may be absent in a given chunk — ``role`` typically only on the + first, ``content`` on body chunks, ``tool_calls`` when the model decides + to call a tool. ``reasoning_content`` / ``thinking`` appear on + reasoning-capable upstreams. + """ + + role: Optional[Literal["system", "user", "assistant", "tool"]] = None + content: Optional[str] = None + tool_calls: Optional[List[ChatChunkToolCall]] = None + reasoning_content: Optional[str] = None + thinking: Optional[str] = None + + class Config: + extra = "allow" + + +class ChatChunkChoice(BaseModel): + """One choice within a streaming chunk.""" + + index: int + delta: ChatChunkDelta + finish_reason: Optional[str] = None # OpenAI-compatible; upstreams may add new values + + class Config: + extra = "allow" + + +class ChatCompletionChunk(BaseModel): + """A single SSE chunk emitted by ``/v1/chat/completions`` when stream=True.""" + + id: str + object: str = "chat.completion.chunk" + created: int + model: str + choices: List[ChatChunkChoice] + # Usage is populated only on the final chunk for providers that support it + # (some upstreams omit it entirely — callers must tolerate ``None``). + usage: Optional[ChatUsage] = None + citations: Optional[List[str]] = None # xAI Live Search citation URLs (final chunk only) + + class Config: + extra = "allow" + + +def stream_choice_content(choice: Any) -> Optional[str]: + """Text delta from a streaming choice, tolerant of a raw ``dict`` choice. + + A chunk that fails strict validation falls back to ``model_construct``, + which leaves nested ``choices`` as plain dicts. Defensive accessors keep the + stream-archiving loop from crashing on those (``'dict' object has no + attribute 'delta'``); a tool-call frame simply has no content and yields + ``None``. + """ + if isinstance(choice, dict): + delta = choice.get("delta") + return delta.get("content") if isinstance(delta, dict) else None + delta = getattr(choice, "delta", None) + return getattr(delta, "content", None) if delta is not None else None + + +def stream_choice_finish_reason(choice: Any) -> Optional[str]: + """``finish_reason`` from a streaming choice, tolerant of a raw dict choice.""" + if isinstance(choice, dict): + return choice.get("finish_reason") + return getattr(choice, "finish_reason", None) + + +def chunk_meta(chunk: Any) -> "tuple[Optional[str], Optional[str], Optional[int]]": + """``(id, model, created)`` of a chunk, tolerant of a ``model_construct``'d + chunk that omits required fields. + + ``model_construct`` does not populate missing required fields, so a drifted + frame that lost its top-level ``id`` yields a chunk object with no ``id`` + attribute. Reading ``chunk.id`` directly would then raise ``AttributeError`` + and crash the stream-archiving loop — the same failure class the other + accessors here guard against. ``getattr`` keeps those reads safe. + """ + return ( + getattr(chunk, "id", None), + getattr(chunk, "model", None), + getattr(chunk, "created", None), + ) + + +def chunk_usage_dict(chunk: Any) -> Optional[Dict[str, Any]]: + """``usage`` of a chunk as a dict, tolerant of a model_construct'd chunk + whose ``usage`` is a raw dict (no ``.model_dump``).""" + usage = getattr(chunk, "usage", None) + if usage is None: + return None + if isinstance(usage, dict): + return {k: v for k, v in usage.items() if v is not None} + return usage.model_dump(exclude_none=True) + + +class Model(BaseModel): + """Available model information.""" + + id: str + name: str + provider: str + description: str + input_price: float # Per 1M tokens (0 when billing_mode != "paid") + output_price: float # Per 1M tokens (0 when billing_mode != "paid") + context_window: int + max_output: int + available: bool = True + # Extended metadata surfaced by /v1/models. `billing_mode` is one of + # "paid" (per-token), "flat" (flat_price per request) or "free". + billing_mode: Optional[Literal["paid", "flat", "free"]] = None + flat_price: Optional[float] = None + categories: Optional[List[str]] = None # e.g. ["chat","reasoning","coding","vision"] + hidden: Optional[bool] = None # True for deprecated/superseded models still routable + + +class PaymentRequirement(BaseModel): + """x402 payment requirement.""" + + scheme: str + network: str + asset: str + amount: str + pay_to: str + max_timeout_seconds: int = 300 + + +class PaymentRequired(BaseModel): + """x402 payment required response.""" + + x402_version: int = 1 + accepts: List[PaymentRequirement] + + +class BlockrunError(Exception): + """Base exception for BlockRun SDK.""" + + +class RetiredEndpointError(BlockrunError): + """Raised by a helper whose upstream endpoint no longer exists. + + Kept as a raising method rather than deleted so upgrading does not break + imports or attribute access — the failure is explicit and immediate instead + of a paid round trip that returns 410/404. + """ + + +class PaymentError(BlockrunError): + """Payment-related error. + + Optionally carries ``status_code`` and ``response`` so callers and + upstream proxies can surface the gateway's real failure reason + (e.g. a Solana facilitator ``transaction_simulation_failed``) + instead of seeing only a generic SDK message. + """ + + def __init__( + self, + message: str, + *, + status_code: Optional[int] = None, + response: Optional[dict] = None, + ) -> None: + super().__init__(message) + self.status_code = status_code + self.response = response + + +class SpendLimitError(PaymentError): + """A quote exceeded a spend limit the caller configured, so it was refused. + + Raised *before* the paid request goes out, so nothing settles: the quote is + declined locally and no funds move. Subclasses :class:`PaymentError` so + existing ``except PaymentError`` handlers keep working, and so the model + fallback chain refuses it — retrying another model after declining on cost + would defeat the limit. + + ``quoted_usd`` is what the gateway asked for, ``limit_usd`` is the ceiling + that refused it, and ``scope`` is ``"call"`` or ``"session"``. + """ + + def __init__( + self, + message: str, + *, + quoted_usd: float, + limit_usd: float, + scope: str, + ) -> None: + super().__init__(message) + self.quoted_usd = quoted_usd + self.limit_usd = limit_usd + self.scope = scope + + +class APIError(BlockrunError): + """API-related error. + + ``retry_after`` carries the gateway's ``Retry-After`` header verbatim when + there was one. It is the whole mechanism by which a rate-limited caller is + told how long to wait, and it matters most on the account rail, where + limits are per key and a 429 is the normal way a busy customer is asked to + slow down. Dropping it leaves every consumer guessing or spinning against + the limit the header exists to prevent. + + Kept as the raw string the header carried rather than a parsed number: the + HTTP spec allows both a delay in seconds and an HTTP-date, and inventing a + number for the date form would be worse than handing back what arrived. + ``retry_after_seconds`` parses the common form when a caller wants one. + """ + + #: Sanitizer placeholders. Appending one of these tells the caller nothing + #: that ``HTTP 500`` did not already say. + _EMPTY_UPSTREAM = frozenset({"api request failed", "request failed", "stream request failed"}) + + def __init__( + self, + message: str, + status_code: int, + response: Optional[dict] = None, + retry_after: Optional[str] = None, + ): + self.status_code = status_code + self.response = response + self.retry_after = retry_after + super().__init__(self._with_upstream(message, response)) + + @classmethod + def _with_upstream(cls, message: str, response: Optional[dict]) -> str: + """Fold the gateway's own explanation into the message. + + Raise sites build a message from the status code and stash the + sanitized body on ``.response``, so the one line worth reading never + reached ``str(exc)``. A free-tier 429 printed as ``API error: 429`` + while the body said *"Free tier rate limit reached (30 requests/minute + per IP). Retry after 10s, or use a paid model"* — the difference + between a caller who knows what to do and one who guesses. + """ + if not isinstance(response, dict): + return message + upstream = response.get("message") + if not isinstance(upstream, str): + return message + upstream = upstream.strip() + if not upstream or upstream.lower() in cls._EMPTY_UPSTREAM: + return message + if upstream in message: + return message + return f"{message}: {upstream}" + + @classmethod + def from_response( + cls, + response: Any, + message: str, + body: Optional[dict] = None, + ) -> "APIError": + """Build from an HTTP response, keeping its ``Retry-After``. + + The one place that reads the header, so a new raise site cannot forget + it. Takes any object with ``status_code`` and ``headers`` so this + module stays free of an httpx import. + """ + return cls( + message, + response.status_code, + body, + retry_after=retry_after_of(response), + ) + + @property + def retry_after_seconds(self) -> Optional[float]: + """``retry_after`` as seconds, when it is the delay-seconds form. + + ``None`` for the HTTP-date form and for anything unparseable — a caller + that wants to sleep needs a number it can trust, and guessing one from + a date the clocks may disagree about is not that. + """ + if self.retry_after is None: + return None + try: + value = float(self.retry_after.strip()) + except (TypeError, ValueError): + return None + return value if value >= 0 else None + + +def retry_after_of(response: Any) -> Optional[str]: + """Read ``Retry-After`` off a response, tolerating one that has no headers. + + Header lookup is case-insensitive on httpx, but this also runs against test + doubles and the odd hand-built object, so a missing ``headers`` attribute + answers ``None`` instead of raising inside an error path. + """ + headers = getattr(response, "headers", None) + if headers is None: + return None + try: + value = headers.get("retry-after") or headers.get("Retry-After") + except Exception: + return None + if value is None: + return None + value = str(value).strip() + return value or None + + +# Image generation types +class ImageData(BaseModel): + """A single generated image.""" + + url: str + # When the gateway mirrors the asset to its own storage, `url` is the + # permanent blockrun-hosted URL and `source_url` is the original upstream. + # `backed_up` is True iff the mirror step succeeded. For data-URI results + # (e.g. openai/gpt-image-1) both fields are omitted. + source_url: Optional[str] = None + backed_up: Optional[bool] = None + revised_prompt: Optional[str] = None + + +class ImageResponse(BaseModel): + """Response from image generation.""" + + created: int + data: List[ImageData] + + +class ImageModel(BaseModel): + """Available image model information.""" + + id: str + name: str + provider: str + description: str + price_per_image: float + available: bool = True + + +# Music / Audio types + + +class AudioTrack(BaseModel): + """A single generated audio track.""" + + url: str + duration_seconds: Optional[float] = None + lyrics: Optional[str] = None + + +class MusicResponse(BaseModel): + """Response from music generation.""" + + created: int + model: str + data: List[AudioTrack] + txHash: Optional[str] = None + + +class AudioModel(BaseModel): + """Available audio/music model information.""" + + id: str + name: str + provider: str + description: str + price_per_track: float + max_duration_seconds: int + + +# Speech (TTS / sound effects) types + + +class SpeechAudio(BaseModel): + """A single synthesized audio clip.""" + + url: str + format: Optional[str] = None + characters: Optional[int] = None + credits: Optional[float] = None + + +class SpeechResponse(BaseModel): + """Response from speech synthesis or sound-effect generation.""" + + created: int + model: str + data: List[SpeechAudio] + txHash: Optional[str] = None + + +# Multi-chain RPC types + + +class RpcError(BaseModel): + """A JSON-RPC 2.0 error object.""" + + code: Optional[int] = None + message: Optional[str] = None + data: Optional[Any] = None + + +class RpcResponse(BaseModel): + """Response from a multi-chain JSON-RPC call (/v1/rpc/{network}). + + Standard JSON-RPC 2.0 envelope plus BlockRun gateway metadata pulled + from response headers (X-Network / X-Cache / X-Payment-Receipt). + """ + + jsonrpc: Optional[str] = None + id: Optional[Union[str, int]] = None + result: Optional[Any] = None + error: Optional[RpcError] = None + # Gateway metadata (response headers) + network: Optional[str] = None # canonical network key, e.g. "ethereum" + cache_hit: bool = False # served from the gateway's method-aware cache + tx_hash: Optional[str] = None # x402 settlement tx (single calls) + + +# Video generation types + + +class VideoClip(BaseModel): + """A single generated video clip.""" + + url: str # Permanent blockrun-hosted URL (falls back to upstream if backup fails) + source_url: Optional[str] = None # Original upstream URL (e.g. vidgen.x.ai) + duration_seconds: Optional[int] = None + request_id: Optional[str] = None # Upstream provider's request id (xAI) + backed_up: Optional[bool] = None + last_frame_url: Optional[str] = None + last_frame_backed_up: Optional[bool] = None + + +class VideoResponse(BaseModel): + """Response from video generation.""" + + created: int + model: str + data: List[VideoClip] + txHash: Optional[str] = None + + +class VideoModel(BaseModel): + """Available video model information.""" + + id: str + name: str + provider: str + description: str + price_per_second: float + default_duration_seconds: int + max_duration_seconds: int + supports_image_input: bool = False + supports_lyrics: bool + supports_instrumental: bool + available: bool = True + + +# Live Search types +class WebSearchSource(BaseModel): + """Web search source configuration.""" + + type: Literal["web"] = "web" + country: Optional[str] = None # ISO alpha-2 country code + excluded_websites: Optional[List[str]] = None # Max 5 websites + allowed_websites: Optional[List[str]] = ( + None # Max 5 websites (mutually exclusive with excluded) + ) + safe_search: bool = True + + +class XSearchSource(BaseModel): + """X/Twitter search source configuration.""" + + type: Literal["x"] = "x" + included_x_handles: Optional[List[str]] = None # Max 10 handles + excluded_x_handles: Optional[List[str]] = None # Max 10 handles + post_favorite_count: Optional[int] = None # Minimum favorites threshold + post_view_count: Optional[int] = None # Minimum views threshold + + +class NewsSearchSource(BaseModel): + """News search source configuration.""" + + type: Literal["news"] = "news" + country: Optional[str] = None # ISO alpha-2 country code + excluded_websites: Optional[List[str]] = None # Max 5 websites + allowed_websites: Optional[List[str]] = None # Max 5 websites + safe_search: bool = True + + +class RssSearchSource(BaseModel): + """RSS feed search source configuration.""" + + type: Literal["rss"] = "rss" + links: List[str] # RSS feed URLs (currently supports one) + + +SearchSource = Union[ + WebSearchSource, XSearchSource, NewsSearchSource, RssSearchSource, Dict[str, Any] +] + + +class SearchParameters(BaseModel): + """ + Live Search parameters for search-enabled models. + + Enables real-time web and X/Twitter search in chat completions. + Cost: $0.025 per source used. + + Example: + search_params = SearchParameters( + mode="on", + sources=[{"type": "x"}], # Search X/Twitter only + return_citations=True + ) + """ + + mode: Literal["off", "auto", "on"] = "auto" + sources: Optional[List[SearchSource]] = None # Default: web, news, x + return_citations: bool = True + from_date: Optional[str] = None # YYYY-MM-DD format + to_date: Optional[str] = None # YYYY-MM-DD format + max_search_results: int = 10 # Max sources (default 10, ~$0.26 with margin) + + +class SearchUsage(BaseModel): + """Search usage information from xAI Live Search.""" + + num_sources_used: Optional[int] = None + + +class CostEstimate(BaseModel): + """ + Cost estimate from dry-run request. + + Returned when dry_run=True to show expected cost before executing. + """ + + model: str + estimated_input_tokens: int + estimated_output_tokens: int + estimated_cost_usd: float + + def __str__(self) -> str: + return f"💰 Estimated cost: ${self.estimated_cost_usd:.6f} ({self.model})" + + +class SpendingReport(BaseModel): + """ + Spending report returned after each paid call. + + Shows what was spent on the current call and cumulative session total. + """ + + model: str + input_tokens: int + output_tokens: int + cost_usd: float + session_total_usd: float + session_calls: int + + def __str__(self) -> str: + return ( + f"💸 This call: ${self.cost_usd:.6f} | " + f"Session total: ${self.session_total_usd:.6f} ({self.session_calls} calls)" + ) + + +class ChatResponseWithCost(BaseModel): + """ + Chat response with spending report attached. + + The content is in response.choices[0].message.content + The spending report is in spending_report + """ + + response: ChatResponse + spending_report: SpendingReport + + @property + def content(self) -> str: + """Shortcut to get response content.""" + return self.response.choices[0].message.content + + @property + def cost(self) -> float: + """Shortcut to get cost of this call.""" + return self.spending_report.cost_usd + + +# Smart routing types (Router Core integration) +RoutingProfile = Literal["free", "eco", "auto", "premium"] +RoutingTier = Literal["SIMPLE", "MEDIUM", "COMPLEX", "REASONING"] +RoutingMethod = Literal["rules", "llm", "portfolio"] +RoutingTaskType = Literal[ + "chat", + "extraction", + "code_edit", + "code_agent", + "tool_agent", + "tool_agent_parallel", + "debug", + "reasoning", + "reasoning_mcq", + "reasoning_math", + "long_context", + "vision", +] + + +class CandidateScore(BaseModel): + """Per-candidate portfolio score breakdown, ordered with ``candidates``.""" + + model: str + score: float + quality: float + cost: float + speed: float + reliability: float + + +class RoutingDecision(BaseModel): + """Result of smart routing decision.""" + + model: str + tier: RoutingTier + confidence: float + #: "portfolio" for the default V3 strategy, "rules" for the V2 rollback and + #: the free profile. + method: RoutingMethod + reasoning: str + cost_estimate: float + baseline_cost: float + savings: float # 0-1 percentage + fallbacks: List[str] = [] # remaining models in tier order, for runtime fallback + # Router Core metadata — present when the portfolio strategy ran. + candidates: List[str] = [] # ordered, capability-eligible; candidates[0] == model + candidate_scores: List[CandidateScore] = [] + task_type: Optional[RoutingTaskType] = None + router_version: Optional[Literal["v2-rules", "v3-portfolio"]] = None + profile: Optional[Literal["auto", "eco", "premium", "agentic"]] = None + agentic_score: Optional[float] = None + + +class SmartChatCompletionResponse(BaseModel): + """ + Response from smart_chat_completion — the routed full completion. + + ``response`` is the ordinary ChatResponse (choices, usage, citations), so + tool calls and structured output work exactly as with chat_completion. + + Example: + result = client.smart_chat_completion([{"role": "user", "content": "hi"}]) + print(result.model) # the model routing picked + print(result.response.choices[0].message.content) + print(result.routing.task_type) # 'chat' + """ + + response: ChatResponse + model: str + routing: RoutingDecision + + +class SmartChatResponse(BaseModel): + """ + Response from smart_chat with routing information. + + Example: + result = client.smart_chat("What is 2+2?") + print(result.response) # '4' + print(result.model) # 'google/gemini-2.5-flash' + print(f"Saved {result.routing.savings * 100:.0f}%") + """ + + response: str + model: str + routing: RoutingDecision + + +# Standalone search response +class SearchResult(BaseModel): + """Response from standalone search endpoint.""" + + query: str + summary: str + citations: Optional[List[Dict[str, str]]] = None + sources_used: Optional[int] = None + model: Optional[str] = None + + +# Pyth-backed market data types (crypto, stocks, fx, commodity) +class PricePoint(BaseModel): + """A single latest price quote from the Pyth network.""" + + symbol: str + price: float + publish_time: Optional[int] = None # Unix seconds + confidence: Optional[float] = None + feed_id: Optional[str] = None + + class Config: + extra = "allow" + + +class PriceBar(BaseModel): + """OHLC bar in a historical price series.""" + + t: Optional[int] = None # Bar open time (unix seconds) + o: Optional[float] = None + h: Optional[float] = None + l: Optional[float] = None + c: Optional[float] = None + v: Optional[float] = None + + class Config: + extra = "allow" + + +class PriceHistoryResponse(BaseModel): + """Response from a historical price endpoint.""" + + symbol: str + resolution: Optional[str] = None + bars: List[PriceBar] = [] + + class Config: + extra = "allow" + + +class SymbolListResponse(BaseModel): + """Response from a market symbol list endpoint.""" + + symbols: List[Dict[str, Any]] = [] + count: Optional[int] = None + + class Config: + extra = "allow" + + +# Virtual Portrait enrollment types + + +class PortraitUsage(BaseModel): + """How the enrolled portrait can be used.""" + + compatible_models: List[str] = [] + how_to_use: Optional[str] = None + + class Config: + extra = "allow" + + +class PortraitSettlement(BaseModel): + """On-chain settlement of the enrollment payment.""" + + success: bool + tx_hash: Optional[str] = None + network: Optional[str] = None + + class Config: + extra = "allow" + + +class PortraitEnrollment(BaseModel): + """Response from POST /v1/portrait/enroll.""" + + object: str = "virtual_portrait" + asset_id: str # ta_xxxxxxxx — pass as real_face_asset_id on Seedance + group_id: Optional[str] = None + name: str + image_url: str + created_at: Optional[str] = None + usage: Optional[PortraitUsage] = None + price: Optional[Dict[str, Any]] = None # {amount, currency} + settlement: Optional[PortraitSettlement] = None + + class Config: + extra = "allow" + + +class PortraitListItem(BaseModel): + """One row in the wallet portrait list (GET /v1/wallet//portraits).""" + + # Upstream uses camelCase here, keep matching for transparent ingestion. + assetId: str + groupId: Optional[str] = None + name: Optional[str] = None + imageUrl: Optional[str] = None + createdAt: Optional[str] = None + enrollmentTxHash: Optional[str] = None + + class Config: + extra = "allow" + + +class PortraitList(BaseModel): + """Response from GET /v1/wallet/
/portraits.""" + + wallet: str + portraits: List[PortraitListItem] = [] + count: Optional[int] = None + + class Config: + extra = "allow" + + +# RealFace enrollment types +# +# RealFace registers a *real person's* face (vs. Virtual Portrait, which is an +# AI-generated character). Enrollment is a three-step flow: init (free) → +# the person completes a phone liveness check → enroll ($0.01 USDC). The +# resulting ta_xxxxxxxx asset id is interchangeable with a Virtual Portrait's +# on Seedance 2.0 / 2.0-fast, so RealFaceEnrollment reuses PortraitUsage and +# PortraitSettlement (identical shapes) rather than duplicating them. + + +class RealFaceInit(BaseModel): + """Response from POST /v1/realface/init (free, rate-limited).""" + + object: str = "realface.init" + group_id: str # legacy_rf_xxxx — pass to status()/enroll() + h5_link: str # URL the real person scans on their phone for liveness + status: Optional[str] = None # pending_validation | active + expires_in_seconds: Optional[int] = None # H5 session validity (~120s) + next_steps: Optional[Dict[str, Any]] = None + refreshed: Optional[bool] = None # True when re-issued for an existing group + + class Config: + extra = "allow" + + +class RealFaceStatus(BaseModel): + """Response from GET /v1/realface/status?groupId=… (free, rate-limited).""" + + object: str = "realface.status" + group_id: str + status: str # pending_validation | active | … + asset_count: Optional[int] = None + ready_to_finalize: bool = False # True once status == "active" + + class Config: + extra = "allow" + + +class RealFaceEnrollment(BaseModel): + """Response from POST /v1/realface/enroll ($0.01 USDC).""" + + object: str = "realface" + asset_id: str # ta_xxxxxxxx — pass as real_face_asset_id on Seedance + group_id: Optional[str] = None + byteplus_asset_id: Optional[str] = None + name: str + image_url: str + created_at: Optional[str] = None + usage: Optional[PortraitUsage] = None + price: Optional[Dict[str, Any]] = None # {amount, currency} + settlement: Optional[PortraitSettlement] = None + + class Config: + extra = "allow" + + +class RealFaceListItem(BaseModel): + """One row in the wallet RealFace list (GET /v1/wallet//realfaces).""" + + # Upstream uses camelCase here, keep matching for transparent ingestion. + assetId: str + groupId: Optional[str] = None + name: Optional[str] = None + imageUrl: Optional[str] = None + createdAt: Optional[str] = None + enrollmentTxHash: Optional[str] = None + byteplusAssetId: Optional[str] = None + + class Config: + extra = "allow" + + +class RealFaceList(BaseModel): + """Response from GET /v1/wallet/
/realfaces.""" + + wallet: str + realfaces: List[RealFaceListItem] = [] + count: Optional[int] = None + + class Config: + extra = "allow" diff --git a/blockrun_llm/validation.py b/blockrun_llm/validation.py index 0215f8c..0ebf5a5 100644 --- a/blockrun_llm/validation.py +++ b/blockrun_llm/validation.py @@ -1,282 +1,646 @@ -""" -Input validation and security utilities for BlockRun LLM SDK. - -This module provides validation functions to ensure: -- Private keys are properly formatted -- API URLs use HTTPS -- Parameters are within valid ranges -- Server responses don't leak sensitive information -- Resource URLs match expected domains -""" - -import re -from typing import Optional, Dict, Any -from urllib.parse import urlparse - - -# Localhost domains that are allowed to use HTTP -LOCALHOST_DOMAINS = {"localhost", "127.0.0.1"} - -# Known LLM providers (for optional validation) -KNOWN_PROVIDERS = { - "openai", - "anthropic", - "google", - "deepseek", - "mistralai", - "meta-llama", - "together", -} - - -def validate_private_key(key: str) -> None: - """ - Validate that a private key is properly formatted. - - Args: - key: The private key to validate - - Raises: - ValueError: If the key format is invalid - - Example: - >>> validate_private_key("0xac0974bec39a17e36ba4a6b4d238ff944bacb478cbed5efcae784d7bf4f2ff80") - """ - if not isinstance(key, str): - raise ValueError("Private key must be a string") - - # Must start with 0x - if not key.startswith("0x"): - raise ValueError("Private key must start with 0x") - - # Must be exactly 66 characters (0x + 64 hex chars) - if len(key) != 66: - raise ValueError( - "Private key must be 66 characters (0x + 64 hexadecimal characters)" - ) - - # Must contain only valid hexadecimal characters - if not re.match(r"^0x[0-9a-fA-F]{64}$", key): - raise ValueError( - "Private key must contain only hexadecimal characters (0-9, a-f, A-F)" - ) - - -def validate_model(model: str) -> None: - """ - Validate model ID format. - - Args: - model: The model ID (e.g., "openai/gpt-4o", "anthropic/claude-sonnet-4.5") - - Raises: - ValueError: If model is invalid - - Example: - >>> validate_model("openai/gpt-4o") - """ - if not model or not isinstance(model, str): - raise ValueError("Model must be a non-empty string") - - # Optionally validate provider (just a warning, don't fail) - if "/" in model: - provider = model.split("/", 1)[0] - if provider not in KNOWN_PROVIDERS: - # Just log, don't fail (allows new providers) - pass - - -def validate_max_tokens(max_tokens: Optional[int]) -> None: - """ - Validate max_tokens parameter. - - Args: - max_tokens: Maximum number of tokens to generate - - Raises: - ValueError: If max_tokens is invalid - - Example: - >>> validate_max_tokens(1000) - """ - if max_tokens is None: - return - - if not isinstance(max_tokens, int): - raise ValueError("max_tokens must be an integer") - - if max_tokens < 1: - raise ValueError("max_tokens must be positive (minimum: 1)") - - if max_tokens > 100000: - raise ValueError("max_tokens too large (maximum: 100000)") - - -def validate_temperature(temperature: Optional[float]) -> None: - """ - Validate temperature parameter. - - Args: - temperature: Sampling temperature (0-2) - - Raises: - ValueError: If temperature is invalid - - Example: - >>> validate_temperature(0.7) - """ - if temperature is None: - return - - if not isinstance(temperature, (int, float)): - raise ValueError("temperature must be a number") - - if temperature < 0 or temperature > 2: - raise ValueError("temperature must be between 0 and 2") - - -def validate_top_p(top_p: Optional[float]) -> None: - """ - Validate top_p parameter (nucleus sampling). - - Args: - top_p: Top-p sampling parameter (0-1) - - Raises: - ValueError: If top_p is invalid - - Example: - >>> validate_top_p(0.9) - """ - if top_p is None: - return - - if not isinstance(top_p, (int, float)): - raise ValueError("top_p must be a number") - - if top_p < 0 or top_p > 1: - raise ValueError("top_p must be between 0 and 1") - - -def validate_api_url(url: str) -> None: - """ - Validate that an API URL is secure and properly formatted. - - Args: - url: The API URL to validate - - Raises: - ValueError: If the URL is invalid or insecure - - Example: - >>> validate_api_url("https://api.blockrun.ai") - >>> validate_api_url("http://localhost:3000") # OK for development - """ - try: - parsed = urlparse(url) - except Exception as e: - raise ValueError(f"Invalid API URL: {e}") - - if not parsed.scheme: - raise ValueError("API URL must include scheme (http:// or https://)") - - if not parsed.netloc: - raise ValueError("API URL must include domain") - - # Require HTTPS for non-localhost URLs - is_localhost = parsed.netloc.split(":")[0] in LOCALHOST_DOMAINS - - if parsed.scheme != "https" and not is_localhost: - raise ValueError( - "API URL must use HTTPS for non-localhost endpoints. " - f"Use https:// instead of {parsed.scheme}://" - ) - - -def sanitize_error_response(error_body: Any) -> Dict[str, Any]: - """ - Sanitize API error responses to prevent information leakage. - - Only exposes safe error fields to the caller, filtering out: - - Internal stack traces - - Server-side paths - - API keys or tokens - - Debugging information - - Args: - error_body: The raw error response from the API - - Returns: - Sanitized error dict with only safe fields - - Example: - >>> sanitize_error_response({ - ... "error": "Invalid model", - ... "internal_stack": "/var/app/handler.py:123", - ... "api_key": "secret" - ... }) - {'message': 'Invalid model', 'code': None} - """ - if not isinstance(error_body, dict): - return {"message": "API request failed", "code": None} - - # Only expose safe fields - return { - "message": ( - error_body.get("error") - if isinstance(error_body.get("error"), str) - else "API request failed" - ), - "code": ( - error_body.get("code") - if isinstance(error_body.get("code"), str) - else None - ), - } - - -def validate_resource_url(url: str, base_url: str) -> str: - """ - Validate a resource URL from the server to prevent redirection attacks. - - Ensures that the resource URL's hostname matches the API's hostname. - If domains don't match, returns a safe default URL instead. - - Args: - url: The resource URL provided by the server - base_url: The base API URL (trusted) - - Returns: - The validated URL or a safe default - - Example: - >>> validate_resource_url( - ... "https://api.blockrun.ai/v1/chat", - ... "https://api.blockrun.ai" - ... ) - 'https://api.blockrun.ai/v1/chat' - - >>> validate_resource_url( - ... "https://malicious.com/steal", - ... "https://api.blockrun.ai" - ... ) - 'https://api.blockrun.ai/v1/chat/completions' - """ - try: - parsed = urlparse(url) - base_parsed = urlparse(base_url) - - # Resource URL hostname must match API hostname - if parsed.netloc != base_parsed.netloc: - # Return safe default - return f"{base_url}/v1/chat/completions" - - # Ensure resource uses same protocol as base - if parsed.scheme != base_parsed.scheme: - return f"{base_url}/v1/chat/completions" - - return url - - except Exception: - # Invalid URL format, return safe default - return f"{base_url}/v1/chat/completions" +""" +Input validation and security utilities for BlockRun LLM SDK. + +This module provides validation functions to ensure: +- Private keys are properly formatted +- API URLs use HTTPS +- Parameters are within valid ranges +- Server responses don't leak sensitive information +- Resource URLs match expected domains +""" + +from __future__ import annotations + +import re +from typing import TYPE_CHECKING, Any, NoReturn +from urllib.parse import urlparse + +if TYPE_CHECKING: + from .types import PaymentError + + +# Localhost domains that are allowed to use HTTP +LOCALHOST_DOMAINS = {"localhost", "127.0.0.1"} + +# Known LLM providers (for optional validation) +KNOWN_PROVIDERS = { + "openai", + "anthropic", + "google", + "deepseek", + "mistralai", + "meta-llama", + "together", + "xai", + "moonshot", + "nvidia", + "minimax", + "zai", +} + +# Seed modes a caller may assert via `input_type` on /v1/videos/generations. +# Mirrors the gateway enum; the gateway stays the authority on whether the +# declared mode matches the seed fields actually sent. +VIDEO_INPUT_TYPES = ("text", "image", "first_last_frame", "reference") + +# Latency/fidelity levels for `quality` on Solana image generation + editing. +# Mirrors the gateway enum, which accepts the field for openai/gpt-image-* only. +IMAGE_QUALITY_LEVELS = ("low", "medium", "high", "auto") + + +# Base58 alphabet characters that never appear in a hex string. Their presence +# is a strong signal that a key is a base58-encoded Solana key, not an EVM key. +_BASE58_ONLY_CHARS = frozenset("GHJKLMNPQRSTUVWXYZghijkmnopqrstuvwxyz") + + +def _looks_like_solana_key(key: str) -> bool: + """ + Heuristically detect a base58-encoded Solana secret key. + + Solana secret keys are base58, not hex: a 32-byte seed is ~43-44 chars and a + 64-byte keypair is ~87-88 chars. An EVM key is exactly 64 hex chars (sans the + ``0x`` prefix). We treat a key as Solana when it contains a base58-only + character (one absent from the hex alphabet) and its length is outside the + EVM 64-char range — so a malformed 64-char hex key still routes to the + regular hex error rather than the Solana hint. + """ + candidate = key.removeprefix("0x") + if len(candidate) == 64 or not (40 <= len(candidate) <= 90): + return False + return any(c in _BASE58_ONLY_CHARS for c in candidate) + + +# bool is a subclass of int in Python, so `isinstance(True, int)` is True and a +# bare numeric type check lets booleans straight through. Before this was fixed, +# validate_max_tokens(True), validate_temperature(True) and validate_top_p(False) +# all passed — the value then serialized as JSON `true`/`false` and went to the +# gateway as a request parameter. `max_tokens=False` was caught, but by the +# positivity check, so a type error was reported as a range error. +# +# Every numeric validator below excludes bool explicitly. Keep it that way when +# adding one. + + +def validate_private_key(key: str) -> None: + """ + Validate that a private key is properly formatted. + + Args: + key: The private key to validate + + Raises: + ValueError: If the key format is invalid + + Example: + >>> validate_private_key("0xac0974bec39a17e36ba4a6b4d238ff944bacb478cbed5efcae784d7bf4f2ff80") + """ + if not isinstance(key, str): + raise ValueError("Private key must be a string") + + # Detect a base58 Solana key fed into the EVM (Base) client and point the + # user at the right entry point instead of the cryptic "66 characters" error. + if _looks_like_solana_key(key): + raise ValueError( + "This looks like a Solana (base58) private key, but this client uses " + "the Base (EVM) chain. Use the Solana client instead:\n" + " from blockrun_llm import SolanaLLMClient\n" + ' client = SolanaLLMClient(private_key="")\n' + "Or for agent use:\n" + " from blockrun_llm import setup_agent_solana_wallet\n" + " client = setup_agent_solana_wallet()\n" + 'Install Solana support first: pip install "blockrun-llm[solana]"' + ) + + # Must start with 0x + if not key.startswith("0x"): + raise ValueError("Private key must start with 0x") + + # Must be exactly 66 characters (0x + 64 hex chars) + if len(key) != 66: + raise ValueError("Private key must be 66 characters (0x + 64 hexadecimal characters)") + + # Must contain only valid hexadecimal characters + if not re.match(r"^0x[0-9a-fA-F]{64}$", key): + raise ValueError("Private key must contain only hexadecimal characters (0-9, a-f, A-F)") + + +def validate_eth_address(address: str) -> None: + """ + Validate that a value is a well-formed Ethereum / Base address. + + Args: + address: The 0x-prefixed 20-byte address to validate + + Raises: + ValueError: If the address format is invalid + + Example: + >>> validate_eth_address("0x036CbD53842c5426634e7929541eC2318f3dCF7e") + """ + if not isinstance(address, str): + raise ValueError("Address must be a string") + + # Must be a 0x-prefixed 40-character hexadecimal string + if not re.match(r"^0x[0-9a-fA-F]{40}$", address): + raise ValueError("Address must be a 0x-prefixed 40-character hexadecimal string") + + +def validate_model(model: str) -> None: + """ + Validate model ID format. + + Args: + model: The model ID (e.g., "openai/gpt-5.2", "anthropic/claude-sonnet-4.5") + + Raises: + ValueError: If model is invalid + + Example: + >>> validate_model("openai/gpt-5.2") + """ + if not model or not isinstance(model, str): + raise ValueError("Model must be a non-empty string") + + # Optionally validate provider (just a warning, don't fail) + if "/" in model: + provider = model.split("/", 1)[0] + if provider not in KNOWN_PROVIDERS: + # Just log, don't fail (allows new providers) + pass + + +def validate_video_input_type(input_type: str | None) -> None: + """ + Validate the optional `input_type` seed-mode assertion on video generation. + + Only the spelling is checked. Whether the declared mode agrees with the + seed fields actually sent is the gateway's call — it infers the mode and + rejects with 400 *before* charging, so re-deriving that inference here + would add a second copy to keep in sync for no benefit. + + Args: + input_type: One of VIDEO_INPUT_TYPES, or None to leave it unset. + + Raises: + ValueError: If input_type is not one of the accepted values. + + Example: + >>> validate_video_input_type("first_last_frame") + """ + if input_type is None: + return + if input_type not in VIDEO_INPUT_TYPES: + raise ValueError( + f"input_type must be one of {', '.join(VIDEO_INPUT_TYPES)}; got {input_type!r}." + ) + + +def validate_image_quality(quality: str | None) -> None: + """ + Validate the optional `quality` knob on Solana image generation/editing. + + Model compatibility is left to the gateway, which accepts `quality` only + for openai/gpt-image-* and returns a clear error otherwise — encoding that + model list here would go stale every time the catalog changes. + + Args: + quality: One of IMAGE_QUALITY_LEVELS, or None to leave it unset. + + Raises: + ValueError: If quality is not one of the accepted values. + + Example: + >>> validate_image_quality("low") + """ + if quality is None: + return + if quality not in IMAGE_QUALITY_LEVELS: + raise ValueError( + f"quality must be one of {', '.join(IMAGE_QUALITY_LEVELS)}; got {quality!r}." + ) + + +# Client-side typo guard, NOT a model limit. +# +# This was 100000, which sat below what models actually serve: zai/glm-5.2 +# serves 262144 and the common ceiling is 128000, so the SDK — not the model — +# was the binding constraint, and callers got a ValueError naming a limit no +# provider had set. +# +# The gateway does NOT reject an over-ceiling max_tokens. It silently clamps to +# the model's ceiling and quotes payment for the clamped value (probed against +# the live 402 leg 2026-07-21: opus-4.8 sent 262144 and 1000000 both quote the +# 128000 price; gpt-5.2 sent 1e12 returns a quote, not a 400). So there is no +# server-side rejection to fall back on — whatever passes here gets priced, and +# anything above the model's ceiling is money spent on tokens you won't get. +# ``LLMClient`` warns when it sees the gateway clamp; see ``_warn_if_clamped``. +# +# Keep a bound so an obvious mistake (1e9, a byte count, a timestamp) fails +# fast locally instead of becoming a payment quote. Set it far above any real +# model so it can never be the binding constraint again. +MAX_TOKENS_SANITY_LIMIT = 1_000_000 + + +def validate_max_tokens(max_tokens: int | None) -> None: + """ + Validate max_tokens parameter. + + Rejects only values no request could have meant. The gateway does not + reject an over-ceiling ``max_tokens`` — it clamps to the model's own + ceiling and charges for the clamped value — so this guard exists to stop a + typo locally, not to enforce any model's limit. + + Args: + max_tokens: Maximum number of tokens to generate + + Raises: + ValueError: If max_tokens is invalid + + Example: + >>> validate_max_tokens(1000) + """ + if max_tokens is None: + return + + # bool is an int subclass, so `isinstance(True, int)` is True and a stray + # flag threaded into the wrong keyword would sail through and reach the + # wire as `"max_tokens": true`. Say so explicitly — "must be an integer" + # reads as wrong to anyone who knows bool is one. + if isinstance(max_tokens, bool): + raise ValueError("max_tokens must be an integer, got a bool") + + if not isinstance(max_tokens, int): + raise ValueError("max_tokens must be an integer") + + if max_tokens < 1: + raise ValueError("max_tokens must be positive (minimum: 1)") + + if max_tokens > MAX_TOKENS_SANITY_LIMIT: + raise ValueError( + f"max_tokens implausibly large (client-side sanity limit: " + f"{MAX_TOKENS_SANITY_LIMIT}). This is not a model limit — no " + f"provider set it. Anything under it is sent to the gateway, " + f"which clamps to the model's own ceiling and charges for the " + f"clamped value rather than rejecting." + ) + + +def validate_temperature(temperature: float | None) -> None: + """ + Validate temperature parameter. + + Args: + temperature: Sampling temperature (0-2) + + Raises: + ValueError: If temperature is invalid + + Example: + >>> validate_temperature(0.7) + """ + if temperature is None: + return + + if isinstance(temperature, bool): + raise ValueError("temperature must be a number, got a bool") + + if not isinstance(temperature, (int, float)): + raise ValueError("temperature must be a number") + + if temperature < 0 or temperature > 2: + raise ValueError("temperature must be between 0 and 2") + + +def validate_top_p(top_p: float | None) -> None: + """ + Validate top_p parameter (nucleus sampling). + + Args: + top_p: Top-p sampling parameter (0-1) + + Raises: + ValueError: If top_p is invalid + + Example: + >>> validate_top_p(0.9) + """ + if top_p is None: + return + + if isinstance(top_p, bool): + raise ValueError("top_p must be a number, got a bool") + + if not isinstance(top_p, (int, float)): + raise ValueError("top_p must be a number") + + if top_p < 0 or top_p > 1: + raise ValueError("top_p must be between 0 and 1") + + +def validate_api_url(url: str) -> None: + """ + Validate that an API URL is secure and properly formatted. + + Args: + url: The API URL to validate + + Raises: + ValueError: If the URL is invalid or insecure + + Example: + >>> validate_api_url("https://blockrun.ai/api") + >>> validate_api_url("http://localhost:3000") # OK for development + """ + try: + parsed = urlparse(url) + except Exception as e: + raise ValueError(f"Invalid API URL: {e}") + + if not parsed.scheme: + raise ValueError("API URL must include scheme (http:// or https://)") + + if not parsed.netloc: + raise ValueError("API URL must include domain") + + # Require HTTPS for non-localhost URLs + is_localhost = parsed.netloc.split(":")[0] in LOCALHOST_DOMAINS + + if parsed.scheme != "https" and not is_localhost: + raise ValueError( + "API URL must use HTTPS for non-localhost endpoints. " + f"Use https:// instead of {parsed.scheme}://" + ) + + +def build_payment_rejected_error(response: Any) -> PaymentError: + """Translate a 402 retry response into a :class:`PaymentError` that + preserves the gateway's original failure reason. + + Without this helper, clients used to throw a generic + ``"Payment rejected. Check your wallet balance."`` and the real + facilitator reason (e.g. ``transaction_simulation_failed``, + ``insufficient_funds``) was lost. + + The gateway's ``details`` field on a 402 settlement-failed response + is the x402 facilitator's well-defined error enum — safe to surface + verbatim. We bound the length defensively in case a future server + bug widens the field. + + Args: + response: An ``httpx.Response`` with status 402 from a paid + retry. Anything with a ``.json()`` method works for tests. + + Returns: + A :class:`PaymentError` carrying ``status_code=402`` and a + ``response`` dict that includes the gateway's ``details``. + """ + # Local import to avoid a circular module dependency at import time. + from .types import PaymentError + + try: + body = response.json() + except Exception: + body = {} + if not isinstance(body, dict): + body = {} + sanitized = dict(sanitize_error_response(body)) + raw_details = body.get("details") + if isinstance(raw_details, str) and 0 < len(raw_details) < 256: + sanitized["details"] = raw_details + # The x402 facilitator's `invalidMessage` — the simulation-level cause that + # the coarse `invalidReason` enum collapses away (an unfunded wallet and a + # stale blockhash both arrive as transaction_simulation_failed). Same + # provenance and safety rationale as `details` above: a facilitator error + # string, not upstream text, so it's safe to surface verbatim — bounded + # defensively all the same. Still folded into the message for human-readable + # errors and for `str(exc)` consumers; `_is_safe_resign_error` reads the + # structured `exc.response` keys directly. + raw_invalid_message = body.get("invalidMessage") + if isinstance(raw_invalid_message, str) and 0 < len(raw_invalid_message) < 256: + sanitized["invalidMessage"] = raw_invalid_message + # Machine-readable payment classification used by the Solana client's + # re-sign phase gate (:func:`solana_client._is_safe_resign_error`). These are + # gateway-owned enums, not raw upstream text. Preserve only short strings; + # debug remains intentionally filtered. + raw_reason = body.get("reason") + if isinstance(raw_reason, str) and 0 < len(raw_reason) < 128: + sanitized["reason"] = raw_reason + raw_code = body.get("code") + if isinstance(raw_code, str) and 0 < len(raw_code) < 128: + sanitized["code"] = raw_code + detail_part = sanitized.get("details") or sanitized.get("message") or "" + invalid_message = sanitized.get("invalidMessage") + if invalid_message: + detail_part = f"{detail_part} ({invalid_message})" if detail_part else invalid_message + msg = ( + f"Payment rejected by gateway: {detail_part}" + if detail_part + else "Payment rejected by gateway" + ) + return PaymentError(msg, status_code=402, response=sanitized) + + +def sanitize_error_response(error_body: Any) -> dict[str, Any]: + """ + Sanitize API error responses to prevent information leakage. + + Only exposes safe error fields to the caller, filtering out: + - Internal stack traces + - Server-side paths + - API keys or tokens + - Debugging information + + Args: + error_body: The raw error response from the API + + Returns: + Sanitized error dict with only safe fields + + Example: + >>> sanitize_error_response({ + ... "error": "Invalid model", + ... "internal_stack": "/var/app/handler.py:123", + ... "api_key": "secret" + ... }) + {'message': 'Invalid model', 'code': None} + """ + if not isinstance(error_body, dict): + return {"message": "API request failed", "code": None} + + # The gateway returns OpenAI-compatible *nested* errors: + # {"error": {"message", "type", "code", "param"}, "message", "code", "debug"} + # while older endpoints (and the SDK's own fallbacks) still use the *flat* shape: + # {"error": "Request failed", "code": "..."} + # Pass the real message/code through for either shape. Never surface `debug` + # (raw upstream error text — may leak internal paths/keys). + nested = error_body.get("error") + + if isinstance(nested, dict): + message = nested.get("message") + code = nested.get("code") or error_body.get("code") + result: dict[str, Any] = { + "message": message if isinstance(message, str) else "API request failed", + "code": code if isinstance(code, str) else None, + } + # Pass through OpenAI error metadata when present. + if isinstance(nested.get("type"), str): + result["type"] = nested["type"] + if isinstance(nested.get("param"), str): + result["param"] = nested["param"] + return result + + # Flat shape: `error` is the human-readable title; fall back to top-level `message`. + if isinstance(nested, str): + message = nested + elif isinstance(error_body.get("message"), str): + message = error_body["message"] + else: + message = "API request failed" + + return { + "message": message, + "code": (error_body.get("code") if isinstance(error_body.get("code"), str) else None), + } + + +def validate_resource_url(url: str, base_url: str) -> str: + """ + Validate a resource URL from the server to prevent redirection attacks. + + Ensures that the resource URL's hostname matches the API's hostname. + If domains don't match, returns a safe default URL instead. + + Args: + url: The resource URL provided by the server + base_url: The base API URL (trusted) + + Returns: + The validated URL or a safe default + + Example: + >>> validate_resource_url( + ... "https://blockrun.ai/api/v1/chat", + ... "https://blockrun.ai/api" + ... ) + 'https://blockrun.ai/api/v1/chat' + + >>> validate_resource_url( + ... "https://malicious.com/steal", + ... "https://blockrun.ai/api" + ... ) + 'https://blockrun.ai/api/v1/chat/completions' + """ + try: + parsed = urlparse(url) + base_parsed = urlparse(base_url) + + # Resource URL hostname must match API hostname + if parsed.netloc != base_parsed.netloc: + # Return safe default + return f"{base_url}/v1/chat/completions" + + # Ensure resource uses same protocol as base + if parsed.scheme != base_parsed.scheme: + return f"{base_url}/v1/chat/completions" + + return url + + except Exception: + # Invalid URL format, return safe default + return f"{base_url}/v1/chat/completions" + + +def resolve_spend_limit(explicit: float | None, env_var: str) -> float | None: + """Resolve a spend limit from the constructor argument or its env var. + + ``None`` means unlimited, which is the default and the pre-1.9.0 behavior. + An unparseable or non-positive env value is ignored rather than raising: + a malformed env var must not brick every client in a deployment, and the + explicit argument always wins. + """ + if explicit is not None: + limit = float(explicit) + if limit <= 0: + raise ValueError(f"spend limit must be positive; got {explicit!r}") + return limit + + import os + + raw = os.environ.get(env_var) + if not raw: + return None + try: + limit = float(raw) + except ValueError: + return None + return limit if limit > 0 else None + + +def check_spend_limits( + cost_usd: float, + *, + max_cost_per_call: float | None, + max_session_cost: float | None, + session_spent_usd: float, + model: str | None = None, +) -> None: + """Refuse a quote that would breach a caller-configured spend limit. + + Call this after the gateway's price is known and BEFORE the paid request is + sent. Signing alone moves no money — the gateway submitting the signed + authorization does — so declining here means nothing settles. + + Both limits are opt-in. With neither set this is a no-op, which is why + adding it changes no existing behavior. + + Raises: + SpendLimitError: If the quote exceeds the per-call limit, or if it would + push the session past its total. + """ + from .types import SpendLimitError + + where = f" for {model}" if model else "" + + if max_cost_per_call is not None and cost_usd > max_cost_per_call: + raise SpendLimitError( + f"Refused a ${cost_usd:.6f} quote{where}: it exceeds the per-call " + f"limit of ${max_cost_per_call:.6f}. Nothing was sent and nothing " + f"was charged. Raise max_cost_per_call to allow it.", + quoted_usd=cost_usd, + limit_usd=max_cost_per_call, + scope="call", + ) + + if max_session_cost is not None and session_spent_usd + cost_usd > max_session_cost: + remaining = max_session_cost - session_spent_usd + raise SpendLimitError( + f"Refused a ${cost_usd:.6f} quote{where}: this client has spent " + f"${session_spent_usd:.6f} of its ${max_session_cost:.6f} session " + f"limit, leaving ${remaining:.6f}. Nothing was sent and nothing was " + f"charged.", + quoted_usd=cost_usd, + limit_usd=max_session_cost, + scope="session", + ) + + +def raise_api_error(resp: Any, prefix: str) -> NoReturn: + """Turn a failed HTTP response into an ``APIError``, keeping ``Retry-After``. + + Three clients carried byte-identical private copies of this. One copy means + a change to how failures are reported — such as keeping the rate-limit + header — lands everywhere at once instead of in two places out of three. + """ + # Local, like build_payment_rejected_error below: types.py is imported by + # every client, and importing it at module scope here would close a cycle. + from .types import APIError + + try: + error_body = resp.json() + except Exception: + error_body = {"error": "Request failed"} + raise APIError.from_response( + resp, + f"{prefix}: HTTP {resp.status_code}", + sanitize_error_response(error_body), + ) diff --git a/blockrun_llm/video.py b/blockrun_llm/video.py new file mode 100644 index 0000000..9e22471 --- /dev/null +++ b/blockrun_llm/video.py @@ -0,0 +1,623 @@ +""" +BlockRun Video Client - Generate short AI videos via x402 micropayments. + +SECURITY NOTE - Private Key Handling: +===================================== +Your private key NEVER leaves your machine. Here's what happens: + +1. Key stays local - only used to sign an EIP-712 typed data message +2. Only the SIGNATURE is sent in the PAYMENT-SIGNATURE header +3. BlockRun verifies the signature on-chain via Coinbase CDP facilitator + +Async flow (client-polled): + POST /v1/videos/generations -> 402 -> sign -> 202 { id, poll_url } + GET /v1/videos/generations/{id} -> loop until status=completed + +The client signs once and replays the same PAYMENT-SIGNATURE on every poll, +re-signing automatically if the 600s authorization window lapses mid-poll. +Settlement happens only on the first completed poll, so upstream failure or +the caller giving up = zero charge. + +Usage: + from blockrun_llm import VideoClient + + client = VideoClient() # Uses BLOCKRUN_WALLET_KEY from env + + result = client.generate("a red apple slowly spinning on a wooden table") + print(result.data[0].url) # permanent blockrun-hosted MP4 URL + print(result.data[0].duration_seconds) +""" + +from __future__ import annotations + +import os +import time +from typing import Any + +import httpx +from dotenv import load_dotenv +from eth_account import Account + +from .apikey import ( + api_key_base_url, + auth_headers, + missing_credential_error, + payment_mode, + raise_for_api_key_402, + resolve_api_key, + resolve_poll_url, +) +from .types import APIError, PaymentError, VideoResponse, retry_after_of +from .validation import ( + raise_api_error, + sanitize_error_response, + validate_api_url, + validate_private_key, + validate_video_input_type, +) +from .x402 import create_payment_payload, extract_payment_details, parse_payment_required + +load_dotenv() + + +class VideoClient: + """ + BlockRun Video Generation Client. + + Supports xAI Grok Imagine Video and ByteDance Seedance (1.5 Pro / + 2.0 Fast / 2.0 Pro) with automatic x402 micropayments on Base. + + Pricing (approx. 5s 720p clip): + xai/grok-imagine-video $0.050/sec (8s default → ~$0.40) + bytedance/seedance-1.5-pro $4.32/M tok flat (~$0.46 / 5s) + bytedance/seedance-2.0-fast $11.20/M text or $6.60/M image (~$1.19 / $0.70 / 5s) + bytedance/seedance-2.0 $14.00/M text or $8.60/M image (~$1.49 / $0.91 / 5s) + + Seedance 2.0 fast/pro additionally accept `real_face_asset_id` — + a `ta_xxxxxx` face/character asset for consistency across multiple + videos. The asset can be either: + - a Virtual Portrait (AI-generated character) enrolled via + `PortraitClient` / `POST /v1/portrait/enroll` ($0.01 USDC), or + - a RealFace (a real person's likeness) enrolled via + `RealFaceClient` / `POST /v1/realface/enroll` ($0.01 USDC, no + KYC — just a brief on-phone liveness check). + Both flows return the same `ta_` id. seedance-1.5-pro does NOT + support these assets. Mutually exclusive with `image_url`. + Resolution and generate_audio can be overridden per call. Returned + URLs are permanent (mirrored to BlockRun storage). + """ + + DEFAULT_API_URL = "https://blockrun.ai/api" + DEFAULT_MODEL = "xai/grok-imagine-video" + DEFAULT_TIMEOUT = 360.0 # per-HTTP-call timeout (submit / each poll) + POLL_INTERVAL_SECONDS = 5.0 + # 15 min: generation itself is 1-3 min, but the upstream pipeline can lag + # the status read-path several minutes behind actual completion (observed: + # video done in 100s, status flipped ~7.5min later). Jobs stay claimable + # ~48h, so a patient default beats a premature give-up. + DEFAULT_GENERATE_BUDGET_SECONDS = 900.0 + # Advertised signed-auth window. Server-side default is 300s; we bump to + # 600s so the signature stays valid across the async polling window. + # Budgets longer than this window are handled by re-signing mid-poll. + MAX_TIMEOUT_SECONDS = 600 + # Max mid-poll re-signs after a 402 (signature expiry). A fresh signature + # that 402s again means a genuine payment problem, not expiry. + MAX_POLL_RESIGNS = 2 + + def __init__( + self, + private_key: str | None = None, + api_url: str | None = None, + timeout: float = 360.0, + ): + """ + Initialize the BlockRun Video client. + + Args: + private_key: EVM wallet private key (or set BLOCKRUN_WALLET_KEY env var) + api_url: API endpoint URL (default: https://blockrun.ai/api) + timeout: Per-HTTP-call timeout in seconds (submit+each poll). + """ + from .wallet import load_wallet + + # Account rail first, and before any wallet variable is read: an API + # key is a complete credential on its own, so demanding a private key + # alongside it would make every API-key user invent a wallet they + # never use. See apikey.py for the precedence rule. + api_key = resolve_api_key(private_key) + key = ( + None + if api_key + else ( + private_key + or os.environ.get("BLOCKRUN_WALLET_KEY") + or os.environ.get("BASE_CHAIN_WALLET_KEY") + or load_wallet() + ) + ) + if not api_key and not key: + raise missing_credential_error() + + if key: + validate_private_key(key) + self.api_key = api_key + # No wallet on the account rail: nothing is signed locally. + self.account = Account.from_key(key) if key else None + + # BLOCKRUN_API_URL names an x402 gateway; an API-key client must not + # follow it and hand the key to a host set up for another rail. + api_url_raw = ( + api_key_base_url(api_url) + if api_key + else (api_url or os.environ.get("BLOCKRUN_API_URL") or self.DEFAULT_API_URL) + ) + validate_api_url(api_url_raw) + self.api_url = api_url_raw.rstrip("/") + + self.timeout = timeout + self._client = httpx.Client(headers=auth_headers(api_key), timeout=timeout) + + def generate( + self, + prompt: str, + *, + model: str | None = None, + image_url: str | None = None, + last_frame_url: str | None = None, + reference_image_urls: list[str] | None = None, + reference_videos: list[dict[str, str]] | None = None, + reference_audios: list[dict[str, str]] | None = None, + bitrate_mode: str | None = None, + output_format: str | None = None, + camera_fixed: bool | None = None, + safety_identifier: str | None = None, + real_face_asset_id: str | None = None, + duration_seconds: int | None = None, + aspect_ratio: str | None = None, + resolution: str | None = None, + generate_audio: bool | None = None, + seed: int | None = None, + watermark: bool | None = None, + return_last_frame: bool | None = None, + input_type: str | None = None, + budget_seconds: float | None = None, + ) -> VideoResponse: + """ + Generate a video clip from a text prompt (or text + image / face asset). + + Submits an async job, then polls until the video is ready. Typical + total wall-time is 60-180s, but upstream status can lag several + minutes behind actual completion. If upstream takes longer than the + budget (default 15min), we raise without charging — the job stays + claimable ~48h via the poll_url in the error details. + + Args: + prompt: Text description of the video. + model: Model ID (default: xai/grok-imagine-video). + image_url: Optional seed image URL for image-to-video. + last_frame_url: First-and-last-frame interpolation — a second + image that seeds the FINAL frame so the model tweens from + `image_url` -> `last_frame_url`. Requires `image_url` and a + Seedance model (bytedance/seedance-1.5-pro, seedance-2.0, + or seedance-2.0-fast). Priced identically to image-to-video. + reference_image_urls: Omni / multi-reference — up to 9 (2.0) or 30 (2.5) reference + image URLs for character/style consistency (Seedance 2.0/2.5). + Cite them as "image 1", "image 2" in the prompt. + Mutually exclusive with `image_url`, `last_frame_url`, and + `real_face_asset_id`. + reference_videos: Up to 3 http(s) motion references on Seedance 2.0. + May be combined with reference_image_urls and reference_audios. + reference_audios: Up to 3 http(s) audio references on Seedance 2.0. + Requires at least one reference image or video. + bitrate_mode: Seedance 2.x output bitrate, "standard" or "high". + output_format: Seedance 2.5 output container, "mp4" or "mov". + camera_fixed: Seedance 1.5-pro fixed-camera control. + safety_identifier: Safety identifier forwarded with a Seedance request. + real_face_asset_id: A `ta_xxxxxx` face/character asset for + identity consistency — either a Virtual Portrait (AI + character, via `PortraitClient`, $0.01) or a RealFace + (real person, via `RealFaceClient`, $0.01, no KYC). + Seedance 2.0 fast/pro only. Mutually exclusive with + `image_url`. + duration_seconds: Billed duration (defaults to model's default). + aspect_ratio: `adaptive` / `16:9` / `9:16` / `1:1` / `4:3` / + `3:4` / `21:9` / `9:21` (Seedance only; Grok ignores). + resolution: Output resolution — `360p` / `480p` / `720p` / + `1080p` / `4K`. Seedance defaults to `720p`; Grok ignores. + generate_audio: Synced audio in the output. Seedance defaults + to `True` for text-to-video and `False` for image- or + face-conditioned generation. Grok ignores this field. + seed: Deterministic generation seed (Seedance only). + watermark: Add the provider watermark (Seedance only). + return_last_frame: Also return the final frame as an image + (Seedance only). + input_type: Optional assertion of the seed mode you intend — + `text` / `image` / `first_last_frame` / `reference`. Purely a + guard: the gateway infers the mode from the seed fields above + and rejects with 400 (before charging) if your declared value + disagrees. Use it when a caller builds the seed fields + dynamically and a silently-wrong mode would be expensive — a + dropped `image_url` yields a text-to-video clip you still pay + for, whereas declaring `input_type="image"` turns that into an + error. Leave unset to accept whatever the inputs imply. + budget_seconds: Overall polling budget (default 900s). + + Returns: + VideoResponse with the clip URL, duration, upstream request_id, + and the settlement tx hash. + + Raises: + ValueError: If mutually-exclusive image inputs are combined + (see above), `last_frame_url` is passed without `image_url`, + `real_face_asset_id` is malformed, or `input_type` is not one + of the four accepted values. + PaymentError: If wallet balance is insufficient. + APIError: If upstream fails, the job times out, or any transport + error occurs. + """ + if image_url and real_face_asset_id: + raise ValueError( + "image_url and real_face_asset_id are mutually exclusive; pass at most one." + ) + if last_frame_url and not image_url: + raise ValueError( + "last_frame_url requires image_url: image_url seeds the FIRST frame and " + "last_frame_url the FINAL frame — send both." + ) + if last_frame_url and real_face_asset_id: + raise ValueError( + "last_frame_url and real_face_asset_id are mutually exclusive; " + "first-and-last-frame uses image_url + last_frame_url." + ) + if reference_image_urls: + if image_url or last_frame_url or real_face_asset_id: + raise ValueError( + "reference_image_urls is mutually exclusive with image_url, " + "last_frame_url, and real_face_asset_id." + ) + image_limit = 30 if (model or "").removeprefix("bytedance/") == "seedance-2.5" else 9 + if len(reference_image_urls) > image_limit: + raise ValueError(f"reference_image_urls accepts at most {image_limit} images.") + if (reference_videos or reference_audios) and ( + image_url or last_frame_url or real_face_asset_id + ): + raise ValueError( + "reference media is mutually exclusive with frame-seed inputs; use reference_image_urls." + ) + for clips in (reference_videos, reference_audios): + if clips is not None: + if not 1 <= len(clips) <= 3: + raise ValueError("reference media accepts 1 to 3 clips per type.") + if any( + not isinstance(clip, dict) + or not isinstance(clip.get("url"), str) + or not clip["url"].startswith(("https://", "http://")) + or clip.get("role", "reference") != "reference" + for clip in clips + ): + raise ValueError( + "reference clips require an http(s) URL and optional reference role." + ) + if real_face_asset_id is not None and not real_face_asset_id.startswith("ta_"): + raise ValueError( + "real_face_asset_id must start with 'ta_' " + "(a Virtual Portrait or RealFace asset id, e.g. 'ta_abc123xyz' — " + "enroll via PortraitClient / POST /v1/portrait/enroll or " + "RealFaceClient / POST /v1/realface/enroll)" + ) + validate_video_input_type(input_type) + + body: dict[str, Any] = { + "model": model or self.DEFAULT_MODEL, + "prompt": prompt, + } + if image_url: + body["image_url"] = image_url + if last_frame_url: + body["last_frame_url"] = last_frame_url + if reference_image_urls: + body["reference_image_urls"] = reference_image_urls + if reference_videos is not None: + body["reference_videos"] = reference_videos + if reference_audios is not None: + body["reference_audios"] = reference_audios + if bitrate_mode is not None: + body["bitrate_mode"] = bitrate_mode + if output_format is not None: + body["output_format"] = output_format + if camera_fixed is not None: + body["camera_fixed"] = camera_fixed + if safety_identifier is not None: + body["safety_identifier"] = safety_identifier + if real_face_asset_id: + body["real_face_asset_id"] = real_face_asset_id + if duration_seconds is not None: + body["duration_seconds"] = duration_seconds + if aspect_ratio is not None: + body["aspect_ratio"] = aspect_ratio + if resolution is not None: + body["resolution"] = resolution + if generate_audio is not None: + body["generate_audio"] = generate_audio + if seed is not None: + body["seed"] = seed + if watermark is not None: + body["watermark"] = watermark + if return_last_frame is not None: + body["return_last_frame"] = return_last_frame + if input_type is not None: + body["input_type"] = input_type + + budget = ( + budget_seconds if budget_seconds is not None else self.DEFAULT_GENERATE_BUDGET_SECONDS + ) + + return self._submit_and_poll(body, budget) + + def generate_from_content( + self, + content: list[dict[str, Any]], + *, + model: str | None = None, + budget_seconds: float | None = None, + **options: Any, + ) -> VideoResponse: + """ + Generate a video from a standard Seedance ``content[]`` body. + + This targets the gateway's ``POST /v1/videos`` endpoint, which accepts + the mainstream multimodal ``content`` array (text + a single reference + image) used by other Seedance APIs, so callers already holding a + ``content[]``-shaped request can submit it unchanged. The gateway + validates unsupported inputs *before* charging and then delegates to + the same x402 submit+poll pipeline as :meth:`generate`. + + Most SDK users should prefer :meth:`generate` (structured kwargs like + ``image_url`` / ``last_frame_url``) — this method exists for migrating + existing ``content[]`` payloads with no reshaping. + + Args: + content: The Seedance ``content`` array, e.g. + ``[{"type": "text", "text": "a red apple spinning"}]`` or a + text item plus ``{"type": "image_url", "image_url": {...}}``. + model: Model ID (default: the gateway's standard Seedance model). + budget_seconds: Overall polling budget (default 900s). + **options: Extra top-level body fields forwarded verbatim + (``resolution``, ``duration_seconds``, ``aspect_ratio``, + ``generate_audio``, ``seed``, ``watermark`` …). + + Returns: + VideoResponse with the clip URL, duration, upstream request_id, + and the settlement tx hash. + """ + if not content: + raise ValueError("content must be a non-empty list of Seedance content items.") + + body: dict[str, Any] = {"content": content, **options} + if model is not None: + body["model"] = model + + budget = ( + budget_seconds if budget_seconds is not None else self.DEFAULT_GENERATE_BUDGET_SECONDS + ) + return self._submit_and_poll(body, budget, submit_path="/v1/videos") + + # ------------------------------------------------------------------ + # Internal: async submit + poll + # ------------------------------------------------------------------ + + def _submit_and_poll( + self, + body: dict[str, Any], + budget_seconds: float, + submit_path: str = "/v1/videos/generations", + ) -> VideoResponse: + submit_url = f"{self.api_url}{submit_path}" + + # Step 1: unauth POST -> 402 with payment requirements + resp402 = self._client.post( + submit_url, + json=body, + headers={"Content-Type": "application/json"}, + ) + + if self.api_key: + # Account rail: the API key IS the payment, so the FIRST post already + # carries it and comes back with the async envelope. There is no 402 + # to answer here — one means the account is out of credit. + raise_for_api_key_402(resp402, self.api_key) + if resp402.status_code not in (200, 202): + self._raise_api_error(resp402, "Submit failed") + payment_payload = None + submit_resp = resp402 + else: + if resp402.status_code != 402: + self._raise_api_error(resp402, "Expected 402 on first POST") + + payment_payload = self._sign_from_challenge(resp402, submit_url) + + # Step 2: submit job with payment -> 202 { id, poll_url } + submit_resp = self._client.post( + submit_url, + json=body, + headers={ + "Content-Type": "application/json", + "PAYMENT-SIGNATURE": payment_payload, + }, + ) + + if submit_resp.status_code == 402: + raise PaymentError("Payment was rejected. Check your wallet balance.") + + if submit_resp.status_code not in (200, 202): + self._raise_api_error(submit_resp, "Submit failed") + + submit_data = submit_resp.json() + job_id = submit_data.get("id") + poll_url_rel = submit_data.get("poll_url") + if not job_id or not poll_url_rel: + raise APIError( + "Submit response missing id/poll_url", + submit_resp.status_code, + {"response": submit_data}, + retry_after=retry_after_of(submit_resp), + ) + + poll_url = self._absolute(poll_url_rel) + + # Step 3: poll with the same PAYMENT-SIGNATURE until completed. The + # signed authorization is valid for MAX_TIMEOUT_SECONDS (600s); when a + # poll 402s after that window, we fetch a fresh challenge from the + # same poll_url and re-sign with the same wallet — the gateway + # enforces wallet binding, not signature equality. + deadline = time.monotonic() + budget_seconds + last_status = submit_data.get("status", "queued") + resigns_left = self.MAX_POLL_RESIGNS + + while time.monotonic() < deadline: + time.sleep(self.POLL_INTERVAL_SECONDS) + + poll_resp = self._client.get( + poll_url, + headers={"PAYMENT-SIGNATURE": payment_payload} if payment_payload else {}, + ) + + try: + poll_data = poll_resp.json() + except Exception: + poll_data = {} + + last_status = poll_data.get("status", last_status) + + if poll_resp.status_code == 202 and last_status in ("queued", "in_progress"): + continue + + if last_status == "failed": + raise APIError( + f"Upstream generation failed: {poll_data.get('error', 'unknown')}", + poll_resp.status_code, + sanitize_error_response(poll_data), + retry_after=retry_after_of(poll_resp), + ) + + # Terminal success is keyed on status, NOT the HTTP code — the + # gateway settles the moment a poll reports completed, so coupling + # success to a literal 200 would spin to the deadline (and report + # "not charged") on a completed-but-non-200 poll the caller was + # already charged for. Mirrors the Go/TS SDKs. + if last_status == "completed": + tx_hash = poll_resp.headers.get("x-payment-receipt") or poll_resp.headers.get( + "X-Payment-Receipt" + ) + if tx_hash: + poll_data["txHash"] = tx_hash + return VideoResponse(**poll_data) + + if poll_resp.status_code == 402: + # Account rail: a 402 is the account being out of credit, not a + # challenge to sign. Nothing here can sign, so say so plainly. + raise_for_api_key_402(poll_resp, self.api_key) + # Mid-poll 402 = the signed authorization expired (600s + # window) on a budget longer than that. Re-challenge + + # re-sign and keep going. A fresh signature that 402s again + # is a genuine payment problem. + if resigns_left > 0: + resigns_left -= 1 + challenge = self._client.get(poll_url) + if challenge.status_code == 402: + payment_payload = self._sign_from_challenge(challenge, poll_url) + continue + raise PaymentError( + "Payment verification failed mid-poll (not a signature-expiry). " + "Check the wallet balance and that you poll from the wallet " + "that submitted the job." + ) + + if poll_resp.status_code not in (200, 202, 504): + self._raise_api_error(poll_resp, "Poll failed") + # status 504 on a poll = transient upstream hiccup; retry + + raise APIError( + f"Video generation did not complete within {budget_seconds:.0f}s " + f"(last status: {last_status}). No payment was taken. The job is " + f"NOT lost: it stays claimable for ~48h — re-GET poll_url with a " + f"fresh signature from the same wallet to fetch (and settle) the " + f"finished video.", + 504, + {"id": job_id, "last_status": last_status, "poll_url": poll_url}, + ) + + def _sign_from_challenge(self, resp402: httpx.Response, fallback_url: str) -> str: + """Parse an x402 challenge response and sign a payment payload for it. + + Used for the initial submit AND for mid-poll re-signing after the + 600s authorization window lapses on long polls. + """ + payment_required = self._extract_payment_required(resp402) + details = extract_payment_details(payment_required) + resource = details.get("resource") or {} + extensions = payment_required.get("extensions", {}) + return create_payment_payload( + account=self.account, + recipient=details["recipient"], + amount=details["amount"], + network=details.get("network", "eip155:8453"), + resource_url=resource.get("url", fallback_url), + resource_description=resource.get("description", "BlockRun Video Generation"), + # Cover as much of the polling window as the auth allows. + max_timeout_seconds=max( + details.get("maxTimeoutSeconds", 0) or 0, self.MAX_TIMEOUT_SECONDS + ), + extra=details.get("extra"), + asset=details.get("asset"), + extensions=extensions, + ) + + def _absolute(self, url: str) -> str: + if url.startswith(("http://", "https://")): + return url + # self.api_url already ends without '/'; poll_url starts with '/api/...' + return resolve_poll_url(url, self.api_url, self.api_key) + + def _extract_payment_required(self, resp: httpx.Response) -> dict[str, Any]: + header = resp.headers.get("payment-required") + if header: + return parse_payment_required(header) + # Fallback: body contains the x402 PaymentRequired document + try: + body = resp.json() + except Exception: + body = None + if isinstance(body, dict) and ("x402Version" in body or "accepts" in body): + return body + raise PaymentError("402 response but no payment requirements found") + + def _raise_api_error(self, resp: httpx.Response, prefix: str) -> None: + raise_api_error(resp, prefix) + + @property + def payment_mode(self) -> str: + """Which rail this client pays on: ``"apikey"`` or ``"wallet"``. + + Worth checking once at startup when both a key and a wallet are + configured in the environment: it is the difference between + spending credit and spending USDC.""" + return payment_mode(self) + + def get_wallet_address(self) -> str: + # No address on the account rail: payment comes from prepaid + # credit, so there is nothing to return but the empty string. + if self.api_key: + return "" + """Get the wallet address being used for payments.""" + return self.account.address + + def close(self): + """Close the HTTP client.""" + self._client.close() + + def __enter__(self): + return self + + def __exit__(self, exc_type, exc_val, exc_tb): + self.close() diff --git a/blockrun_llm/voice.py b/blockrun_llm/voice.py new file mode 100644 index 0000000..6ce2de0 --- /dev/null +++ b/blockrun_llm/voice.py @@ -0,0 +1,415 @@ +""" +BlockRun Voice Call Client - AI-powered outbound phone calls via x402 micropayments. + +The AI agent calls a phone number (E.164) and conducts a conversation based on your +'task' instructions. Speech-to-text, LLM reasoning, and text-to-speech are all handled +upstream by Bland.ai; BlockRun handles billing through x402. + +SECURITY NOTE - Private Key Handling: +===================================== +Your private key NEVER leaves your machine. Here's what happens: + +1. Key stays local - only used to sign an EIP-712 typed data message +2. Only the SIGNATURE is sent in the PAYMENT-SIGNATURE header +3. BlockRun verifies the signature on-chain via Coinbase CDP facilitator + +Usage: + from blockrun_llm import VoiceClient + + client = VoiceClient() # Uses BLOCKRUN_WALLET_KEY from env + + # Initiate a call (paid, $0.54) + result = client.call( + to="+14155552671", + task="You are a friendly assistant calling to confirm a 3pm dentist appointment.", + max_duration=5, + ) + print(result["call_id"]) + + # Poll for status, transcript, and recording (free) + status = client.get_status(result["call_id"]) + print(status) + +Pricing: $0.54 per outbound call (regardless of duration up to max_duration). +""" + +from __future__ import annotations + +import os +from typing import Any + +import httpx +from dotenv import load_dotenv +from eth_account import Account + +from .apikey import ( + api_key_base_url, + auth_headers, + missing_credential_error, + payment_mode, + raise_for_api_key_402, + resolve_api_key, +) +from .tx_log import paid_request_error_prefix +from .types import APIError, PaymentError, retry_after_of +from .validation import ( + sanitize_error_response, + validate_api_url, + validate_private_key, +) +from .x402 import create_payment_payload, extract_payment_details, parse_payment_required + +load_dotenv() + + +# Built-in Bland.ai voice presets — any string accepted by Bland is also valid. +VOICE_PRESETS: list[str] = ["nat", "josh", "maya", "june", "paige", "derek", "florian"] + +# Bland.ai conversation models +CALL_MODELS: list[str] = ["base", "enhanced", "turbo"] + +# Settled price per call (USD) +CALL_PRICE_USD: float = 0.54 + + +class VoiceClient: + """ + BlockRun Voice Call Client. + + Initiates AI-powered outbound phone calls. The AI agent dials the recipient and + conducts a real-time conversation following your 'task' description. + + Pricing: $0.54 per call. Status polling is free. + + Caller-ID requirements: every call needs a `from` number your wallet owns. + Provision one with PhoneClient.buy_number() before placing calls; if your + wallet owns exactly one active number, the backend auto-picks it. + """ + + DEFAULT_API_URL = "https://blockrun.ai/api" + DEFAULT_TIMEOUT = 60.0 # call initiation returns quickly; long-poll status separately + + def __init__( + self, + private_key: str | None = None, + api_url: str | None = None, + timeout: float = 60.0, + ): + """ + Initialize the BlockRun Voice client. + + Args: + private_key: EVM wallet private key (or set BLOCKRUN_WALLET_KEY env var) + api_url: API endpoint URL (default: https://blockrun.ai/api) + timeout: Request timeout in seconds + """ + from .wallet import load_wallet + + # Account rail first, and before any wallet variable is read: an API + # key is a complete credential on its own, so demanding a private key + # alongside it would make every API-key user invent a wallet they + # never use. See apikey.py for the precedence rule. + api_key = resolve_api_key(private_key) + key = ( + None + if api_key + else ( + private_key + or os.environ.get("BLOCKRUN_WALLET_KEY") + or os.environ.get("BASE_CHAIN_WALLET_KEY") + or load_wallet() + ) + ) + if not api_key and not key: + raise missing_credential_error() + + if key: + validate_private_key(key) + self.api_key = api_key + # No wallet on the account rail: nothing is signed locally. + self.account = Account.from_key(key) if key else None + + # BLOCKRUN_API_URL names an x402 gateway; an API-key client must not + # follow it and hand the key to a host set up for another rail. + api_url_raw = ( + api_key_base_url(api_url) + if api_key + else (api_url or os.environ.get("BLOCKRUN_API_URL") or self.DEFAULT_API_URL) + ) + validate_api_url(api_url_raw) + self.api_url = api_url_raw.rstrip("/") + + self.timeout = timeout + self._client = httpx.Client(headers=auth_headers(api_key), timeout=timeout) + + def call( + self, + to: str, + task: str, + *, + from_: str | None = None, + voice: str | None = None, + max_duration: int = 5, + language: str = "en-US", + first_sentence: str | None = None, + wait_for_greeting: bool | None = None, + interruption_threshold: int | None = None, + model: str | None = None, + ) -> dict[str, Any]: + """ + Initiate an AI-powered outbound phone call. + + Args: + to: Destination phone number in E.164 format (e.g. "+14155552671"). + US and Canada supported. + task: Natural-language instructions for the AI agent + (10-4000 chars). Describe what the call should accomplish. + from_: Your provisioned BlockRun phone number (E.164). Shown as caller ID. + Must be owned by your wallet — buy one via PhoneClient.buy_number() + or POST /v1/phone/numbers/buy ($5 / 30-day lease). + Use the trailing-underscore form because 'from' is a Python keyword. + + If omitted: + - wallet owns exactly 1 active number → that number is used automatically + - wallet owns 0 → APIError(403) "no_active_number" (buy one first) + - wallet owns 2+ → APIError(400) "ambiguous_from" (pass `from_` explicitly; + the error body lists your_active_numbers so the agent can pick) + voice: One of VOICE_PRESETS (nat, josh, maya, june, paige, derek, florian) + or any custom Bland.ai voice ID. + max_duration: Maximum call length in minutes (1-30, default 5). + language: BCP-47 language code for STT/TTS (default "en-US"). + first_sentence: Optional opening line the agent says before listening. + wait_for_greeting: If True, wait for the recipient to speak first. + interruption_threshold: Sensitivity for detecting recipient interruptions + (50-500ms). Lower = quicker to yield the floor. + model: Conversation model — "base", "enhanced", or "turbo". + + Returns: + Dict with keys: + - call_id (str): Bland.ai call identifier + - status (str): Initial status (usually "queued") + - poll_url (str): URL to poll for transcript/recording + - message (str): Human-readable note + - txHash (str, optional): On-chain payment receipt + + Raises: + ValueError: If arguments are out of range + PaymentError: If wallet has insufficient balance + APIError: If the API or upstream provider returns an error + + Example: + result = client.call( + to="+14155552671", + task="Call the user and confirm they want to reschedule to Tuesday 2pm.", + voice="maya", + max_duration=3, + ) + print(result["call_id"]) + """ + if not to or not to.strip(): + raise ValueError("'to' phone number is required (E.164 format)") + if not task or len(task.strip()) < 10: + raise ValueError("'task' must be at least 10 characters") + if len(task) > 4000: + raise ValueError("'task' must be at most 4000 characters") + if max_duration < 1 or max_duration > 30: + raise ValueError("max_duration must be between 1 and 30 minutes") + if model is not None and model not in CALL_MODELS: + raise ValueError(f"model must be one of {CALL_MODELS}") + if interruption_threshold is not None and not (50 <= interruption_threshold <= 500): + raise ValueError("interruption_threshold must be between 50 and 500") + + body: dict[str, Any] = { + "to": to.strip(), + "task": task.strip(), + "max_duration": max_duration, + "language": language, + } + if from_: + body["from"] = from_.strip() + if voice: + body["voice"] = voice + if first_sentence: + body["first_sentence"] = first_sentence.strip() + if wait_for_greeting is not None: + body["wait_for_greeting"] = wait_for_greeting + if interruption_threshold is not None: + body["interruption_threshold"] = interruption_threshold + if model: + body["model"] = model + + return self._request_with_payment("/v1/voice/call", body) + + def get_status(self, call_id: str) -> dict[str, Any]: + """ + Poll the status of an in-progress or completed call. Free — no payment. + + Args: + call_id: The 'call_id' returned by call(). + + Returns: + Dict with the full Bland.ai call record, including: + - status: "queued" | "in-progress" | "completed" | "failed" | ... + - transcripts: List of turns once available + - recording_url: Audio URL once the call ends + - duration, started_at, ended_at, etc. + + Raises: + APIError: If the call is not found (404) or upstream errors. + """ + if not call_id or not call_id.strip(): + raise ValueError("call_id is required") + + url = f"{self.api_url}/v1/voice/call/{call_id.strip()}" + response = self._client.get(url, headers={"Accept": "application/json"}) + raise_for_api_key_402(response, self.api_key) + + if response.status_code == 404: + raise APIError(f"Call not found: {call_id}", 404, {"call_id": call_id}) + + if response.status_code != 200: + try: + error_body = response.json() + except Exception: + error_body = {"error": "Request failed"} + raise APIError( + f"API error: {response.status_code}", + response.status_code, + sanitize_error_response(error_body), + retry_after=retry_after_of(response), + ) + + return response.json() + + def _request_with_payment(self, endpoint: str, body: dict[str, Any]) -> dict[str, Any]: + """Make a POST with automatic x402 payment handling.""" + url = f"{self.api_url}{endpoint}" + + response = self._client.post( + url, + json=body, + headers={"Content-Type": "application/json"}, + ) + + if response.status_code == 402: + # Account rail: a 402 is the account being out of credit, not a + # challenge to sign. Nothing here can sign, so say so plainly. + raise_for_api_key_402(response, self.api_key) + return self._handle_payment_and_retry(url, body, response) + + if response.status_code != 200: + try: + error_body = response.json() + except Exception: + error_body = {"error": "Request failed"} + raise APIError( + f"API error: {response.status_code}", + response.status_code, + sanitize_error_response(error_body), + retry_after=retry_after_of(response), + ) + + return response.json() + + def _handle_payment_and_retry( + self, + url: str, + body: dict[str, Any], + response: httpx.Response, + ) -> dict[str, Any]: + """Handle 402: parse requirements, sign payment, retry.""" + payment_header = response.headers.get("payment-required") + if not payment_header: + try: + resp_body = response.json() + if "x402" in resp_body or "accepts" in resp_body: + payment_header = resp_body + except Exception: + pass + + if not payment_header: + raise PaymentError("402 response but no payment requirements found") + + if isinstance(payment_header, str): + payment_required = parse_payment_required(payment_header) + else: + payment_required = payment_header + + details = extract_payment_details(payment_required) + resource = details.get("resource") or {} + extensions = payment_required.get("extensions", {}) + + payment_payload = create_payment_payload( + account=self.account, + recipient=details["recipient"], + amount=details["amount"], + network=details.get("network", "eip155:8453"), + resource_url=resource.get("url", f"{self.api_url}/v1/voice/call"), + resource_description=resource.get("description", "BlockRun Voice Call"), + max_timeout_seconds=details.get("maxTimeoutSeconds", 300), + extra=details.get("extra"), + asset=details.get("asset"), + extensions=extensions, + ) + + retry_response = self._client.post( + url, + json=body, + headers={ + "Content-Type": "application/json", + "PAYMENT-SIGNATURE": payment_payload, + }, + ) + + if retry_response.status_code == 402: + # Account rail: a 402 is the account being out of credit, not a + # challenge to sign. Nothing here can sign, so say so plainly. + raise_for_api_key_402(retry_response, self.api_key) + raise PaymentError("Payment was rejected. Check your wallet balance.") + + if retry_response.status_code != 200: + try: + error_body = retry_response.json() + except Exception: + error_body = {"error": "Request failed"} + raise APIError( + f"{paid_request_error_prefix(retry_response.headers)}: {retry_response.status_code}", + retry_response.status_code, + sanitize_error_response(error_body), + retry_after=retry_after_of(retry_response), + ) + + data = retry_response.json() + tx_hash = retry_response.headers.get("x-payment-receipt") or retry_response.headers.get( + "X-Payment-Receipt" + ) + if tx_hash: + data["txHash"] = tx_hash + return data + + @property + def payment_mode(self) -> str: + """Which rail this client pays on: ``"apikey"`` or ``"wallet"``. + + Worth checking once at startup when both a key and a wallet are + configured in the environment: it is the difference between + spending credit and spending USDC.""" + return payment_mode(self) + + def get_wallet_address(self) -> str: + # No address on the account rail: payment comes from prepaid + # credit, so there is nothing to return but the empty string. + if self.api_key: + return "" + """Get the wallet address being used for payments.""" + return self.account.address + + def close(self): + """Close the HTTP client.""" + self._client.close() + + def __enter__(self): + return self + + def __exit__(self, exc_type, exc_val, exc_tb): + self.close() diff --git a/blockrun_llm/wallet.py b/blockrun_llm/wallet.py new file mode 100644 index 0000000..007a408 --- /dev/null +++ b/blockrun_llm/wallet.py @@ -0,0 +1,724 @@ +""" +BlockRun Wallet Management - Auto-create and manage wallets. + +Provides frictionless wallet setup for new users: +- Auto-creates wallet if none exists +- Stores key securely at ~/.blockrun/.session +- Generates EIP-681 QR codes for easy MetaMask funding +""" + +from __future__ import annotations + +import json +import os +import time +from pathlib import Path +from typing import TYPE_CHECKING + +from eth_account import Account + +if TYPE_CHECKING: + from blockrun_llm import LLMClient + +# Wallet storage location +WALLET_DIR = Path.home() / ".blockrun" +WALLET_FILE = WALLET_DIR / ".session" # Wallet key file +QR_FILE = WALLET_DIR / "qr.png" +QR_ASCII_FILE = WALLET_DIR / "qr.txt" + +# USDC on Base contract address +USDC_BASE_CONTRACT = "0x833589fCD6eDb6E08f4c7C32D4f71b54bdA02913" +BASE_CHAIN_ID = "8453" + + +def create_wallet() -> tuple[str, str]: + """ + Create a new Ethereum wallet. + + Returns: + Tuple of (address, private_key) + """ + account = Account.create() + private_key = "0x" + account.key.hex() + return account.address, private_key + + +def save_wallet(private_key: str) -> Path: + """ + Save wallet private key to ~/.blockrun/.session + + Args: + private_key: Private key string (with or without 0x prefix) + + Returns: + Path to saved wallet file + """ + WALLET_DIR.mkdir(exist_ok=True) + WALLET_FILE.write_text(private_key) + WALLET_FILE.chmod(0o600) # Owner read/write only + return WALLET_FILE + + +def scan_wallets() -> list[dict[str, str]]: + """ + Discover ~/./wallet.json files from other providers. + + Each file should contain JSON with "privateKey" and "address" fields. + Results are sorted by modification time (most recent first). Discovery is + opt-in and must never replace the canonical BlockRun wallet automatically. + + Returns: + List of dicts with 'private_key', 'address' and 'source', most recent + first. 'address' is the file's own claim — use list_discovered_wallets() + for an address derived from the key. + """ + home = Path.home() + results: list[tuple] = [] # (mtime, private_key, address, source) + + try: + for entry in home.iterdir(): + if not entry.name.startswith(".") or not entry.is_dir(): + continue + wallet_file = entry / "wallet.json" + if not wallet_file.is_file(): + continue + try: + data = json.loads(wallet_file.read_text()) + pk = data.get("privateKey", "") + addr = data.get("address", "") + if pk and addr: + mtime = wallet_file.stat().st_mtime + results.append((mtime, pk, addr, str(wallet_file))) + except (json.JSONDecodeError, OSError): + continue + except OSError: + pass + + # Sort by modification time, most recent first + results.sort(key=lambda x: x[0], reverse=True) + return [{"private_key": pk, "address": addr, "source": src} for _, pk, addr, src in results] + + +def list_discovered_wallets() -> list[dict[str, str]]: + """ + List wallets from other applications, safe to show to a user. + + Unlike scan_wallets(), the private key is not returned and the address is + derived from the key rather than read from the file, so a wallet file + cannot claim an address it has no key for. + + Nothing here is active. Adopt one deliberately with import_wallet(). + + Returns: + List of dicts with 'address' and 'source', most recent first + """ + listed = [] + for entry in scan_wallets(): + try: + address = Account.from_key(entry["private_key"]).address + except Exception: + continue + listed.append({"address": address, "source": entry.get("source", "")}) + return listed + + +def import_wallet(address: str) -> str: + """ + Adopt a discovered wallet by address, making it the active BlockRun wallet. + + This is the deliberate migration path: automatic selection never adopts a + discovered wallet, but you can choose one whose funds you want to spend. + Matching is done against the address *derived from each discovered key*, so + a wallet file claiming someone else's address can never be selected by it. + + The current ~/.blockrun/.session is backed up beside itself before being + overwritten, so adopting a wallet can't strand the funds in the old one. + + Args: + address: Address to adopt, as shown by list_discovered_wallets() + + Returns: + The adopted address + + Raises: + ValueError: If no discovered wallet derives to that address + """ + wanted = address.strip().lower() + + for entry in scan_wallets(): + try: + derived = Account.from_key(entry["private_key"]).address + except Exception: + continue + + if derived.lower() != wanted: + continue + + # Preserve the outgoing wallet — it may hold funds. + if WALLET_FILE.exists(): + current = WALLET_FILE.read_text().strip() + if current and current != entry["private_key"]: + backup = WALLET_FILE.with_name(f".session.backup-{int(time.time())}") + backup.write_text(current) + backup.chmod(0o600) + + save_wallet(entry["private_key"]) + return derived + + available = [w["address"] for w in list_discovered_wallets()] + raise ValueError( + f"No discovered wallet controls {address}. " + f"Available: {', '.join(available) if available else 'none'}" + ) + + +def load_wallet() -> str | None: + """ + Load wallet private key from file. + + Priority: + 1. ~/.blockrun/.session + 2. ~/.blockrun/wallet.key (legacy) + + Returns: + Private key string or None if not found + """ + # The canonical BlockRun wallet always wins. Do not implicitly adopt a + # wallet discovered in another application's private storage. + if WALLET_FILE.exists(): + key = WALLET_FILE.read_text().strip() + if key: + return key + + # Check legacy wallet.key + legacy_file = WALLET_DIR / "wallet.key" + if legacy_file.exists(): + key = legacy_file.read_text().strip() + if key: + return key + + return None + + +def get_or_create_wallet() -> tuple[str, str, bool]: + """ + Get existing wallet or create new one. + + Priority: + 1. BLOCKRUN_WALLET_KEY / BASE_CHAIN_WALLET_KEY environment variable + 2. ~/.blockrun/.session file + 3. ~/.blockrun/wallet.key (legacy) + 4. Create new wallet + + Returns: + Tuple of (address, private_key, is_new) + is_new is True if wallet was just created + """ + # 1. Check environment variable first + key = os.environ.get("BLOCKRUN_WALLET_KEY") or os.environ.get("BASE_CHAIN_WALLET_KEY") + if key: + account = Account.from_key(key) + return account.address, key, False + + # 2-3. Canonical BlockRun wallet, then the legacy wallet.key. Delegating to + # load_wallet() keeps this in step with the TypeScript SDK, which resolves + # the same two files. scan_wallets() is exposed for an explicit migration + # flow only and must not affect automatic selection. + file_key = load_wallet() + if file_key: + account = Account.from_key(file_key) + return account.address, file_key, False + + # 4. Create new wallet + address, key = create_wallet() + save_wallet(key) + return address, key, True + + +def get_wallet_address() -> str | None: + """ + Get wallet address without exposing private key. + + Returns: + Wallet address or None if no wallet configured + """ + key = os.environ.get("BLOCKRUN_WALLET_KEY") or os.environ.get("BASE_CHAIN_WALLET_KEY") + if key: + return Account.from_key(key).address + + key = load_wallet() + if key: + return Account.from_key(key).address + + return None + + +def get_eip681_uri(address: str, amount_usdc: float = 1.0) -> str: + """ + Generate EIP-681 URI for USDC transfer on Base. + + Args: + address: Recipient Ethereum address + amount_usdc: Amount in USDC (default 1.0) + + Returns: + EIP-681 URI string for MetaMask/wallet scanning + """ + # USDC has 6 decimals + amount_wei = int(amount_usdc * 1_000_000) + return f"ethereum:{USDC_BASE_CONTRACT}@{BASE_CHAIN_ID}/transfer?address={address}&uint256={amount_wei}" + + +def generate_wallet_qr_ascii(address: str) -> str: + """ + Generate ASCII QR code for wallet funding (EIP-681 format). + Caches to ~/.blockrun/qr.txt for fast loading. + + Args: + address: Ethereum address + + Returns: + ASCII art QR code string + """ + # Use EIP-681 format for MetaMask compatibility + eip681_uri = get_eip681_uri(address) + + # Cache key includes EIP-681 URI to invalidate old format caches + cache_key = f"v2:{eip681_uri}" + + # Try to load from cache first + if QR_ASCII_FILE.exists(): + try: + cached = QR_ASCII_FILE.read_text() + # Format: first line is cache key (v2:eip681_uri), rest is QR + lines = cached.split("\n", 1) + if len(lines) == 2 and lines[0] == cache_key: + return lines[1] + except Exception: + pass + + # Generate new QR + try: + from io import StringIO + + import qrcode + + qr = qrcode.QRCode( + version=1, + error_correction=qrcode.constants.ERROR_CORRECT_L, + box_size=1, + border=1, + ) + qr.add_data(eip681_uri) + qr.make(fit=True) + + f = StringIO() + qr.print_ascii(out=f, invert=True) + qr_ascii = f.getvalue() + + # Cache it with versioned key + try: + WALLET_DIR.mkdir(exist_ok=True) + QR_ASCII_FILE.write_text(f"{cache_key}\n{qr_ascii}") + except Exception: + pass + + return qr_ascii + + except ImportError: + return f"[QR code requires 'qrcode' package: pip install qrcode[pil]]\nAddress: {address}" + + +def save_wallet_qr(address: str, path: str | None = None, with_logo: bool = True) -> str: + """ + Save QR code as PNG image (EIP-681 format with optional Base logo). + + Args: + address: Ethereum address + path: Optional custom path (default: ~/.blockrun/qr.png) + with_logo: Whether to embed Base logo in center (default: True) + + Returns: + Path to saved QR image + """ + try: + import io + import urllib.request + + import qrcode + from PIL import Image + + # Use EIP-681 format for MetaMask compatibility + eip681_uri = get_eip681_uri(address) + + # Use high error correction when adding logo + error_correction = ( + qrcode.constants.ERROR_CORRECT_H if with_logo else qrcode.constants.ERROR_CORRECT_L + ) + + qr = qrcode.QRCode( + version=4, + error_correction=error_correction, + box_size=10, + border=2, + ) + qr.add_data(eip681_uri) + qr.make(fit=True) + + img = qr.make_image(fill_color="black", back_color="white").convert("RGB") + + # Add Base logo to center + if with_logo: + try: + logo_url = "https://avatars.githubusercontent.com/u/108554348?s=200&v=4" + with urllib.request.urlopen(logo_url, timeout=5) as response: + logo_data = response.read() + logo = Image.open(io.BytesIO(logo_data)) + + # Resize logo to ~20% of QR size + qr_width, qr_height = img.size + logo_size = int(qr_width * 0.2) + logo = logo.resize((logo_size, logo_size), Image.Resampling.LANCZOS) + + # Paste in center + pos = ((qr_width - logo_size) // 2, (qr_height - logo_size) // 2) + img.paste(logo, pos) + except Exception: + pass # Continue without logo if fetch fails + + save_path = Path(path) if path else QR_FILE + save_path.parent.mkdir(exist_ok=True) + img.save(str(save_path)) + + return str(save_path) + + except ImportError: + return "" + + +def open_wallet_qr(address: str) -> str: + """ + Generate QR code and open it in the default image viewer. + + Args: + address: Ethereum address + + Returns: + Path to saved QR image + """ + import platform + import subprocess + + qr_path = save_wallet_qr(address) + if qr_path: + try: + if platform.system() == "Darwin": # macOS + subprocess.run(["open", qr_path], check=True) + elif platform.system() == "Windows": + subprocess.run(["start", qr_path], shell=True, check=True) + else: # Linux + subprocess.run(["xdg-open", qr_path], check=True) + except Exception: + pass # Silently fail if can't open + return qr_path + + +def get_payment_links(address: str) -> dict: + """ + Generate payment links for the wallet address. + + Args: + address: Ethereum address + + Returns: + Dict with various payment links + """ + return { + # View address on basescan + "basescan": f"https://basescan.org/address/{address}", + # EIP-681 payment link (opens wallet apps) + "wallet_link": f"ethereum:{USDC_BASE_CONTRACT}@{BASE_CHAIN_ID}/transfer?address={address}", + # Simple ethereum link (some wallets) + "ethereum": f"ethereum:{address}@{BASE_CHAIN_ID}", + # BlockRun funding page (if available) + "blockrun": f"https://blockrun.ai/fund?address={address}", + } + + +def format_wallet_migration_notice(new_address: str) -> str | None: + """ + Warn when a new wallet was created while other provider wallets exist. + + Automatic selection deliberately ignores wallets discovered in other + applications' directories, so a user who previously relied on that + discovery would otherwise land on an empty wallet with no explanation of + where their funds went. This notice names the discovered addresses and + tells them how to import one on purpose. + + Addresses are derived from the discovered private key rather than read + from the file's "address" field, so a file claiming an address it cannot + sign for cannot trick the user into importing it. + + Args: + new_address: Address of the wallet that was just created + + Returns: + Formatted notice, or None if nothing was discovered + """ + try: + discovered = scan_wallets() + except Exception: + return None + + addresses = [] + for entry in discovered: + try: + addresses.append(Account.from_key(entry["private_key"]).address) + except Exception: + continue + + if not addresses: + return None + + found = "\n".join(f" {addr}" for addr in addresses) + return f""" +NOTICE: BlockRun created a new wallet, but also found existing wallet(s) +belonging to other applications on this system: + +{found} + +BlockRun now uses only its own wallet: + + {new_address} + +Discovered wallets are never adopted automatically — one may belong to a +different application, or have been planted to make you fund an address you +do not control. + +If an address above is yours and holds your USDC, adopt it deliberately: + + from blockrun_llm import import_wallet + import_wallet("") + +Your current wallet is backed up first. You can also set +BLOCKRUN_WALLET_KEY= for a single run without changing anything. +""" + + +def format_wallet_created_message(address: str, open_qr: bool = True) -> str: + """ + Format the message shown when a new wallet is created. + + Args: + address: New wallet address + open_qr: Whether to open QR code in image viewer (default: True) + + Returns: + Formatted message string + """ + qr_ascii = generate_wallet_qr_ascii(address) + # Generate and optionally open QR code + if open_qr: + qr_path = open_wallet_qr(address) + else: + qr_path = save_wallet_qr(address) + links = get_payment_links(address) + + message = f""" +I'm your BlockRun Agent! I can access GPT-5, Claude, Gemini, and more. + +Please send $1-5 USDC on Base to start: + +{address} + +{qr_ascii} +""" + + if qr_path: + message += f"QR saved: {qr_path}\n" + + message += f""" +What is Base? Base is Coinbase's blockchain network. +You can buy USDC on Coinbase and send it directly to me. + +What $1 USDC gets you: +- ~1,000 GPT-5.2 calls +- ~100 image generations +- ~10,000 DeepSeek calls + +Quick links: +- Check my balance: {links['basescan']} +- Get USDC: https://www.coinbase.com or https://bridge.base.org + +Questions? care@blockrun.ai | Issues? github.com/BlockRunAI/blockrun-llm/issues + +Key stored securely in ~/.blockrun/ +Your private key never leaves your machine - only signatures are sent. +""" + return message + + +def format_needs_funding_message(address: str, open_qr: bool = True) -> str: + """ + Format the message shown when wallet needs more funds. + + Args: + address: Wallet address + open_qr: Whether to open QR code in image viewer (default: True) + + Returns: + Formatted message string + """ + qr_ascii = generate_wallet_qr_ascii(address) + # Open QR for easy scanning + if open_qr: + open_wallet_qr(address) + links = get_payment_links(address) + + return f""" +I've run out of funds! Please send more USDC on Base to continue helping you. + +Send to my address: +{address} + +{qr_ascii} + +Check my balance: {links['basescan']} + +What $1 USDC gets you: ~1,000 GPT-5.2 calls or ~100 images. +Questions? care@blockrun.ai | Issues? github.com/BlockRunAI/blockrun-llm/issues + +Your private key never leaves your machine - only signatures are sent. +""" + + +def format_funding_message_compact(address: str) -> str: + """ + Compact funding message (no QR) for repeated displays. + + Args: + address: Wallet address + + Returns: + Short formatted message string + """ + links = get_payment_links(address) + + return f"""I need a little top-up to keep helping you! Send USDC on Base to: {address} +Check my balance: {links['basescan']}""" + + +def setup_agent_wallet(silent: bool = False) -> LLMClient: + """ + Set up wallet for agent use and return an LLMClient. + + This is the entry point for Claude Code skills and other agent runtimes. + It auto-creates a wallet if needed and shows the welcome/funding message. + + Args: + silent: If True, don't print welcome message (default: False) + + Returns: + Configured LLMClient ready for use + + Example: + from blockrun_llm import setup_agent_wallet + + client = setup_agent_wallet() # Shows welcome message if new wallet + response = client.chat("openai/gpt-5.2", "Hello!") + """ + import sys + + from .apikey import resolve_api_key + + # With an API key configured this mints nothing: the account rail already + # has a funded identity, and writing a keyfile for a wallet that will never + # sign anything is a private key to lose for no benefit. This is what lets + # a skill or agent call setup_agent_wallet() unconditionally and work on + # either rail. + api_key = resolve_api_key(None) + if api_key: + from .client import LLMClient + + return LLMClient(private_key=api_key) + + address, key, is_new = get_or_create_wallet() + + if is_new: + # Printed even when silent: `silent` suppresses the welcome banner, and + # losing sight of a funded wallet is not something to stay quiet about. + notice = format_wallet_migration_notice(address) + if notice: + print(notice, file=sys.stderr) + + if not silent: + print(format_wallet_created_message(address), file=sys.stderr) + + # Import here to avoid circular import + from .client import LLMClient + + return LLMClient(private_key=key) + + +def status() -> dict: + """ + Print wallet status and return info dict. + + One-command verification that shows wallet address and balance. + Creates wallet if needed (silently). + + Returns: + Dict with 'address' and 'balance' keys + + Example: + python3 -c "from blockrun_llm import status; status()" + """ + client = setup_agent_wallet(silent=True) + addr = client.get_wallet_address() + bal = client.get_balance() + print(f"Wallet: {addr}") + print(f"Balance: ${bal:.2f} USDC") + return {"address": addr, "balance": bal} + + +# GitHub issue link for error reporting +ISSUES_URL = "https://github.com/BlockRunAI/blockrun-llm/issues" + + +def format_error_message(error: str, context: str = "") -> str: + """ + Format error message with pre-filled report template. + + Args: + error: The error message + context: Optional context about what was happening + + Returns: + Formatted error message with copy-paste template + """ + import platform + from datetime import datetime + + timestamp = datetime.now().strftime("%Y-%m-%d %H:%M") + os_info = platform.system() + + template = f"""Error: {error} +Context: {context or 'N/A'} +Time: {timestamp} +OS: {os_info}""" + + # URL encode for GitHub issue link + encoded_title = error[:50].replace(" ", "+").replace("\n", "") + encoded_body = template.replace("\n", "%0A").replace(" ", "+") + + return f""" +Something went wrong: {error} + +Report this issue (click or copy): +{ISSUES_URL}/new?title={encoded_title}&body={encoded_body} + +Or copy this and email to care@blockrun.ai: +--- +{template} +--- +""" diff --git a/blockrun_llm/x402.py b/blockrun_llm/x402.py index 3da7302..af10d78 100644 --- a/blockrun_llm/x402.py +++ b/blockrun_llm/x402.py @@ -1,185 +1,311 @@ -""" -x402 Payment Protocol v2 Implementation for BlockRun. - -This module handles creating signed payment payloads for the x402 v2 protocol. -The private key is used ONLY for local signing and NEVER leaves the client. -""" - -import json -import time -import base64 -import secrets -from typing import Dict, Any, Optional -from eth_account import Account -from eth_account.messages import encode_typed_data - - -# Chain and token constants -BASE_CHAIN_ID = 8453 -USDC_BASE = "0x833589fCD6eDb6E08f4c7C32D4f71b54bdA02913" - - -def create_nonce() -> str: - """Generate a random bytes32 nonce.""" - return "0x" + secrets.token_hex(32) - - -def create_payment_payload( - account: Account, - recipient: str, - amount: str, # In micro USDC (6 decimals) - network: str = "eip155:8453", - resource_url: str = "https://blockrun.ai/api/v1/chat/completions", - resource_description: str = "BlockRun AI API call", - max_timeout_seconds: int = 300, - extra: Optional[Dict[str, str]] = None, - extensions: Optional[Dict[str, Any]] = None, -) -> str: - """ - Create a signed x402 v2 payment payload. - - This uses EIP-712 typed data signing to create a payment authorization - that the CDP facilitator can verify and settle. - - Args: - account: eth-account Account instance - recipient: Payment recipient address (checksummed) - amount: Amount in micro USDC (6 decimals, e.g., "1000" = $0.001) - network: Network identifier (default: Base mainnet) - resource_url: URL of the resource being accessed - resource_description: Description of the resource - max_timeout_seconds: Max timeout for the payment (default: 300) - extra: Extra info for USDC domain (name, version) - - Returns: - Base64-encoded signed payment payload - """ - # Current timestamp - now = int(time.time()) - valid_after = now - 600 # 10 minutes before (allows for clock skew) - valid_before = now + max_timeout_seconds - - # Generate random nonce - nonce = create_nonce() - - # EIP-712 domain for Base USDC - domain = { - "name": extra.get("name", "USD Coin") if extra else "USD Coin", - "version": extra.get("version", "2") if extra else "2", - "chainId": BASE_CHAIN_ID, - "verifyingContract": USDC_BASE, - } - - # EIP-712 types for TransferWithAuthorization - types = { - "TransferWithAuthorization": [ - {"name": "from", "type": "address"}, - {"name": "to", "type": "address"}, - {"name": "value", "type": "uint256"}, - {"name": "validAfter", "type": "uint256"}, - {"name": "validBefore", "type": "uint256"}, - {"name": "nonce", "type": "bytes32"}, - ], - } - - # Message to sign - message = { - "from": account.address, - "to": recipient, - "value": int(amount), - "validAfter": valid_after, - "validBefore": valid_before, - "nonce": bytes.fromhex(nonce[2:]), # Remove 0x prefix - } - - # Sign using EIP-712 - signable = encode_typed_data(domain, types, message) - signed = account.sign_message(signable) - - # Create x402 v2 payment payload - payment_data = { - "x402Version": 2, - "resource": { - "url": resource_url, - "description": resource_description, - "mimeType": "application/json", - }, - "accepted": { - "scheme": "exact", - "network": network, - "amount": amount, - "asset": USDC_BASE, - "payTo": recipient, - "maxTimeoutSeconds": max_timeout_seconds, - "extra": extra or {"name": "USD Coin", "version": "2"}, - }, - "payload": { - "signature": "0x" + signed.signature.hex() if not signed.signature.hex().startswith("0x") else signed.signature.hex(), - "authorization": { - "from": account.address, - "to": recipient, - "value": amount, - "validAfter": str(valid_after), - "validBefore": str(valid_before), - "nonce": nonce, - }, - }, - "extensions": extensions or {}, - } - - # Encode as base64 - return base64.b64encode(json.dumps(payment_data).encode()).decode() - - -def parse_payment_required(header_value: str) -> Dict[str, Any]: - """ - Parse the X-Payment-Required header from a 402 response. - - Args: - header_value: Base64-encoded payment requirements - - Returns: - Decoded payment requirements dict - """ - try: - decoded = base64.b64decode(header_value) - return json.loads(decoded) - except Exception: - # Don't expose internal error details - raise ValueError("Failed to parse payment required header: invalid format") - - -def extract_payment_details(payment_required: Dict[str, Any]) -> Dict[str, Any]: - """ - Extract payment details from parsed payment required response. - - Supports both v1 and v2 formats. - - Args: - payment_required: Parsed payment required dict - - Returns: - Dict with amount, recipient, network, asset, and extra info - """ - accepts = payment_required.get("accepts", []) - if not accepts: - raise ValueError("No payment options in payment required response") - - # Take the first option - option = accepts[0] - - # Support both v1 (maxAmountRequired) and v2 (amount) formats - amount = option.get("amount") or option.get("maxAmountRequired") - if not amount: - raise ValueError("No amount found in payment requirements") - - return { - "amount": amount, - "recipient": option.get("payTo"), - "network": option.get("network"), - "asset": option.get("asset"), - "scheme": option.get("scheme"), - "maxTimeoutSeconds": option.get("maxTimeoutSeconds", 300), - "extra": option.get("extra"), - "resource": payment_required.get("resource"), - } +""" +x402 Payment Protocol v2 Implementation for BlockRun. + +This module handles creating signed payment payloads for the x402 v2 protocol. +The private key is used ONLY for local signing and NEVER leaves the client. +""" + +from __future__ import annotations + +import base64 +import json +import secrets +import time +from typing import Any + +from eth_account import Account +from eth_account.messages import encode_typed_data + +# Chain and token constants for mainnet +BASE_CHAIN_ID = 8453 +USDC_BASE = "0x833589fCD6eDb6E08f4c7C32D4f71b54bdA02913" + +# Chain and token constants for testnet (Base Sepolia) +BASE_SEPOLIA_CHAIN_ID = 84532 +USDC_BASE_SEPOLIA = "0x036CbD53842c5426634e7929541eC2318f3dCF7e" + +# Circle's Arc (arc.blockrun.ai). USDC is the chain's native token, exposed as +# the ERC-20 at 0x3600…0000; its EIP-712 domain name is "USDC", not Base's +# "USD Coin". +ARC_CHAIN_ID = 5042 +USDC_ARC = "0x3600000000000000000000000000000000000000" + +# The EVM networks a BlockRun gateway settles on, keyed by the CAIP-2 `network` +# a 402 carries, each with the SDK's OWN chain id, USDC address and EIP-712 +# domain. The 402 SELECTS a network from this table; it never supplies the +# domain — until the Arc release the table fell back to Base for any network +# it did not know and took `asset` and `extra` from the 402 as given, which on +# arc.blockrun.ai signed chainId 8453 against Arc's contract: an invalid +# signature, a 401 from the facilitator, after the SDK had reported a payment. +EVM_NETWORKS: dict[str, dict] = { + "eip155:8453": { + "name": "Base", + "chain_id": BASE_CHAIN_ID, + "usdc": USDC_BASE, + "domain": { + "name": "USD Coin", + "version": "2", + "chainId": BASE_CHAIN_ID, + "verifyingContract": USDC_BASE, + }, + }, + "eip155:5042": { + "name": "Arc", + "chain_id": ARC_CHAIN_ID, + "usdc": USDC_ARC, + "domain": { + "name": "USDC", + "version": "2", + "chainId": ARC_CHAIN_ID, + "verifyingContract": USDC_ARC, + }, + }, + "eip155:84532": { + "name": "Base Sepolia", + "chain_id": BASE_SEPOLIA_CHAIN_ID, + "usdc": USDC_BASE_SEPOLIA, + "domain": { + "name": "USDC", + "version": "2", + "chainId": BASE_SEPOLIA_CHAIN_ID, + "verifyingContract": USDC_BASE_SEPOLIA, + }, + }, +} +# The pre-CAIP alias this SDK accepted for the testnet. +_NETWORK_ALIASES = {"base-sepolia": "eip155:84532"} + + +def evm_network(network: str) -> dict: + """The table entry for a 402's `network`, or a ValueError naming what IS supported.""" + net = EVM_NETWORKS.get(_NETWORK_ALIASES.get(network, network)) + if net is None: + raise ValueError( + f'Unsupported x402 network "{network}": this SDK signs USDC payments on ' + + ", ".join(EVM_NETWORKS) + ) + return net + + +# BlockRun's x402 builder code — the ERC-8021 Schema 2 service code (`s`) that +# tags every payment this SDK signs as BlockRun-originated for on-chain +# attribution. See https://docs.cdp.coinbase.com/x402/core-concepts/builder-codes +BLOCKRUN_SERVICE_CODE = "blockrun" + + +def with_builder_code_service_code( + extensions: dict[str, Any] | None, +) -> dict[str, Any]: + """Merge BlockRun's service code (``s``) into the payload's ``builder-code`` + extension, preserving any app code (``a``) the server echoed back in its 402. + + The CDP facilitator reads ``builder-code.info.s`` and encodes it into the + settlement calldata suffix — no CBOR/encoding happens client-side. + """ + merged: dict[str, Any] = dict(extensions or {}) + existing = dict(merged.get("builder-code") or {}) + info = dict(existing.get("info") or {}) + info["s"] = [BLOCKRUN_SERVICE_CODE] + existing["info"] = info + merged["builder-code"] = existing + return merged + + +def get_chain_config(network: str) -> tuple[int, str]: + """Chain ID and USDC contract for a network — see EVM_NETWORKS. Raises for an unknown one.""" + net = evm_network(network) + return net["chain_id"], net["usdc"] + + +def get_usdc_domain_name(network: str) -> str: + """The EIP-712 domain name for USDC on a network — "USD Coin" on Base, "USDC" on Arc and Base Sepolia.""" + return evm_network(network)["domain"]["name"] + + +def create_nonce() -> str: + """Generate a random bytes32 nonce.""" + return "0x" + secrets.token_hex(32) + + +def create_payment_payload( + account: Account, + recipient: str, + amount: str, # In micro USDC (6 decimals) + network: str = "eip155:8453", + resource_url: str = "https://blockrun.ai/api/v1/chat/completions", + resource_description: str = "BlockRun AI API call", + max_timeout_seconds: int = 300, + extra: dict[str, str] | None = None, + extensions: dict[str, Any] | None = None, + asset: str | None = None, +) -> str: + """ + Create a signed x402 v2 payment payload. + + This uses EIP-712 typed data signing to create a payment authorization + that the CDP facilitator can verify and settle. + + Args: + account: eth-account Account instance + recipient: Payment recipient address (checksummed) + amount: Amount in micro USDC (6 decimals, e.g., "1000" = $0.001) + network: Network identifier (e.g., "eip155:8453" for Base mainnet, "eip155:84532" for Base Sepolia) + resource_url: URL of the resource being accessed + resource_description: Description of the resource + max_timeout_seconds: Max timeout for the payment (default: 300) + extra: The 402's `extra`. Accepted for compatibility; the domain comes from EVM_NETWORKS. + asset: The 402's `asset`. Checked against the network's USDC; a mismatch raises ValueError. + + Returns: + Base64-encoded signed payment payload + """ + # Current timestamp + now = int(time.time()) + valid_after = now - 600 # 10 minutes before (allows for clock skew) + valid_before = now + max_timeout_seconds + + # Generate random nonce + nonce = create_nonce() + + # The domain is the SDK's own value for the 402's network — never the 402's + # `extra` (see EVM_NETWORKS). A 402 naming a network the table lacks, or an + # asset that is not that network's USDC, is refused rather than signed. + net = evm_network(network) + usdc_address = net["usdc"] + if asset and asset.lower() != usdc_address.lower(): + raise ValueError( + f"x402 asset mismatch: the 402 asks for {asset} on {network}, " + f"but this SDK only pays USDC there ({usdc_address})" + ) + domain = dict(net["domain"]) + + # EIP-712 types for TransferWithAuthorization + types = { + "TransferWithAuthorization": [ + {"name": "from", "type": "address"}, + {"name": "to", "type": "address"}, + {"name": "value", "type": "uint256"}, + {"name": "validAfter", "type": "uint256"}, + {"name": "validBefore", "type": "uint256"}, + {"name": "nonce", "type": "bytes32"}, + ], + } + + # Message to sign + message = { + "from": account.address, + "to": recipient, + "value": int(amount), + "validAfter": valid_after, + "validBefore": valid_before, + "nonce": bytes.fromhex(nonce[2:]), # Remove 0x prefix + } + + # Sign using EIP-712 + signable = encode_typed_data(domain, types, message) + signed = account.sign_message(signable) + + # Create x402 v2 payment payload + payment_data = { + "x402Version": 2, + "resource": { + "url": resource_url, + "description": resource_description, + "mimeType": "application/json", + }, + "accepted": { + "scheme": "exact", + "network": network, + "amount": amount, + "asset": usdc_address, + "payTo": recipient, + "maxTimeoutSeconds": max_timeout_seconds, + "extra": {"name": domain["name"], "version": domain["version"]}, + }, + "payload": { + "signature": ( + "0x" + signed.signature.hex() + if not signed.signature.hex().startswith("0x") + else signed.signature.hex() + ), + "authorization": { + "from": account.address, + "to": recipient, + "value": amount, + "validAfter": str(valid_after), + "validBefore": str(valid_before), + "nonce": nonce, + }, + }, + "extensions": with_builder_code_service_code(extensions), + } + + # Encode as base64 + return base64.b64encode(json.dumps(payment_data).encode()).decode() + + +def parse_payment_required(header_value: str) -> dict[str, Any]: + """ + Parse the X-Payment-Required header from a 402 response. + + Args: + header_value: Base64-encoded payment requirements + + Returns: + Decoded payment requirements dict + """ + try: + decoded = base64.b64decode(header_value) + return json.loads(decoded) + except Exception: + # Don't expose internal error details + raise ValueError("Failed to parse payment required header: invalid format") + + +def extract_payment_details(payment_required: dict[str, Any]) -> dict[str, Any]: + """ + Extract payment details from parsed payment required response. + + Supports both v1 and v2 formats. + + Args: + payment_required: Parsed payment required dict + + Returns: + Dict with amount, recipient, network, asset, and extra info + """ + accepts = payment_required.get("accepts", []) + if not accepts: + raise ValueError("No payment options in payment required response") + + # Take the first option + option = accepts[0] + + # Support both v1 (maxAmountRequired) and v2 (amount) formats + amount = option.get("amount") or option.get("maxAmountRequired") + if not amount: + raise ValueError("No amount found in payment requirements") + + return { + "amount": amount, + "recipient": option.get("payTo"), + "network": option.get("network"), + "asset": option.get("asset"), + "scheme": option.get("scheme"), + "maxTimeoutSeconds": option.get("maxTimeoutSeconds", 300), + "extra": option.get("extra"), + "resource": payment_required.get("resource"), + } + + +# ============================================================ +# Solana x402 Payment — delegated to official x402 SDK +# ============================================================ +# The Solana payment implementation has been replaced by the +# official x402 Python SDK (pip install x402[svm]). +# See solana_client.py for usage. + + +def is_solana_network(network: str) -> bool: + """Check if a network string represents Solana.""" + return network.startswith("solana:") diff --git a/brand-numbers.json b/brand-numbers.json new file mode 100644 index 0000000..42af21d --- /dev/null +++ b/brand-numbers.json @@ -0,0 +1,37 @@ +{ + "$schema": "https://blockrun.ai/brand/numbers.schema.json", + "version": 1, + "models": { + "chatVisible": 78, + "totalVisible": 105, + "free": 6, + "freeWithheld": 27, + "image": 12, + "video": 8, + "music": 1, + "speech": 5, + "soundfx": 1, + "withFallback": 23, + "withFallbackAllEntries": 62 + }, + "clawrouter": { + "dimensions": 15, + "tiers": 4, + "profiles": 4, + "aliases": 259 + }, + "mcp": { + "tools": 19, + "contextTokens": 12767, + "contextTokensTrading": 5270, + "contextCutPct": 59 + }, + "chains": { + "rpc": 40 + }, + "savings": { + "baselineModel": "anthropic/claude-opus-5", + "ecoVsBaselinePct": 98, + "autoVsBaselinePct": 84 + } +} diff --git a/docs/plans/2026-02-27-solana-client.md b/docs/plans/2026-02-27-solana-client.md new file mode 100644 index 0000000..e0fb944 --- /dev/null +++ b/docs/plans/2026-02-27-solana-client.md @@ -0,0 +1,1010 @@ +# SolanaLLMClient Python SDK Implementation Plan + +> **For Claude:** REQUIRED SUB-SKILL: Use superpowers:executing-plans to implement this plan task-by-task. + +**Goal:** Add `SolanaLLMClient` class to `blockrun-llm` Python SDK so Solana developers can pay for AI calls with Solana USDC via x402. + +**Architecture:** New `SolanaLLMClient` class in `blockrun_llm/solana_client.py` (mirrors `LLMClient` but uses Solana keypair). New `create_solana_payment_payload` in `blockrun_llm/x402.py`. Solana keypair management in `blockrun_llm/solana_wallet.py`. Uses `solders` for keypair/transaction and `httpx` (already a dep) for Solana RPC. All existing Base/EVM code is untouched. + +**Tech Stack:** Python 3.9+, `solders>=0.21.0` (new optional dep), `httpx` (already a dep), `base58>=2.1.0` (new optional dep) + +--- + +### Task 1: Add Solana deps to pyproject.toml + +**Files:** +- Modify: `pyproject.toml` + +**Step 1: Add optional Solana deps** + +In `[project.optional-dependencies]`, add: + +```toml +[project.optional-dependencies] +dev = [ + "pytest>=7.0.0", + "pytest-asyncio>=0.21.0", + "black==24.10.0", + "mypy>=1.0.0", + "ruff>=0.1.0", +] +solana = [ + "solders>=0.21.0", + "base58>=2.1.0", +] +``` + +**Step 2: Install Solana extras in dev env** + +```bash +cd /Users/vickyfu/Documents/blockrun-web/blockrun-llm +source /Users/vickyfu/myenv_py313/bin/activate +pip install solders>=0.21.0 base58>=2.1.0 +``` +Expected: both packages install + +**Step 3: Commit** + +```bash +git add pyproject.toml +git commit -m "feat: add optional Solana dependencies to pyproject.toml" +``` + +--- + +### Task 2: Add Solana wallet utilities + +**Files:** +- Create: `blockrun_llm/solana_wallet.py` +- Test: `tests/unit/test_solana_wallet.py` + +**Context:** Solana keys are bs58-encoded 64-byte secret keys. Address is base58 public key. Key stored at `~/.blockrun/.solana-session`. + +**Step 1: Write failing tests** + +Create `tests/unit/test_solana_wallet.py`: + +```python +"""Unit tests for Solana wallet utilities.""" +import pytest +from blockrun_llm.solana_wallet import ( + create_solana_wallet, + solana_key_to_bytes, + get_solana_public_key, +) + +# A valid test bs58 secret key (64 bytes encoded) +TEST_BS58_KEY = "5MaiiCavjCmn9Hs1o3eznqDEhRwxo7pXiAYez7keQUviQeRjpzKCY8trDwpvBMTKTpNFbCJsBZthJ4tCs6o62rr" + + +class TestCreateSolanaWallet: + def test_returns_address_and_key(self): + wallet = create_solana_wallet() + assert "address" in wallet + assert "private_key" in wallet + assert len(wallet["address"]) >= 32 # base58 pubkey + assert len(wallet["private_key"]) >= 86 # bs58 64-byte key + + def test_unique_wallets(self): + w1 = create_solana_wallet() + w2 = create_solana_wallet() + assert w1["address"] != w2["address"] + assert w1["private_key"] != w2["private_key"] + + +class TestSolanaKeyToBytes: + def test_valid_key(self): + b = solana_key_to_bytes(TEST_BS58_KEY) + assert isinstance(b, bytes) + assert len(b) == 64 + + def test_invalid_key_raises(self): + with pytest.raises(ValueError, match="Invalid Solana private key"): + solana_key_to_bytes("not-a-valid-key!!!") + + +class TestGetSolanaPublicKey: + def test_returns_base58_address(self): + addr = get_solana_public_key(TEST_BS58_KEY) + assert isinstance(addr, str) + assert len(addr) >= 32 + # Should be valid base58 (only alphanumeric, no 0/O/I/l) + import re + assert re.match(r'^[1-9A-HJ-NP-Za-km-z]+$', addr) +``` + +**Step 2: Run to verify fails** + +```bash +source /Users/vickyfu/myenv_py313/bin/activate +cd /Users/vickyfu/Documents/blockrun-web/blockrun-llm +pytest tests/unit/test_solana_wallet.py -v +``` +Expected: FAIL "ImportError: cannot import name" + +**Step 3: Implement `blockrun_llm/solana_wallet.py`** + +```python +""" +BlockRun Solana Wallet Management. + +Stores keys as bs58-encoded strings at ~/.blockrun/.solana-session. +Requires: solders>=0.21.0, base58>=2.1.0 +""" +from __future__ import annotations + +import os +from pathlib import Path +from typing import Dict, Optional + +WALLET_DIR = Path.home() / ".blockrun" +SOLANA_WALLET_FILE = WALLET_DIR / ".solana-session" + + +def _require_solders() -> None: + try: + import solders # noqa: F401 + except ImportError: + raise ImportError( + "Solana support requires 'solders' and 'base58' packages. " + "Install with: pip install blockrun-llm[solana]" + ) + + +def create_solana_wallet() -> Dict[str, str]: + """ + Create a new Solana wallet. + + Returns: + Dict with 'address' (base58 pubkey) and 'private_key' (bs58 secret key) + """ + _require_solders() + from solders.keypair import Keypair # type: ignore + import base58 # type: ignore + + kp = Keypair() + secret = bytes(kp) # 64 bytes + return { + "address": str(kp.pubkey()), + "private_key": base58.b58encode(secret).decode(), + } + + +def solana_key_to_bytes(private_key: str) -> bytes: + """ + Convert a bs58 private key string to bytes (64 bytes). + + Args: + private_key: bs58-encoded 64-byte Solana secret key + + Returns: + 64-byte secret key as bytes + + Raises: + ValueError: If key is invalid + """ + try: + import base58 # type: ignore + decoded = base58.b58decode(private_key) + if len(decoded) != 64: + raise ValueError(f"Expected 64 bytes, got {len(decoded)}") + return decoded + except Exception as e: + raise ValueError(f"Invalid Solana private key: {e}") from e + + +def get_solana_public_key(private_key: str) -> str: + """ + Get the Solana public key (address) from a bs58 private key. + + Args: + private_key: bs58-encoded 64-byte Solana secret key + + Returns: + Base58 public key string + """ + _require_solders() + from solders.keypair import Keypair # type: ignore + + secret = solana_key_to_bytes(private_key) + kp = Keypair.from_bytes(secret) + return str(kp.pubkey()) + + +def save_solana_wallet(private_key: str) -> Path: + WALLET_DIR.mkdir(exist_ok=True) + SOLANA_WALLET_FILE.write_text(private_key) + SOLANA_WALLET_FILE.chmod(0o600) + return SOLANA_WALLET_FILE + + +def load_solana_wallet() -> Optional[str]: + if SOLANA_WALLET_FILE.exists(): + key = SOLANA_WALLET_FILE.read_text().strip() + if key: + return key + return None + + +def get_or_create_solana_wallet() -> Dict[str, object]: + """ + Get existing Solana wallet or create new one. + + Priority: SOLANA_WALLET_KEY env var → ~/.blockrun/.solana-session → create new + + Returns: + Dict with 'address', 'private_key', 'is_new' + """ + env_key = os.environ.get("SOLANA_WALLET_KEY") + if env_key: + return {"private_key": env_key, "address": get_solana_public_key(env_key), "is_new": False} + + file_key = load_solana_wallet() + if file_key: + return {"private_key": file_key, "address": get_solana_public_key(file_key), "is_new": False} + + wallet = create_solana_wallet() + save_solana_wallet(wallet["private_key"]) + return {**wallet, "is_new": True} +``` + +**Step 4: Run tests** + +```bash +pytest tests/unit/test_solana_wallet.py -v +``` +Expected: PASS (5 tests) + +**Step 5: Commit** + +```bash +git add blockrun_llm/solana_wallet.py tests/unit/test_solana_wallet.py +git commit -m "feat: add Solana wallet utilities" +``` + +--- + +### Task 3: Add create_solana_payment_payload to x402.py + +**Files:** +- Modify: `blockrun_llm/x402.py` +- Test: `tests/unit/test_x402.py` + +**Context:** The 402 response from `sol.blockrun.ai` has: +- `network: "solana:5eykt4UsFv8P8NJdTREpY1vzqKqZKvdp"` +- `amount: "1000"` (micro USDC, 6 decimals) +- `payTo: "AQqnMFBwGZEoti85aTVRy8XYpKrho7GaMDx9ZB3CEeKA"` +- `asset: "EPjFWdd5AufqSSqeM2qN1xzybapC8G4wEGGkZwyTDt1v"` (Solana USDC) +- `extra.feePayer: "2wKupLR9q6wXYppw8Gr2NvWxKBUqm4PPJKkQfoxHDBg4"` (CDP fee payer) + +The transaction is an SPL Token TransferChecked with compute budget instructions, signed by the user's keypair. The feePayer (CDP) will co-sign on the server side. + +The ATA (Associated Token Account) is derived as PDA: `[owner_bytes, TOKEN_PROGRAM_ID_bytes, mint_bytes]` under ASSOCIATED_TOKEN_PROGRAM_ID. + +**Step 1: Write failing tests** + +Add to `tests/unit/test_x402.py`: + +```python +# Add at the end of test_x402.py: + +class TestCreateSolanaPaymentPayload: + """Tests for Solana payment payload creation.""" + + TEST_BS58_KEY = "5MaiiCavjCmn9Hs1o3eznqDEhRwxo7pXiAYez7keQUviQeRjpzKCY8trDwpvBMTKTpNFbCJsBZthJ4tCs6o62rr" + TEST_FEE_PAYER = "2wKupLR9q6wXYppw8Gr2NvWxKBUqm4PPJKkQfoxHDBg4" + TEST_RECIPIENT = "AQqnMFBwGZEoti85aTVRy8XYpKrho7GaMDx9ZB3CEeKA" + + def test_payload_structure(self): + """Should create valid Solana payment payload.""" + from blockrun_llm.x402 import create_solana_payment_payload + import json, base64 + + payload = create_solana_payment_payload( + private_key=self.TEST_BS58_KEY, + recipient=self.TEST_RECIPIENT, + amount="1000", + fee_payer=self.TEST_FEE_PAYER, + ) + + assert isinstance(payload, str) + decoded = json.loads(base64.b64decode(payload)) + assert decoded["x402Version"] == 2 + assert "transaction" in decoded["payload"] + assert decoded["accepted"]["network"].startswith("solana:") + assert decoded["accepted"]["asset"] == "EPjFWdd5AufqSSqeM2qN1xzybapC8G4wEGGkZwyTDt1v" + + def test_payload_transaction_is_base64(self): + """Transaction field should be base64-encoded.""" + from blockrun_llm.x402 import create_solana_payment_payload + import json, base64 + + payload = create_solana_payment_payload( + private_key=self.TEST_BS58_KEY, + recipient=self.TEST_RECIPIENT, + amount="1000", + fee_payer=self.TEST_FEE_PAYER, + ) + decoded = json.loads(base64.b64decode(payload)) + # Should be valid base64 + tx_bytes = base64.b64decode(decoded["payload"]["transaction"]) + assert len(tx_bytes) > 0 +``` + +**Step 2: Run to verify fails** + +```bash +pytest tests/unit/test_x402.py::TestCreateSolanaPaymentPayload -v +``` +Expected: FAIL "cannot import name 'create_solana_payment_payload'" + +**Step 3: Add `create_solana_payment_payload` to `blockrun_llm/x402.py`** + +Append to the end of `blockrun_llm/x402.py`: + +```python +# ============================================================ +# Solana x402 Payment +# ============================================================ + +SOLANA_NETWORK = "solana:5eykt4UsFv8P8NJdTREpY1vzqKqZKvdp" +USDC_SOLANA = "EPjFWdd5AufqSSqeM2qN1xzybapC8G4wEGGkZwyTDt1v" + +# SPL program IDs +TOKEN_PROGRAM_ID = "TokenkegQfeZyiNwAJbNbGKPFXCWuBvf9Ss623VQ5DA" +ASSOCIATED_TOKEN_PROGRAM_ID = "ATokenGPvbdGVxr1b2hvZbsiqW5xWH25efTNsLJe1bRS" + +# Compute budget defaults (match @x402/svm) +DEFAULT_COMPUTE_UNIT_PRICE_MICROLAMPORTS = 1 +DEFAULT_COMPUTE_UNIT_LIMIT = 8000 + + +def _get_ata(owner: str, mint: str) -> str: + """Derive Associated Token Account address.""" + from solders.pubkey import Pubkey # type: ignore + + owner_pk = Pubkey.from_string(owner) + mint_pk = Pubkey.from_string(mint) + token_program = Pubkey.from_string(TOKEN_PROGRAM_ID) + assoc_program = Pubkey.from_string(ASSOCIATED_TOKEN_PROGRAM_ID) + + seeds = [bytes(owner_pk), bytes(token_program), bytes(mint_pk)] + ata, _ = Pubkey.find_program_address(seeds, assoc_program) + return str(ata) + + +def _get_latest_blockhash(rpc_url: str) -> str: + """Fetch latest blockhash from Solana RPC.""" + import httpx + resp = httpx.post( + rpc_url, + json={"jsonrpc": "2.0", "id": 1, "method": "getLatestBlockhash", + "params": [{"commitment": "finalized"}]}, + timeout=10, + ) + resp.raise_for_status() + return resp.json()["result"]["value"]["blockhash"] + + +def create_solana_payment_payload( + private_key: str, + recipient: str, + amount: str, + fee_payer: str, + resource_url: str = "https://sol.blockrun.ai/api/v1/chat/completions", + resource_description: str = "BlockRun Solana AI API call", + max_timeout_seconds: int = 300, + extra: Optional[Dict[str, Any]] = None, + extensions: Optional[Dict[str, Any]] = None, + rpc_url: str = "https://api.mainnet-beta.solana.com", +) -> str: + """ + Create a signed Solana x402 v2 payment payload. + + Builds an SPL TransferChecked transaction signed by the user's Solana keypair. + The CDP facilitator (feePayer) co-signs on the server side. + + Args: + private_key: bs58-encoded 64-byte Solana secret key + recipient: Payment recipient Solana address (base58) + amount: Amount in micro USDC (6 decimals, e.g. "1000" = $0.001) + fee_payer: CDP facilitator address that pays SOL transaction fees (base58) + resource_url: URL of the resource being accessed + resource_description: Description for the payment + max_timeout_seconds: Max timeout for the payment + extra: Extra info included in payment (e.g. feePayer) + extensions: x402 extensions dict + rpc_url: Solana RPC endpoint + + Returns: + Base64-encoded signed payment payload + """ + try: + from solders.keypair import Keypair # type: ignore + from solders.pubkey import Pubkey # type: ignore + from solders.hash import Hash # type: ignore + from solders.instruction import Instruction, AccountMeta # type: ignore + from solders.message import MessageV0 # type: ignore + from solders.transaction import VersionedTransaction # type: ignore + import base58 # type: ignore + except ImportError: + raise ImportError( + "Solana payment requires 'solders' and 'base58'. " + "Install with: pip install blockrun-llm[solana]" + ) + + # Load keypair + secret = base58.b58decode(private_key) + keypair = Keypair.from_bytes(secret) + owner_pubkey = keypair.pubkey() + + # Derive ATAs + source_ata = _get_ata(str(owner_pubkey), USDC_SOLANA) + dest_ata = _get_ata(recipient, USDC_SOLANA) + + # Get latest blockhash + blockhash = _get_latest_blockhash(rpc_url) + + # Build compute budget instructions + # ComputeBudgetProgram.setComputeUnitLimit + compute_budget_id = Pubkey.from_string("ComputeBudget111111111111111111111111111111") + + # setComputeUnitLimit instruction: discriminator=2, units=u32 LE + import struct + limit_data = bytes([2]) + struct.pack(" bool: + """Check if a network string represents Solana.""" + return network.startswith("solana:") + + +def extract_solana_payment_details(payment_required: Dict[str, Any]) -> Dict[str, Any]: + """ + Extract Solana payment details from a 402 response. + Finds the Solana network option in accepts[]. + """ + accepts = payment_required.get("accepts", []) + option = next((o for o in accepts if is_solana_network(o.get("network", ""))), None) + if not option: + raise ValueError("No Solana payment option found in 402 response") + + amount = option.get("amount") or option.get("maxAmountRequired") + if not amount: + raise ValueError("No amount in Solana payment requirements") + + return { + "amount": amount, + "recipient": option.get("payTo"), + "network": option.get("network"), + "asset": option.get("asset"), + "max_timeout_seconds": option.get("maxTimeoutSeconds", 300), + "extra": option.get("extra", {}), + "resource": payment_required.get("resource"), + } +``` + +**Step 4: Run tests** + +```bash +pytest tests/unit/test_x402.py::TestCreateSolanaPaymentPayload -v +``` +Expected: PASS (2 tests) + +**Step 5: Commit** + +```bash +git add blockrun_llm/x402.py tests/unit/test_x402.py +git commit -m "feat: add create_solana_payment_payload to x402" +``` + +--- + +### Task 4: Add SolanaLLMClient class + +**Files:** +- Create: `blockrun_llm/solana_client.py` +- Test: `tests/unit/test_solana_client.py` + +**Step 1: Write failing tests** + +Create `tests/unit/test_solana_client.py`: + +```python +"""Unit tests for SolanaLLMClient.""" +import pytest +import os +from blockrun_llm.solana_client import SolanaLLMClient + +TEST_BS58_KEY = "5MaiiCavjCmn9Hs1o3eznqDEhRwxo7pXiAYez7keQUviQeRjpzKCY8trDwpvBMTKTpNFbCJsBZthJ4tCs6o62rr" + + +class TestSolanaLLMClientInit: + def test_init_with_key(self): + client = SolanaLLMClient(private_key=TEST_BS58_KEY) + assert client is not None + + def test_init_from_env(self): + os.environ["SOLANA_WALLET_KEY"] = TEST_BS58_KEY + client = SolanaLLMClient() + assert client is not None + del os.environ["SOLANA_WALLET_KEY"] + + def test_raises_without_key(self): + saved = os.environ.pop("SOLANA_WALLET_KEY", None) + with pytest.raises(ValueError, match="private key required"): + SolanaLLMClient() + if saved: + os.environ["SOLANA_WALLET_KEY"] = saved + + def test_default_api_url(self): + client = SolanaLLMClient(private_key=TEST_BS58_KEY) + assert client.is_solana() + + def test_custom_api_url(self): + client = SolanaLLMClient( + private_key=TEST_BS58_KEY, + api_url="https://custom.example.com/api" + ) + assert not client.is_solana() + + def test_get_wallet_address(self): + client = SolanaLLMClient(private_key=TEST_BS58_KEY) + addr = client.get_wallet_address() + assert isinstance(addr, str) + assert len(addr) >= 32 + + def test_get_spending_initial(self): + client = SolanaLLMClient(private_key=TEST_BS58_KEY) + spending = client.get_spending() + assert spending["total_usd"] == 0.0 + assert spending["calls"] == 0 +``` + +**Step 2: Run to verify fails** + +```bash +pytest tests/unit/test_solana_client.py -v +``` +Expected: FAIL "No module named 'blockrun_llm.solana_client'" + +**Step 3: Implement `blockrun_llm/solana_client.py`** + +```python +""" +BlockRun Solana LLM Client. + +Usage: + from blockrun_llm import SolanaLLMClient + + # SOLANA_WALLET_KEY env var (bs58-encoded Solana secret key) + client = SolanaLLMClient() + + # Or pass key directly + client = SolanaLLMClient(private_key="your-bs58-key") + + # Same API as LLMClient + response = client.chat("openai/gpt-4o", "gm Solana") + print(response) +""" +from __future__ import annotations + +import os +from typing import Any, Dict, List, Optional + +import httpx + +from .types import ChatResponse, APIError, PaymentError +from .x402 import ( + create_solana_payment_payload, + extract_solana_payment_details, + parse_payment_required, + SOLANA_NETWORK, +) +from .solana_wallet import get_solana_public_key +from .validation import validate_api_url, sanitize_error_response, validate_resource_url + +SOLANA_API_URL = "https://sol.blockrun.ai/api" +DEFAULT_MAX_TOKENS = 1024 +DEFAULT_TIMEOUT = 60.0 + + +def _get_user_agent() -> str: + from . import __version__ + return f"blockrun-python/{__version__}" + + +class SolanaLLMClient: + """ + BlockRun LLM Client for Solana — pays via Solana USDC x402. + + Connects to sol.blockrun.ai by default. + """ + + SOLANA_API_URL = SOLANA_API_URL + + def __init__( + self, + private_key: Optional[str] = None, + api_url: str = SOLANA_API_URL, + rpc_url: str = "https://api.mainnet-beta.solana.com", + timeout: float = DEFAULT_TIMEOUT, + ) -> None: + key = private_key or os.environ.get("SOLANA_WALLET_KEY") + if not key: + raise ValueError( + "Private key required. Pass private_key or set SOLANA_WALLET_KEY env var." + ) + self._private_key = key + validate_api_url(api_url) + self._api_url = api_url.rstrip("/") + self._rpc_url = rpc_url + self._timeout = timeout + self._session_total_usd = 0.0 + self._session_calls = 0 + self._address: Optional[str] = None + + def get_wallet_address(self) -> str: + if not self._address: + self._address = get_solana_public_key(self._private_key) + return self._address + + def is_solana(self) -> bool: + return "sol.blockrun.ai" in self._api_url + + def get_spending(self) -> Dict[str, Any]: + return {"total_usd": self._session_total_usd, "calls": self._session_calls} + + def chat( + self, + model: str, + prompt: str, + system: Optional[str] = None, + max_tokens: int = DEFAULT_MAX_TOKENS, + temperature: Optional[float] = None, + search: bool = False, + ) -> str: + """Simple 1-line chat.""" + messages: List[Dict[str, str]] = [] + if system: + messages.append({"role": "system", "content": system}) + messages.append({"role": "user", "content": prompt}) + result = self.chat_completion( + model, messages, + max_tokens=max_tokens, + temperature=temperature, + search=search, + ) + return result.choices[0].message.content or "" + + def chat_completion( + self, + model: str, + messages: List[Dict[str, Any]], + max_tokens: int = DEFAULT_MAX_TOKENS, + temperature: Optional[float] = None, + top_p: Optional[float] = None, + search: bool = False, + search_parameters: Optional[Dict[str, Any]] = None, + ) -> ChatResponse: + """Full chat completion (OpenAI-compatible).""" + body: Dict[str, Any] = {"model": model, "messages": messages, "max_tokens": max_tokens} + if temperature is not None: + body["temperature"] = temperature + if top_p is not None: + body["top_p"] = top_p + if search_parameters: + body["search_parameters"] = search_parameters + elif search: + body["search_parameters"] = {"mode": "on"} + return self._request_with_payment("/v1/chat/completions", body) + + def list_models(self) -> List[Dict[str, Any]]: + with httpx.Client(timeout=self._timeout) as http: + resp = http.get(f"{self._api_url}/v1/models") + resp.raise_for_status() + return resp.json().get("data", []) + + def _request_with_payment( + self, endpoint: str, body: Dict[str, Any] + ) -> ChatResponse: + url = f"{self._api_url}{endpoint}" + headers = {"Content-Type": "application/json", "User-Agent": _get_user_agent()} + + with httpx.Client(timeout=self._timeout) as http: + response = http.post(url, json=body, headers=headers) + + if response.status_code == 402: + return self._handle_payment_and_retry(url, body, response) + + if not response.is_success: + try: + error_body = response.json() + except Exception: + error_body = {"error": "Request failed"} + raise APIError( + f"API error: {response.status_code}", + response.status_code, + sanitize_error_response(error_body), + ) + + return ChatResponse(**response.json()) + + def _handle_payment_and_retry( + self, url: str, body: Dict[str, Any], response: httpx.Response + ) -> ChatResponse: + # Get payment header + payment_header = response.headers.get("payment-required") + if not payment_header: + try: + resp_body = response.json() + if resp_body.get("accepts") or resp_body.get("x402Version"): + import base64, json + payment_header = base64.b64encode( + json.dumps(resp_body).encode() + ).decode() + except Exception: + pass + + if not payment_header: + raise PaymentError("402 response but no payment requirements found") + + payment_required = parse_payment_required(payment_header) + details = extract_solana_payment_details(payment_required) + + if not details["network"].startswith("solana:"): + raise PaymentError( + f"Expected Solana network, got: {details['network']}. " + "Use LLMClient for Base payments." + ) + + fee_payer = (details.get("extra") or {}).get("feePayer") + if not fee_payer: + raise PaymentError("Missing feePayer in 402 extra field") + + resource_info = details.get("resource") or {} + resource_url = validate_resource_url( + resource_info.get("url") or f"{self._api_url}/v1/chat/completions", + self._api_url, + ) + + payment_payload = create_solana_payment_payload( + private_key=self._private_key, + recipient=details["recipient"], + amount=details["amount"], + fee_payer=fee_payer, + resource_url=resource_url, + resource_description=resource_info.get("description") or "BlockRun Solana AI API call", + max_timeout_seconds=details["max_timeout_seconds"], + extra=details.get("extra"), + rpc_url=self._rpc_url, + ) + + headers = { + "Content-Type": "application/json", + "User-Agent": _get_user_agent(), + "PAYMENT-SIGNATURE": payment_payload, + } + + with httpx.Client(timeout=self._timeout) as http: + retry_response = http.post(url, json=body, headers=headers) + + if retry_response.status_code == 402: + raise PaymentError("Payment rejected. Check your Solana USDC balance.") + + if not retry_response.is_success: + try: + error_body = retry_response.json() + except Exception: + error_body = {"error": "Request failed"} + raise APIError( + f"API error after payment: {retry_response.status_code}", + retry_response.status_code, + sanitize_error_response(error_body), + ) + + cost_usd = float(details["amount"]) / 1e6 + self._session_calls += 1 + self._session_total_usd += cost_usd + + return ChatResponse(**retry_response.json()) +``` + +**Step 4: Run tests** + +```bash +pytest tests/unit/test_solana_client.py -v +``` +Expected: PASS (7 tests) + +**Step 5: Commit** + +```bash +git add blockrun_llm/solana_client.py tests/unit/test_solana_client.py +git commit -m "feat: add SolanaLLMClient for Solana USDC payments" +``` + +--- + +### Task 5: Update exports and version + +**Files:** +- Modify: `blockrun_llm/__init__.py` +- Modify: `pyproject.toml` (version bump) + +**Step 1: Read current `__init__.py` exports** + +```bash +head -60 blockrun_llm/__init__.py +``` + +**Step 2: Add SolanaLLMClient to exports** + +In `blockrun_llm/__init__.py`, add alongside `LLMClient`: + +```python +from .solana_client import SolanaLLMClient +``` + +And in `__all__` (if present): +```python +"SolanaLLMClient", +``` + +**Step 3: Bump version in `pyproject.toml`** + +Change `version = "0.4.1"` → `version = "0.5.0"` (minor bump, new feature). + +Also update `blockrun_llm/__init__.py` version string if present. + +**Step 4: Run full test suite** + +```bash +pytest tests/unit/ -v +``` +Expected: all pass + +**Step 5: Commit** + +```bash +git add blockrun_llm/__init__.py pyproject.toml +git commit -m "feat: export SolanaLLMClient and bump to 0.5.0" +``` + +--- + +### Task 6: Update README and publish + +**Files:** +- Modify: `README.md` + +**Step 1: Add Solana section to README** + +Find the "Supported Chains" table and update it: + +```markdown +| Chain | Network | Payment | Status | +|-------|---------|---------|--------| +| **Base** | Base Mainnet (Chain ID: 8453) | USDC | ✅ Primary | +| **Base Testnet** | Base Sepolia (Chain ID: 84532) | Testnet USDC | ✅ Development | +| **Solana** | Solana Mainnet | USDC (SPL) | ✅ New | +``` + +Add a new section after Quick Start: + +```markdown +## Solana Support + +Pay for AI calls with Solana USDC via [sol.blockrun.ai](https://sol.blockrun.ai): + +\`\`\`python +from blockrun_llm import SolanaLLMClient + +# SOLANA_WALLET_KEY env var (bs58-encoded Solana secret key) +client = SolanaLLMClient() + +# Or pass key directly +client = SolanaLLMClient(private_key="your-bs58-solana-key") + +# Same API as LLMClient +response = client.chat("openai/gpt-4o", "gm Solana") +print(response) + +# Live Search with Grok (Solana payment) +tweet = client.chat("xai/grok-3-mini", "What is trending on X?", search=True) +\`\`\` + +**Setup:** +\`\`\`bash +pip install blockrun-llm[solana] +export SOLANA_WALLET_KEY="your-bs58-solana-key" +\`\`\` + +**Endpoint:** `https://sol.blockrun.ai/api` +**Payment:** Solana USDC (SPL Token, mainnet) +``` + +**Step 2: Commit** + +```bash +git add README.md +git commit -m "docs: add Solana section to README" +``` + +**Step 3: Build and publish** + +```bash +source /Users/vickyfu/myenv_py313/bin/activate +pip install build +python -m build +pip install twine +twine upload dist/blockrun_llm-0.5.0* +``` + +**Step 4: Push to GitHub** + +```bash +git push +``` diff --git a/examples/arbitrage_analyzer.py b/examples/arbitrage_analyzer.py new file mode 100644 index 0000000..477fec8 --- /dev/null +++ b/examples/arbitrage_analyzer.py @@ -0,0 +1,278 @@ +""" +BlockRun Integration Example: Crypto Arbitrage Analysis + +This example shows how to integrate BlockRun's AI capabilities into +a cryptocurrency arbitrage bot (like Polymarket-Kalshi BTC arbitrage). + +Setup: + pip install blockrun-llm + export BASE_CHAIN_WALLET_KEY=0x... # Your Base wallet private key + +Usage: + python arbitrage_analyzer.py +""" + +from dataclasses import dataclass + +from blockrun_llm import APIError, AsyncLLMClient, LLMClient, PaymentError + + +@dataclass +class ArbitrageOpportunity: + """Represents a detected arbitrage opportunity.""" + + platform_a: str + platform_b: str + price_a: float # e.g., 0.52 (52% probability) + price_b: float # e.g., 0.47 (47% probability) + spread: float # Combined cost below $1.00 + expiry: str + market: str # e.g., "BTC > $100,000" + + +class ArbitrageAnalyzer: + """ + AI-powered arbitrage opportunity analyzer using BlockRun. + + Provides risk assessment, market sentiment, and execution recommendations + for detected arbitrage opportunities. + """ + + # Model recommendations by use case + MODELS = { + "fast": "openai/gpt-5.4-nano", # $0.20/M input - quick analysis + "balanced": "anthropic/claude-haiku-4.5", # $1.00/M input - good reasoning + "deep": "anthropic/claude-sonnet-4.6", # $3.00/M input - thorough analysis + "frontier": "openai/gpt-5.5", # $5.00/M input - latest capabilities (1M context) + } + + def __init__(self, model_tier: str = "fast"): + """ + Initialize the analyzer. + + Args: + model_tier: One of "fast", "balanced", "deep", "frontier" + """ + self.client = LLMClient() + self.model = self.MODELS.get(model_tier, self.MODELS["fast"]) + + def analyze_opportunity(self, opp: ArbitrageOpportunity) -> dict: + """ + Analyze an arbitrage opportunity for risk and execution. + + Args: + opp: The detected arbitrage opportunity + + Returns: + Analysis dict with risk_score, recommendation, and reasoning + """ + prompt = f"""Analyze this prediction market arbitrage opportunity: + +Market: {opp.market} +Platform A ({opp.platform_a}): {opp.price_a:.2%} probability +Platform B ({opp.platform_b}): {opp.price_b:.2%} probability +Combined cost: ${opp.spread:.4f} (potential profit: ${1 - opp.spread:.4f}) +Expiry: {opp.expiry} + +Evaluate: +1. Is this spread large enough to be worth executing after fees? +2. What are the execution risks (slippage, timing, liquidity)? +3. Any concerns about the market or timing? + +Provide a risk score (1-10, 10=highest risk) and clear recommendation.""" + + try: + response = self.client.chat( + self.model, + prompt, + system="You are a quantitative trading analyst specializing in prediction market arbitrage. Be concise and actionable.", + ) + + return { + "success": True, + "analysis": response, + "model": self.model, + "cost_estimate": "~$0.001-0.01", + } + + except PaymentError as e: + return {"success": False, "error": f"Payment failed - check USDC balance: {e}"} + except APIError as e: + return {"success": False, "error": f"API error: {e}"} + + def get_market_sentiment(self, asset: str = "BTC") -> dict: + """ + Get AI-powered market sentiment analysis. + + Args: + asset: The asset to analyze (default: BTC) + + Returns: + Sentiment analysis dict + """ + prompt = f"""What is the current market sentiment for {asset}? + +Consider: +- Recent price action and trends +- Market structure (support/resistance levels) +- Macro factors affecting crypto +- Any upcoming events that could impact prices + +Provide a sentiment score (-100 to +100) and brief reasoning.""" + + try: + response = self.client.chat( + self.model, + prompt, + system="You are a crypto market analyst. Provide objective, data-driven analysis.", + ) + + return {"success": True, "asset": asset, "sentiment": response, "model": self.model} + + except (PaymentError, APIError) as e: + return {"success": False, "error": str(e)} + + def compare_opportunities(self, opportunities: list[ArbitrageOpportunity]) -> dict: + """ + Rank multiple opportunities by risk-adjusted return. + + Args: + opportunities: List of detected opportunities + + Returns: + Ranked list with recommendations + """ + opp_descriptions = "\n".join( + [ + f"{i+1}. {o.market}: {o.platform_a} @ {o.price_a:.2%} vs {o.platform_b} @ {o.price_b:.2%}, " + f"spread: ${o.spread:.4f}, expires: {o.expiry}" + for i, o in enumerate(opportunities) + ] + ) + + prompt = f"""Rank these arbitrage opportunities by risk-adjusted return: + +{opp_descriptions} + +Consider: +- Profit potential vs execution risk +- Time to expiry +- Liquidity concerns +- Market volatility + +Return a ranked list with brief reasoning for each.""" + + try: + response = self.client.chat( + self.model, + prompt, + system="You are a quantitative trading analyst. Rank opportunities objectively.", + ) + + return { + "success": True, + "ranking": response, + "count": len(opportunities), + "model": self.model, + } + + except (PaymentError, APIError) as e: + return {"success": False, "error": str(e)} + + +class AsyncArbitrageAnalyzer: + """ + Async version for high-throughput analysis. + + Use this when analyzing multiple opportunities concurrently. + """ + + MODELS = ArbitrageAnalyzer.MODELS + + def __init__(self, model_tier: str = "fast"): + self.model = self.MODELS.get(model_tier, self.MODELS["fast"]) + + async def analyze_batch(self, opportunities: list[ArbitrageOpportunity]) -> list[dict]: + """ + Analyze multiple opportunities concurrently. + + Args: + opportunities: List of opportunities to analyze + + Returns: + List of analysis results + """ + import asyncio + + async with AsyncLLMClient() as client: + tasks = [] + for opp in opportunities: + prompt = f"Quick analysis: {opp.market}, spread ${opp.spread:.4f}, expires {opp.expiry}. Worth it? (Yes/No + 1 sentence)" + tasks.append( + client.chat( + self.model, + prompt, + system="Be extremely concise. Yes/No + one sentence max.", + ) + ) + + results = await asyncio.gather(*tasks, return_exceptions=True) + + return [ + ( + {"opportunity": opp, "analysis": r} + if isinstance(r, str) + else {"opportunity": opp, "error": str(r)} + ) + for opp, r in zip(opportunities, results) + ] + + +# Example usage +if __name__ == "__main__": + # Create sample opportunity + opportunity = ArbitrageOpportunity( + platform_a="Polymarket", + platform_b="Kalshi", + price_a=0.52, + price_b=0.47, + spread=0.99, # $0.99 combined cost + expiry="2024-01-15 17:00 UTC", + market="BTC > $100,000 by Jan 15", + ) + + # Initialize analyzer (uses BASE_CHAIN_WALLET_KEY from env) + analyzer = ArbitrageAnalyzer(model_tier="fast") + + print("=" * 60) + print("BlockRun Arbitrage Analyzer") + print("=" * 60) + print(f"Wallet: {analyzer.client.get_wallet_address()}") + print(f"Model: {analyzer.model}") + print("=" * 60) + + # Analyze the opportunity + print("\n1. Analyzing opportunity...") + result = analyzer.analyze_opportunity(opportunity) + + if result["success"]: + print(f"\nAnalysis ({result['model']}):") + print("-" * 40) + print(result["analysis"]) + else: + print(f"Error: {result['error']}") + + # Get market sentiment + print("\n2. Getting BTC sentiment...") + sentiment = analyzer.get_market_sentiment("BTC") + + if sentiment["success"]: + print(f"\nSentiment ({sentiment['model']}):") + print("-" * 40) + print(sentiment["sentiment"]) + else: + print(f"Error: {sentiment['error']}") + + print("\n" + "=" * 60) + print("Cost: ~$0.002-0.02 total (pay-per-request)") + print("=" * 60) diff --git a/examples/benchmark_claude.py b/examples/benchmark_claude.py new file mode 100755 index 0000000..cfd9b24 --- /dev/null +++ b/examples/benchmark_claude.py @@ -0,0 +1,290 @@ +#!/usr/bin/env python3 +""" +End-to-end performance benchmark for a Claude model through the BlockRun gateway. + +Measures the 12 metrics requested: + 1. 单个请求吞吐 (token/s) per-request output throughput (avg of per-req tokens/s) + 2. 系统级平均token生成速度 system-level aggregate output tokens / wall-clock + 3. 平均 TTFT(s) mean time-to-first-token (streaming) + 4. P50 TTFT(s) + 5. P95 TTFT(s) + 6. P99 TTFT(s) + 7. 平均延迟(s) mean end-to-end latency (request → last token) + 8. P50 延迟(s) + 9. P95 延迟(s) + 10. P99 延迟(s) + 11. 成功率(%) successful requests / total + 12. 缓存命中率(%) cache_read_input_tokens / prompt_tokens (2nd call, + shared long prefix). NON-streaming — the gateway's + SSE chunks carry no usage. Requires the + fingerprint-passthrough code DEPLOYED and the test + wallet in ANTHROPIC_DIRECT_PAYER_ALLOWLIST_EXTRA, + otherwise the gateway strips cache tokens → N/A. + +Each paid request spends USDC via x402. Pick --requests with that in mind. + +Wallet: read by the SDK from --private-key, $SOLANA_WALLET_KEY / $BLOCKRUN_WALLET_KEY, +or ~/.blockrun/.session — the key never leaves the host. + +Examples +-------- + # Solana gateway, claude-opus-4.7, 30 reqs @ concurrency 5, + cache probe + python benchmark_claude.py --chain solana --model anthropic/claude-opus-4.7 \ + --requests 30 --concurrency 5 --cache-probe + + # Base gateway, sonnet, quick 10-req smoke + python benchmark_claude.py --chain base --model anthropic/claude-sonnet-4.6 \ + --requests 10 --concurrency 3 +""" +from __future__ import annotations + +import argparse +import statistics +import time +from concurrent.futures import ThreadPoolExecutor, as_completed +from dataclasses import dataclass, field +from typing import Any + +SOLANA_API_URL = "https://sol.blockrun.ai/api" +BASE_API_URL = "https://blockrun.ai/api" + +# Default workload — a deterministic-ish prompt that yields a few hundred tokens. +DEFAULT_PROMPT = ( + "Explain how an x402 micropayment settles on-chain, step by step, " + "from the 402 challenge to facilitator verification. Be concise." +) + + +def _percentile(values: list[float], pct: float) -> float: + """Nearest-rank percentile (pct in [0,100]). Empty → nan.""" + if not values: + return float("nan") + ordered = sorted(values) + if len(ordered) == 1: + return ordered[0] + # nearest-rank: index = ceil(pct/100 * N) - 1 + import math + + rank = max(1, math.ceil((pct / 100.0) * len(ordered))) + return ordered[min(rank, len(ordered)) - 1] + + +def _count_tokens(text: str, model_hint: str = "") -> int: + """Best-effort output-token count for throughput. Uses tiktoken if present + (o200k_base — closest public BPE), else a ~4-chars/token estimate. Claude's + real tokenizer differs slightly; throughput is reported as an estimate.""" + try: + import tiktoken + + enc = tiktoken.get_encoding("o200k_base") + return len(enc.encode(text)) + except Exception: + return max(1, round(len(text) / 4)) + + +@dataclass +class ReqResult: + ok: bool + ttft: float | None = None # seconds to first content token + latency: float | None = None # seconds request → last token + out_tokens: int = 0 + error: str = "" + + +@dataclass +class Bench: + chain: str + model: str + api_url: str + requests: int + concurrency: int + prompt: str + max_tokens: int + private_key: str | None = None + results: list[ReqResult] = field(default_factory=list) + + def _client(self): + if self.chain == "solana": + from blockrun_llm import SolanaLLMClient + + key = self.private_key + if not key: + # Fall back to the SDK's wallet resolver ($SOLANA_WALLET_KEY → + # ~/.blockrun/.solana-session) so the existing session "just works". + from blockrun_llm.solana_wallet import load_solana_wallet + + key = load_solana_wallet() + return SolanaLLMClient(private_key=key, api_url=self.api_url) + from blockrun_llm import LLMClient + + return LLMClient(private_key=self.private_key, api_url=self.api_url) + + def _one_streaming(self, client) -> ReqResult: + messages = [{"role": "user", "content": self.prompt}] + start = time.perf_counter() + ttft: float | None = None + text_parts: list[str] = [] + try: + for chunk in client.chat_completion_stream( + model=self.model, messages=messages, max_tokens=self.max_tokens + ): + if not chunk.choices: + continue + delta = chunk.choices[0].delta + content = getattr(delta, "content", None) + if content: + if ttft is None: + ttft = time.perf_counter() - start + text_parts.append(content) + latency = time.perf_counter() - start + out = _count_tokens("".join(text_parts), self.model) + return ReqResult(ok=True, ttft=ttft, latency=latency, out_tokens=out) + except Exception as exc: + return ReqResult(ok=False, error=f"{type(exc).__name__}: {exc}") + + def run_throughput_phase(self) -> float: + """Fire `requests` streaming calls at `concurrency`. Returns wall-clock seconds.""" + client = self._client() + wall_start = time.perf_counter() + with ThreadPoolExecutor(max_workers=self.concurrency) as pool: + futures = [pool.submit(self._one_streaming, client) for _ in range(self.requests)] + for fut in as_completed(futures): + self.results.append(fut.result()) + return time.perf_counter() - wall_start + + def cache_probe(self) -> float: + """Two NON-streaming calls sharing a long system prefix. Returns cache hit + rate (%) on the 2nd call: cached_input_tokens / prompt_tokens. + + Reads BOTH provider conventions the gateway may surface (for allowlisted + payers): Anthropic ``cache_read_input_tokens`` and OpenAI + ``prompt_tokens_details.cached_tokens``. Returns 0.0 when nothing cached + or the field is absent (not deployed / not allowlisted / model doesn't + cache) — reported as 0, never "N/A".""" + client = self._client() + long_prefix = ("You are a meticulous protocol analyst. " * 240).strip() + # Identical long system prefix (the cacheable part) but DIFFERENT user + # messages on the two calls — an identical body would trip the gateway's + # x402 replay guard, so vary it while keeping the prefix cache-eligible. + warm = [ + {"role": "system", "content": long_prefix}, + {"role": "user", "content": "Reply with the single word: ready."}, + ] + measure = [ + {"role": "system", "content": long_prefix}, + {"role": "user", "content": "Now reply with the single word: done."}, + ] + client.chat_completion(model=self.model, messages=warm, max_tokens=8) + time.sleep(2.0) + resp = client.chat_completion(model=self.model, messages=measure, max_tokens=8) + usage = getattr(resp, "usage", None) + if usage is None: + return 0.0 + u: dict[str, Any] = ( + usage.model_dump(exclude_none=True) if hasattr(usage, "model_dump") else dict(usage) + ) + prompt_tokens = u.get("prompt_tokens") or 0 + cache_read = u.get("cache_read_input_tokens") or 0 + cache_creation = u.get("cache_creation_input_tokens") or 0 + # Anthropic style: prompt_tokens (= input_tokens) EXCLUDES cached tokens — + # the three counts are disjoint, so total input is their sum. + if cache_read or cache_creation: + total_input = prompt_tokens + cache_read + cache_creation + return 100.0 * cache_read / total_input if total_input else 0.0 + # OpenAI style: cached_tokens is a SUBSET of prompt_tokens. + details = u.get("prompt_tokens_details") or {} + cached = details.get("cached_tokens", 0) if isinstance(details, dict) else 0 + if cached and prompt_tokens: + return 100.0 * cached / prompt_tokens + return 0.0 + + def report(self, wall: float, cache_hit: float | None) -> None: + ok = [r for r in self.results if r.ok] + ttfts = [r.ttft for r in ok if r.ttft is not None] + lats = [r.latency for r in ok if r.latency is not None] + per_req_tps = [ + r.out_tokens / r.latency for r in ok if r.latency and r.latency > 0 and r.out_tokens + ] + total_out = sum(r.out_tokens for r in ok) + succ = 100.0 * len(ok) / self.requests if self.requests else 0.0 + + def fmt(x: float) -> str: + return "nan" if x != x else f"{x:.3f}" # noqa: PLR0124 — x!=x is the NaN test + + print("\n" + "=" * 56) + print(f" Claude E2E benchmark — {self.model} ({self.chain})") + print(f" {self.api_url}") + print( + f" requests={self.requests} concurrency={self.concurrency} " + f"max_tokens={self.max_tokens}" + ) + print("=" * 56) + rows = [ + ("单个请求吞吐 (token/s)", fmt(statistics.mean(per_req_tps)) if per_req_tps else "nan"), + ("系统级平均token生成速度 (token/s)", fmt(total_out / wall) if wall > 0 else "nan"), + ("平均TTFT(s)", fmt(statistics.mean(ttfts)) if ttfts else "nan"), + ("P50 TTFT(s)", fmt(_percentile(ttfts, 50))), + ("P95 TTFT(s)", fmt(_percentile(ttfts, 95))), + ("P99 TTFT(s)", fmt(_percentile(ttfts, 99))), + ("平均延迟(s)", fmt(statistics.mean(lats)) if lats else "nan"), + ("P50延迟(s)", fmt(_percentile(lats, 50))), + ("P95延迟(s)", fmt(_percentile(lats, 95))), + ("P99延迟(s)", fmt(_percentile(lats, 99))), + ("成功率(%)", fmt(succ)), + ("缓存命中率(%)", fmt(cache_hit if cache_hit is not None else 0.0)), + ] + for name, val in rows: + print(f" {name:<34} {val}") + print("-" * 56) + print( + f" 样本: 成功 {len(ok)}/{self.requests} 总输出≈{total_out} tokens wall={wall:.2f}s" + ) + fails = [r for r in self.results if not r.ok] + if fails: + print(f" 失败 {len(fails)} 例,示例: {fails[0].error}") + print("=" * 56 + "\n") + + +def main() -> None: + p = argparse.ArgumentParser(description="Claude E2E benchmark via BlockRun gateway") + p.add_argument("--chain", choices=["solana", "base"], default="solana") + p.add_argument("--model", default="anthropic/claude-opus-4.7") + p.add_argument("--api-url", default=None, help="override gateway URL") + p.add_argument("--requests", type=int, default=20) + p.add_argument("--concurrency", type=int, default=5) + p.add_argument("--max-tokens", type=int, default=256) + p.add_argument("--prompt", default=DEFAULT_PROMPT) + p.add_argument("--private-key", default=None, help="wallet key (else env / ~/.blockrun)") + p.add_argument( + "--cache-probe", + action="store_true", + help="add 2 non-streaming calls to measure cache hit rate (extra spend)", + ) + args = p.parse_args() + + api_url = args.api_url or (SOLANA_API_URL if args.chain == "solana" else BASE_API_URL) + bench = Bench( + chain=args.chain, + model=args.model, + api_url=api_url, + requests=args.requests, + concurrency=args.concurrency, + prompt=args.prompt, + max_tokens=args.max_tokens, + private_key=args.private_key, + ) + print(f"[benchmark] {args.requests} paid streaming requests → {api_url} ({args.model}) …") + wall = bench.run_throughput_phase() + cache_hit = 0.0 + if args.cache_probe: + print("[benchmark] cache probe (2 non-streaming calls) …") + try: + cache_hit = bench.cache_probe() + except Exception as exc: + print(f"[benchmark] cache probe failed (→ 0): {type(exc).__name__}: {exc}") + cache_hit = 0.0 + bench.report(wall, cache_hit) + + +if __name__ == "__main__": + main() diff --git a/examples/sweep_all_chat_models.py b/examples/sweep_all_chat_models.py new file mode 100644 index 0000000..a0914a0 --- /dev/null +++ b/examples/sweep_all_chat_models.py @@ -0,0 +1,705 @@ +"""Sweep test for every chat LLM the BlockRun Python SDK can call on Base. + +Sends a minimal probe ("What is 2+2?") to each model in SWEEP_TARGETS, captures +status / latency / token usage / per-call cost, and prints a grouped report at +the end. Designed to be run manually before releases or after router changes. + +Usage: + export BLOCKRUN_WALLET_KEY=0x... # ≥ $1 USDC on Base mainnet + python examples/sweep_all_chat_models.py + +Optional flags: + --budget-cap 2.50 abort sweep when cumulative spend reaches this + --sleep 1.0 seconds between sequential calls + --skip-async skip the AsyncLLMClient gather() smoke + --only openai,nvidia restrict sweep to specific providers (CSV) + --output-json FILE write per-probe results as JSON +""" + +from __future__ import annotations + +import argparse +import asyncio +import json +import os +import sys +import time +from dataclasses import asdict, dataclass, field +from typing import Any + +import httpx + +from blockrun_llm import AsyncLLMClient, LLMClient +from blockrun_llm.types import APIError, PaymentError + +# --------------------------------------------------------------------------- +# Sweep targets — hardcoded so we also probe hidden / retired model ids that +# the /v1/models endpoint deliberately omits. Mutually-exclusive groups, in +# the order the report displays them. +# --------------------------------------------------------------------------- + +SWEEP_TARGETS: list[str] = [ + # OpenAI + "openai/gpt-5.5", + "openai/gpt-5.4", + "openai/gpt-5.4-pro", + "openai/gpt-5.4-mini", + "openai/gpt-5.4-nano", + "openai/gpt-5.3", + "openai/gpt-5.3-codex", + "openai/gpt-5.2", + "openai/gpt-5.2-pro", + "openai/gpt-5-mini", + "openai/o1", + "openai/o3", + "openai/o3-mini", + # Anthropic + "anthropic/claude-opus-4.8", + "anthropic/claude-opus-4.7", + "anthropic/claude-opus-4.6", + "anthropic/claude-opus-4.5", + "anthropic/claude-sonnet-4.6", + "anthropic/claude-haiku-4.5", + # Google + "google/gemini-3.1-pro", + "google/gemini-3-flash-preview", + "google/gemini-2.5-pro", + "google/gemini-2.5-flash", + "google/gemini-3.1-flash-lite", + "google/gemini-2.5-flash-lite", + # DeepSeek + "deepseek/deepseek-v4-pro", + "deepseek/deepseek-chat", + "deepseek/deepseek-reasoner", + # xAI — resold via OpenRouter credit pool (added 2026-06-04) + "xai/grok-4.3", + "xai/grok-build-0.1", + # MiniMax + "minimax/minimax-m3", + "minimax/minimax-m2.7", + # ZAI + "zai/glm-5.2", + "zai/glm-5.1", + "zai/glm-5", + "zai/glm-5-turbo", + # Moonshot + "moonshot/kimi-k2.5", + "moonshot/kimi-k2.6", + # NVIDIA — free tier + # (qwen3-next-80b-a3b-thinking removed: NVIDIA EOL 2026-05-21, HTTP 410; + # gateway redirects pinned callers to llama-4-maverick) + "nvidia/deepseek-v4-flash", + "nvidia/nemotron-3-nano-omni-30b-a3b-reasoning", + "nvidia/mistral-small-4-119b", + "nvidia/llama-4-maverick", + "nvidia/qwen3-coder-480b", + # NVIDIA — hidden from /v1/models but direct calls still work intentionally + # (re-enabled 2026-04-30). Privacy caveat: NVIDIA's free build.nvidia.com + # tier may use prompts/outputs for service improvement. + "nvidia/gpt-oss-120b", + "nvidia/gpt-oss-20b", + # NVIDIA — hidden, backend redirects to v4-flash + "nvidia/deepseek-v4-pro", + "nvidia/deepseek-v3.2", + "nvidia/glm-4.7", +] + +REASONING_MODELS = { + "openai/o1", + "openai/o3", + "openai/o3-mini", + "openai/gpt-5.3-codex", + "deepseek/deepseek-reasoner", + "deepseek/deepseek-v4-pro", + "xai/grok-4.3", + "zai/glm-5.2", + "nvidia/nemotron-3-nano-omni-30b-a3b-reasoning", +} + +# Hidden from /v1/models (so SmartChat won't auto-pick them) but direct calls +# still work intentionally per the README "Available free models" table notes. +# A 200 OK from these is the expected state, not a privacy violation. +HIDDEN_CALLABLE = { + "anthropic/claude-opus-4.6", + "moonshot/kimi-k2.5", + "nvidia/gpt-oss-120b", + "nvidia/gpt-oss-20b", +} + +# Hidden from /v1/models, backend redirects to a different model id; expect +# response.model != requested model. +HIDDEN_REDIRECTED = { + "nvidia/deepseek-v4-pro", + "nvidia/deepseek-v3.2", + "nvidia/glm-4.7", +} + +ASYNC_SMOKE_MODELS = [ + "deepseek/deepseek-chat", + "anthropic/claude-haiku-4.5", + "google/gemini-2.5-flash-lite", +] + +PROBE_PROMPT = "Reply with the digit 4 only. What is 2+2?" +PROBE_MAX_TOKENS = 8 +PROBE_MAX_TOKENS_REASONING = 512 + + +# --------------------------------------------------------------------------- +# Result record +# --------------------------------------------------------------------------- + + +@dataclass +class ProbeResult: + model_id: str + provider: str + status: str + latency_ms: int + tokens_in: int | None = None + tokens_out: int | None = None + tokens_total: int | None = None + cost_delta_usd: float = 0.0 + expected_cost_usd: float | None = None + cost_drift_pct: float | None = None + redirected_to: str | None = None + response_preview: str = "" + contains_4: bool = False + error_message: str | None = None + timestamp: float = field(default_factory=time.time) + + +# --------------------------------------------------------------------------- +# Helpers +# --------------------------------------------------------------------------- + + +def sanitize(s: str, limit: int = 200) -> str: + s = s.replace("\n", "; ").replace("\r", " ") + for prefix in ("/Users/", "/var/", "/private/", "/tmp/"): + idx = s.find(prefix) + if idx >= 0: + s = s[:idx] + "[path]" + return s[:limit] + + +def mask_address(addr: str) -> str: + return f"{addr[:6]}...{addr[-4:]}" if len(addr) > 10 else addr + + +def fmt_cost(usd: float) -> str: + return f"${usd:.5f}" + + +def provider_of(model_id: str) -> str: + return model_id.split("/", 1)[0] if "/" in model_id else model_id + + +# --------------------------------------------------------------------------- +# Preflight +# --------------------------------------------------------------------------- + + +def preflight() -> LLMClient: + print("=" * 78) + print("BLOCKRUN PYTHON SDK — CHAT-LLM SWEEP") + print("=" * 78) + + found = None + for name in ("BLOCKRUN_WALLET_KEY", "BASE_CHAIN_WALLET_KEY"): + if os.environ.get(name): + found = name + break + if not found: + if os.path.exists(os.path.expanduser("~/.blockrun/.session")): + found = "~/.blockrun/.session" + else: + sys.stderr.write( + "ERROR: no wallet key found.\n" + " set BLOCKRUN_WALLET_KEY or BASE_CHAIN_WALLET_KEY,\n" + " or run setup_agent_wallet() to create ~/.blockrun/.session.\n" + ) + sys.exit(2) + print(f"key source : {found}") + + client = LLMClient() + + if client.is_testnet(): + sys.stderr.write("ERROR: client resolved to testnet. refusing to run.\n") + sys.exit(2) + + print(f"wallet : {mask_address(client.get_wallet_address())}") + print(f"api url : {client.api_url}") + print("network : Base mainnet") + + try: + balance = client.get_balance() + print(f"USDC bal : ${balance:.4f}") + if balance < 1.0: + print("WARN : balance below $1.00 — sweep may abort mid-run") + except Exception as e: + print(f"USDC bal : (unavailable: {sanitize(str(e), 80)})") + + initial = client.get_spending() + print(f"spending : ${initial['total_usd']:.4f} ({initial['calls']} calls)") + print(f"sweep size : {len(SWEEP_TARGETS)} models") + print() + return client + + +# --------------------------------------------------------------------------- +# Forward-compat diff vs /v1/models +# --------------------------------------------------------------------------- + + +def forward_compat_check(client: LLMClient) -> dict[str, dict[str, Any]]: + print(">>> Forward-compat check vs /v1/models") + try: + listed_raw = client.list_models() + except Exception as e: + print(f" list_models() failed ({sanitize(str(e), 60)}); skipping") + print() + return {} + + listed_chat: dict[str, dict[str, Any]] = {} + for m in listed_raw: + cats = m.get("categories") + if cats is None or "chat" in cats: + listed_chat[m["id"]] = m + + listed_ids = set(listed_chat.keys()) + hardcoded = set(SWEEP_TARGETS) + + new_in_api = listed_ids - hardcoded + missing_in_api = hardcoded - listed_ids + + print(f" listed in API : {len(listed_ids)} chat models") + print(f" in our sweep : {len(hardcoded)}") + print(f" overlap : {len(listed_ids & hardcoded)}") + + if missing_in_api: + print(f" {len(missing_in_api)} sweep targets not listed (hidden/retired expected):") + for m in sorted(missing_in_api): + print(f" - {m}") + + if new_in_api: + print(f" {len(new_in_api)} NEW models in API not in sweep list:") + for m in sorted(new_in_api): + print(f" + {m} (consider adding to SWEEP_TARGETS)") + + print() + return listed_chat + + +# --------------------------------------------------------------------------- +# Single probe +# --------------------------------------------------------------------------- + + +def probe_one( + client: LLMClient, + model_id: str, + pricing: dict[str, dict[str, Any]], +) -> ProbeResult: + provider = provider_of(model_id) + max_toks = PROBE_MAX_TOKENS_REASONING if model_id in REASONING_MODELS else PROBE_MAX_TOKENS + pre = client.get_spending()["total_usd"] + t0 = time.monotonic() + try: + response = client.chat_completion( + model_id, + [{"role": "user", "content": PROBE_PROMPT}], + max_tokens=max_toks, + ) + latency_ms = int((time.monotonic() - t0) * 1000) + post = client.get_spending()["total_usd"] + cost_delta = post - pre + + text = "" + try: + text = response.choices[0].message.content or "" + except Exception: + text = "" + + usage = response.usage + tokens_in = usage.prompt_tokens if usage else None + tokens_out = usage.completion_tokens if usage else None + tokens_total = usage.total_tokens if usage else None + + responding_model = getattr(response, "model", model_id) or model_id + # Only treat as "redirected" if the responding model fundamentally differs + # (different base id). Upstreams often append dated suffixes like + # "gpt-5.5-2026-04-20" — that's not a redirect. + requested_tail = model_id.split("/", 1)[-1] + responding_tail = ( + responding_model.split("/", 1)[-1] if "/" in responding_model else responding_model + ) + same_family = ( + requested_tail in responding_model + or responding_tail in model_id + or model_id in HIDDEN_CALLABLE # backend reports canonical id, that's fine + ) + redirected_to = responding_model if responding_model and not same_family else None + + # Cost-drift check vs published pricing. /v1/models returns + # `pricing.input` / `pricing.output` (USD per 1M tokens) for paid models + # and `pricing.flat` (USD per call) for flat-priced models. + expected_cost = None + cost_drift_pct = None + meta = pricing.get(model_id, {}) + price_block = meta.get("pricing") or {} + if tokens_in is not None and tokens_out is not None and price_block: + if "flat" in price_block: + expected_cost = float(price_block["flat"]) + else: + ip = float( + price_block.get( + "input", + meta.get("inputPrice", meta.get("input_price", 0)), + ) + ) + op = float( + price_block.get( + "output", + meta.get("outputPrice", meta.get("output_price", 0)), + ) + ) + expected_cost = (tokens_in * ip + tokens_out * op) / 1_000_000.0 + if expected_cost > 0: + cost_drift_pct = (cost_delta - expected_cost) / expected_cost * 100.0 + + if not text and (usage and usage.completion_tokens == 0): + status = "ok_empty" + elif redirected_to: + status = "ok_redirected" + else: + status = "ok" + + return ProbeResult( + model_id=model_id, + provider=provider, + status=status, + latency_ms=latency_ms, + tokens_in=tokens_in, + tokens_out=tokens_out, + tokens_total=tokens_total, + cost_delta_usd=cost_delta, + expected_cost_usd=expected_cost, + cost_drift_pct=cost_drift_pct, + redirected_to=redirected_to, + response_preview=text[:80], + contains_4="4" in text[:80], + error_message=None, + ) + + except APIError as e: + latency_ms = int((time.monotonic() - t0) * 1000) + status = "http_error" + return ProbeResult( + model_id=model_id, + provider=provider, + status=status, + latency_ms=latency_ms, + cost_delta_usd=0.0, + error_message=f"status={e.status_code}: {sanitize(str(e), 200)}", + ) + except PaymentError as e: + latency_ms = int((time.monotonic() - t0) * 1000) + return ProbeResult( + model_id=model_id, + provider=provider, + status="payment_error", + latency_ms=latency_ms, + error_message=sanitize(str(e), 200), + ) + except httpx.TimeoutException: + latency_ms = int((time.monotonic() - t0) * 1000) + return ProbeResult( + model_id=model_id, + provider=provider, + status="timeout", + latency_ms=latency_ms, + error_message=f"timeout after {latency_ms}ms", + ) + except Exception as e: + latency_ms = int((time.monotonic() - t0) * 1000) + return ProbeResult( + model_id=model_id, + provider=provider, + status="unexpected", + latency_ms=latency_ms, + error_message=f"{type(e).__name__}: {sanitize(str(e), 200)}", + ) + + +# --------------------------------------------------------------------------- +# Main sweep loop +# --------------------------------------------------------------------------- + + +def run_sweep( + client: LLMClient, + targets: list[str], + args: argparse.Namespace, + pricing: dict[str, dict[str, Any]], +) -> list[ProbeResult]: + results: list[ProbeResult] = [] + n = len(targets) + warned = False + + print(f">>> Sweep ({n} models, {args.sleep}s between calls)") + print() + + for i, model_id in enumerate(targets, start=1): + spending = client.get_spending()["total_usd"] + if spending >= args.budget_cap: + print(f"[BUDGET-ABORT] cumulative spend ${spending:.4f} >= ${args.budget_cap:.2f}") + for remaining in targets[i - 1 :]: + results.append( + ProbeResult( + model_id=remaining, + provider=provider_of(remaining), + status="skipped_budget", + latency_ms=0, + error_message=f"budget cap ${args.budget_cap:.2f} reached", + ) + ) + return results + if spending >= args.budget_cap * 0.8 and not warned: + print(f"[BUDGET-WARN] at ${spending:.4f} of ${args.budget_cap:.2f}") + warned = True + + result = probe_one(client, model_id, pricing) + results.append(result) + + token_str = ( + f"{result.tokens_in}/{result.tokens_out}" if result.tokens_in is not None else "-/-" + ) + preview = (result.response_preview or "").replace("\n", " ")[:30] + if result.error_message: + preview = result.error_message[:30] + print( + f"[{i:03d}/{n}] {result.model_id:50s} {result.status:18s} " + f"{fmt_cost(result.cost_delta_usd)} {result.latency_ms:5d}ms " + f"{token_str:>9s} {preview}" + ) + + if i < n: + time.sleep(args.sleep) + + return results + + +# --------------------------------------------------------------------------- +# Async smoke (verifies AsyncLLMClient + asyncio.gather()). +# --------------------------------------------------------------------------- + + +async def _async_probe(client: AsyncLLMClient, model_id: str) -> dict[str, Any]: + t0 = time.monotonic() + try: + response = await client.chat_completion( + model_id, + [{"role": "user", "content": PROBE_PROMPT}], + max_tokens=PROBE_MAX_TOKENS, + ) + latency_ms = int((time.monotonic() - t0) * 1000) + usage = response.usage + return { + "model_id": model_id, + "ok": True, + "latency_ms": latency_ms, + "tokens_in": usage.prompt_tokens if usage else None, + "tokens_out": usage.completion_tokens if usage else None, + "preview": (response.choices[0].message.content or "")[:30], + "error": None, + } + except Exception as e: + latency_ms = int((time.monotonic() - t0) * 1000) + return { + "model_id": model_id, + "ok": False, + "latency_ms": latency_ms, + "tokens_in": None, + "tokens_out": None, + "preview": "", + "error": f"{type(e).__name__}: {sanitize(str(e), 120)}", + } + + +async def _async_smoke() -> list[dict[str, Any]]: + async with AsyncLLMClient() as client: + coros = [_async_probe(client, m) for m in ASYNC_SMOKE_MODELS] + return await asyncio.gather(*coros) + + +def run_async_smoke() -> list[dict[str, Any]]: + print(">>> Async smoke (asyncio.gather over 3 models)") + t0 = time.monotonic() + results = asyncio.run(_async_smoke()) + total_ms = int((time.monotonic() - t0) * 1000) + for r in results: + flag = "ok " if r["ok"] else "FAIL" + token_str = f"{r['tokens_in']}/{r['tokens_out']}" if r["tokens_in"] is not None else "-/-" + detail = r["error"] if not r["ok"] else r["preview"] + print( + f" [async] {r['model_id']:42s} {flag} {r['latency_ms']:5d}ms " + f"{token_str:>7s} {detail}" + ) + print(f" total wall: {total_ms}ms") + print() + return results + + +# --------------------------------------------------------------------------- +# Final report +# --------------------------------------------------------------------------- + + +def report( + client: LLMClient, + results: list[ProbeResult], + async_results: list[dict[str, Any]] | None, + started_at: float, + args: argparse.Namespace, +) -> bool: + failures = [r for r in results if not r.status.startswith("ok")] + drifts = [r for r in results if r.cost_drift_pct is not None and abs(r.cost_drift_pct) > 5.0] + + if failures: + print(">>> Failures") + for r in failures: + print(f" {r.model_id:50s} {r.status:18s}") + if r.error_message: + print(f" {r.error_message}") + print() + + if drifts: + print(">>> Cost drift > 5% (potential billing inconsistency)") + for r in drifts: + print( + f" {r.model_id:50s} actual={fmt_cost(r.cost_delta_usd)} " + f"expected={fmt_cost(r.expected_cost_usd or 0)} " + f"drift={r.cost_drift_pct:+.1f}%" + ) + print() + + print(">>> Provider summary") + by_provider: dict[str, list[ProbeResult]] = {} + for r in results: + by_provider.setdefault(r.provider, []).append(r) + for provider in sorted(by_provider): + rows = by_provider[provider] + ok = sum(1 for r in rows if r.status.startswith("ok")) + cost = sum(r.cost_delta_usd for r in rows) + toks_in = sum(r.tokens_in or 0 for r in rows) + toks_out = sum(r.tokens_out or 0 for r in rows) + line = f" {provider:10s} {ok}/{len(rows)} ok cost={fmt_cost(cost)}" + if toks_in or toks_out: + line += f" tokens={toks_in}/{toks_out}" + print(line) + print() + + spending = client.get_spending() + duration = time.monotonic() - started_at + minutes, seconds = divmod(int(duration), 60) + + # NVIDIA "free path" = models in the README's Available free models table that + # we expect to actually run inference. Excludes the hidden+redirected ids + # (deepseek-v4-pro/v3.2/glm-4.7) which forward to v4-flash on the backend. + nvidia_free = [ + r for r in results if r.provider == "nvidia" and r.model_id not in HIDDEN_REDIRECTED + ] + nvidia_free_ok = all(r.status.startswith("ok") for r in nvidia_free) + + successes = [r for r in results if r.status.startswith("ok")] + success_rate = len(successes) / max(len(results), 1) * 100 + + total_in = sum(r.tokens_in or 0 for r in results) + total_out = sum(r.tokens_out or 0 for r in results) + + async_ok = True + if async_results is not None: + async_ok = all(r["ok"] for r in async_results) + + main_threshold = 40 if not args.only else max(int(len(results) * 0.9), 1) + main_pass = len(successes) >= main_threshold + + overall_pass = main_pass and nvidia_free_ok and async_ok + + print(">>> Summary") + print(f" total cost : {fmt_cost(spending['total_usd'])}") + print(f" total tokens : {total_in} in / {total_out} out / {total_in + total_out} total") + print(f" total calls : {spending['calls']}") + print(f" sweep success : {len(successes)}/{len(results)} ({success_rate:.0f}%)") + print(f" duration : {minutes}m {seconds}s") + print(f" budget used : {fmt_cost(spending['total_usd'])} of {fmt_cost(args.budget_cap)}") + if async_results is not None: + passed = sum(1 for r in async_results if r["ok"]) + print(f" async smoke : {passed}/{len(async_results)}") + print(f" free tier : {'all ok' if nvidia_free_ok else 'FAIL'}") + print(f" status : {'PASS' if overall_pass else 'FAIL'}") + print() + + return overall_pass + + +# --------------------------------------------------------------------------- +# Entry point +# --------------------------------------------------------------------------- + + +def main() -> int: + parser = argparse.ArgumentParser( + description="Sweep test every chat LLM in the BlockRun SDK on Base mainnet." + ) + parser.add_argument("--budget-cap", type=float, default=2.50) + parser.add_argument("--sleep", type=float, default=1.0) + parser.add_argument("--skip-async", action="store_true") + parser.add_argument( + "--only", + type=str, + default=None, + help="comma-separated provider prefixes (e.g. openai,nvidia)", + ) + parser.add_argument("--output-json", type=str, default=None) + args = parser.parse_args() + + started_at = time.monotonic() + + client = preflight() + listed = forward_compat_check(client) + + targets = SWEEP_TARGETS + if args.only: + keep = {p.strip() for p in args.only.split(",") if p.strip()} + targets = [m for m in SWEEP_TARGETS if provider_of(m) in keep] + print(f"--only filter: {sorted(keep)} → {len(targets)} models") + print() + + results = run_sweep(client, targets, args, listed) + + async_results: list[dict[str, Any]] | None = None + if not args.skip_async: + async_results = run_async_smoke() + + overall_pass = report(client, results, async_results, started_at, args) + + if args.output_json: + payload = { + "started_at": started_at, + "args": vars(args), + "results": [asdict(r) for r in results], + "async_results": async_results, + "spending": client.get_spending(), + "pass": overall_pass, + } + with open(args.output_json, "w") as f: + json.dump(payload, f, indent=2) + print(f"results JSON written to {args.output_json}") + + return 0 if overall_pass else 1 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/examples/sweep_all_media_models.py b/examples/sweep_all_media_models.py new file mode 100644 index 0000000..82d7fb7 --- /dev/null +++ b/examples/sweep_all_media_models.py @@ -0,0 +1,456 @@ +"""Sweep test for every image + music model the BlockRun SDK exposes. + +Runs each model with a short fixed prompt, captures status / latency / cost, +and prints a grouped report at the end. Mirror of examples/sweep_all_chat_models.py +but for ImageClient and MusicClient. Video is intentionally separate (single +clip can take >2 min and cost up to $0.30 — run that one manually). + +Usage: + export BLOCKRUN_WALLET_KEY=0x... # ≥ $1 USDC on Base mainnet + python examples/sweep_all_media_models.py + +Optional: + --budget-cap 1.00 abort sweep when cumulative spend reaches this + --skip-image run only music + --skip-music run only image + --output-json FILE write per-probe results as JSON +""" + +from __future__ import annotations + +import argparse +import json +import sys +import time +from dataclasses import asdict, dataclass, field +from typing import Any + +import httpx + +from blockrun_llm import ImageClient, LLMClient, MusicClient +from blockrun_llm.types import APIError, PaymentError + +IMAGE_TARGETS: list[dict[str, Any]] = [ + # Each entry: model_id + size override if model has a constrained set. + {"model": "google/nano-banana", "size": "1024x1024"}, + {"model": "google/nano-banana-pro", "size": "1024x1024"}, + {"model": "openai/dall-e-3", "size": "1024x1024"}, + {"model": "openai/gpt-image-1", "size": "1024x1024"}, + {"model": "openai/gpt-image-2", "size": "1024x1024"}, + {"model": "zai/cogview-4", "size": "1024x1024"}, + {"model": "xai/grok-imagine-image", "size": "1024x1024"}, + {"model": "xai/grok-imagine-image-pro", "size": "1024x1024"}, +] + +MUSIC_TARGETS: list[str] = [ + "minimax/music-2.5+", + "minimax/music-2.5", +] + +IMAGE_PROMPT = "a single red apple on a plain white background, photographic" +MUSIC_PROMPT = "30-second chill lo-fi beat with mellow piano" + + +@dataclass +class ProbeResult: + model_id: str + modality: str # "image" | "music" + status: str # ok / http_error / timeout / payment_error / unexpected + latency_ms: int + cost_delta_usd: float = 0.0 + artifact_url: str | None = None # first asset URL/data preview + error_message: str | None = None + timestamp: float = field(default_factory=time.time) + + +def sanitize(s: str, limit: int = 200) -> str: + s = s.replace("\n", "; ").replace("\r", " ") + for prefix in ("/Users/", "/var/", "/private/", "/tmp/"): + idx = s.find(prefix) + if idx >= 0: + s = s[:idx] + "[path]" + return s[:limit] + + +def fmt_cost(usd: float) -> str: + return f"${usd:.5f}" + + +def mask_address(addr: str) -> str: + return f"{addr[:6]}...{addr[-4:]}" if len(addr) > 10 else addr + + +def preview_url(url: str, max_len: int = 60) -> str: + if not url: + return "" + if url.startswith("data:"): + # Data URL — show prefix + length + comma = url.find(",") + head = url[:comma] if comma >= 0 else url[:60] + body_len = len(url) - comma - 1 if comma >= 0 else 0 + return f"{head[:30]}...({body_len} bytes)" + return url[:max_len] + ("..." if len(url) > max_len else "") + + +def probe_image( + client: ImageClient, + target: dict[str, Any], + pricing: dict[str, float], +) -> ProbeResult: + model_id = target["model"] + t0 = time.monotonic() + try: + result = client.generate(IMAGE_PROMPT, model=model_id, size=target.get("size")) + latency_ms = int((time.monotonic() - t0) * 1000) + url = result.data[0].url if result.data else "" + return ProbeResult( + model_id=model_id, + modality="image", + status="ok", + latency_ms=latency_ms, + cost_delta_usd=pricing.get(model_id, 0.0), + artifact_url=url, + ) + except APIError as e: + latency_ms = int((time.monotonic() - t0) * 1000) + return ProbeResult( + model_id=model_id, + modality="image", + status="http_error", + latency_ms=latency_ms, + error_message=f"status={e.status_code}: {sanitize(str(e))}", + ) + except PaymentError as e: + latency_ms = int((time.monotonic() - t0) * 1000) + return ProbeResult( + model_id=model_id, + modality="image", + status="payment_error", + latency_ms=latency_ms, + error_message=sanitize(str(e)), + ) + except httpx.TimeoutException: + latency_ms = int((time.monotonic() - t0) * 1000) + return ProbeResult( + model_id=model_id, + modality="image", + status="timeout", + latency_ms=latency_ms, + error_message=f"timeout after {latency_ms}ms", + ) + except Exception as e: + latency_ms = int((time.monotonic() - t0) * 1000) + return ProbeResult( + model_id=model_id, + modality="image", + status="unexpected", + latency_ms=latency_ms, + error_message=f"{type(e).__name__}: {sanitize(str(e))}", + ) + + +def probe_music( + client: MusicClient, + model_id: str, + pricing: dict[str, float], +) -> ProbeResult: + t0 = time.monotonic() + try: + result = client.generate(MUSIC_PROMPT, model=model_id, instrumental=True) + latency_ms = int((time.monotonic() - t0) * 1000) + url = result.data[0].url if result.data else "" + return ProbeResult( + model_id=model_id, + modality="music", + status="ok", + latency_ms=latency_ms, + cost_delta_usd=pricing.get(model_id, 0.0), + artifact_url=url, + ) + except APIError as e: + latency_ms = int((time.monotonic() - t0) * 1000) + return ProbeResult( + model_id=model_id, + modality="music", + status="http_error", + latency_ms=latency_ms, + error_message=f"status={e.status_code}: {sanitize(str(e))}", + ) + except PaymentError as e: + latency_ms = int((time.monotonic() - t0) * 1000) + return ProbeResult( + model_id=model_id, + modality="music", + status="payment_error", + latency_ms=latency_ms, + error_message=sanitize(str(e)), + ) + except httpx.TimeoutException: + latency_ms = int((time.monotonic() - t0) * 1000) + return ProbeResult( + model_id=model_id, + modality="music", + status="timeout", + latency_ms=latency_ms, + error_message=f"timeout after {latency_ms}ms", + ) + except Exception as e: + latency_ms = int((time.monotonic() - t0) * 1000) + return ProbeResult( + model_id=model_id, + modality="music", + status="unexpected", + latency_ms=latency_ms, + error_message=f"{type(e).__name__}: {sanitize(str(e))}", + ) + + +def preflight() -> tuple: + """Return (ImageClient, image_pricing, music_pricing, initial_balance). + + ImageClient/MusicClient don't expose spending tracking, so we use a + parallel LLMClient to read wallet balance for the budget guard. + Per-call costs are looked up from the published pricing tables. + """ + print("=" * 78) + print("BLOCKRUN PYTHON SDK — IMAGE + MUSIC SWEEP") + print("=" * 78) + + image_client = ImageClient() + print(f"wallet : {mask_address(image_client.get_wallet_address())}") + print(f"api url : {image_client.api_url}") + + # Use LLMClient for balance + pricing — image/music clients don't have + # those helpers and the wallet is shared. + llm = LLMClient() + initial_balance = 0.0 + try: + initial_balance = llm.get_balance() + print(f"USDC bal : ${initial_balance:.4f}") + if initial_balance < 1.0: + print("WARN : balance below $1.00 — sweep may abort") + except Exception as e: + print(f"USDC bal : (unavailable: {sanitize(str(e), 80)})") + + # Image + music pricing both come from /v1/models filtered by category; + # the legacy /v1/images/models endpoint currently returns 404 server-side + # (2026-05-09), so don't rely on it. + image_pricing: dict[str, float] = {} + music_pricing: dict[str, float] = {} + try: + for m in llm.list_models(): + mid = m.get("id", "") + if not mid: + continue + cats = m.get("categories") or [] + block = m.get("pricing") or {} + if "image" in cats: + price = block.get("per_image") or block.get("flat") or block.get("perImage") + image_pricing[mid] = float(price or 0) + elif "music" in cats or "audio" in cats: + price = block.get("per_track") or block.get("flat") or block.get("perTrack") + music_pricing[mid] = float(price or 0) + except Exception as e: + print(f"WARN : list_models() failed: {sanitize(str(e), 80)}") + + print(f"image price catalog: {len(image_pricing)} models") + print(f"music price catalog: {len(music_pricing)} models") + print() + return image_client, image_pricing, music_pricing, initial_balance + + +def run_image_sweep( + client: ImageClient, + pricing: dict[str, float], + args: argparse.Namespace, + spent_so_far: float = 0.0, +) -> list[ProbeResult]: + print(f">>> Image sweep ({len(IMAGE_TARGETS)} models)") + print() + results: list[ProbeResult] = [] + n = len(IMAGE_TARGETS) + warned = False + spent = spent_so_far + for i, target in enumerate(IMAGE_TARGETS, start=1): + if spent >= args.budget_cap: + print(f"[BUDGET-ABORT] ${spent:.4f} >= ${args.budget_cap:.2f}") + for remaining in IMAGE_TARGETS[i - 1 :]: + results.append( + ProbeResult( + model_id=remaining["model"], + modality="image", + status="skipped_budget", + latency_ms=0, + error_message=f"budget cap ${args.budget_cap:.2f} reached", + ) + ) + return results + if spent >= args.budget_cap * 0.8 and not warned: + print(f"[BUDGET-WARN] at ${spent:.4f} of ${args.budget_cap:.2f}") + warned = True + + result = probe_image(client, target, pricing) + results.append(result) + spent += result.cost_delta_usd + preview = preview_url(result.artifact_url or "") + if result.error_message: + preview = result.error_message[:60] + print( + f"[{i:02d}/{n}] {result.model_id:36s} {result.status:14s} " + f"{fmt_cost(result.cost_delta_usd)} {result.latency_ms:>6d}ms {preview}" + ) + if i < n: + time.sleep(1.0) + return results + + +def run_music_sweep( + pricing: dict[str, float], + args: argparse.Namespace, + spent_so_far: float = 0.0, +) -> list[ProbeResult]: + print(f">>> Music sweep ({len(MUSIC_TARGETS)} models)") + print() + client = MusicClient() + results: list[ProbeResult] = [] + n = len(MUSIC_TARGETS) + spent = spent_so_far + for i, model_id in enumerate(MUSIC_TARGETS, start=1): + if spent >= args.budget_cap: + print(f"[BUDGET-ABORT] ${spent:.4f} >= ${args.budget_cap:.2f}") + for remaining in MUSIC_TARGETS[i - 1 :]: + results.append( + ProbeResult( + model_id=remaining, + modality="music", + status="skipped_budget", + latency_ms=0, + error_message=f"budget cap ${args.budget_cap:.2f} reached", + ) + ) + return results + + result = probe_music(client, model_id, pricing) + results.append(result) + spent += result.cost_delta_usd + preview = preview_url(result.artifact_url or "") + if result.error_message: + preview = result.error_message[:60] + print( + f"[{i:02d}/{n}] {result.model_id:36s} {result.status:14s} " + f"{fmt_cost(result.cost_delta_usd)} {result.latency_ms:>6d}ms {preview}" + ) + if i < n: + time.sleep(1.0) + return results + + +def report( + results: list[ProbeResult], + started_at: float, + args: argparse.Namespace, + initial_balance: float, + final_balance: float, +) -> bool: + failures = [r for r in results if r.status != "ok"] + + if failures: + print() + print(">>> Failures") + for r in failures: + print(f" [{r.modality}] {r.model_id:36s} {r.status:14s}") + if r.error_message: + print(f" {r.error_message}") + + print() + print(">>> Modality summary") + by_mod: dict[str, list[ProbeResult]] = {} + for r in results: + by_mod.setdefault(r.modality, []).append(r) + for modality in sorted(by_mod): + rows = by_mod[modality] + ok = sum(1 for r in rows if r.status == "ok") + cost = sum(r.cost_delta_usd for r in rows) + avg_ms = sum(r.latency_ms for r in rows) / max(len(rows), 1) + print( + f" {modality:6s} {ok}/{len(rows)} ok cost={fmt_cost(cost)} " + f"avg_latency={avg_ms / 1000:.1f}s" + ) + + duration = time.monotonic() - started_at + minutes, seconds = divmod(int(duration), 60) + successes = [r for r in results if r.status == "ok"] + success_rate = len(successes) / max(len(results), 1) * 100 + total_cost = sum(r.cost_delta_usd for r in results) + + overall_pass = len(failures) == 0 + + actual_charged = max(0.0, initial_balance - final_balance) + + print() + print(">>> Summary") + print(f" est cost from pricing: {fmt_cost(total_cost)}") + print( + f" actual USDC charged : {fmt_cost(actual_charged)} " + f"(balance: ${initial_balance:.4f} -> ${final_balance:.4f})" + ) + print(f" total calls : {len(results)}") + print(f" success : {len(successes)}/{len(results)} ({success_rate:.0f}%)") + print(f" duration : {minutes}m {seconds}s") + print(f" budget used : {fmt_cost(actual_charged)} of {fmt_cost(args.budget_cap)}") + print(f" status : {'PASS' if overall_pass else 'FAIL'}") + print() + return overall_pass + + +def main() -> int: + parser = argparse.ArgumentParser( + description="Sweep test every image + music model in the BlockRun SDK." + ) + parser.add_argument("--budget-cap", type=float, default=1.00) + parser.add_argument("--skip-image", action="store_true") + parser.add_argument("--skip-music", action="store_true") + parser.add_argument("--output-json", type=str, default=None) + args = parser.parse_args() + + if args.skip_image and args.skip_music: + sys.stderr.write("ERROR: nothing to do — both --skip-image and --skip-music set\n") + return 2 + + started_at = time.monotonic() + image_client, image_pricing, music_pricing, initial_balance = preflight() + + results: list[ProbeResult] = [] + spent = 0.0 + if not args.skip_image: + image_results = run_image_sweep(image_client, image_pricing, args, spent) + results.extend(image_results) + spent += sum(r.cost_delta_usd for r in image_results) + if not args.skip_music: + results.extend(run_music_sweep(music_pricing, args, spent)) + + # Re-read balance to reconcile actual charged amount. + final_balance = initial_balance + try: + final_balance = LLMClient().get_balance() + except Exception: + pass + + overall_pass = report(results, started_at, args, initial_balance, final_balance) + + if args.output_json: + payload = { + "started_at": started_at, + "args": vars(args), + "results": [asdict(r) for r in results], + "pass": overall_pass, + } + with open(args.output_json, "w") as f: + json.dump(payload, f, indent=2) + print(f"results JSON written to {args.output_json}") + + return 0 if overall_pass else 1 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/pyproject.toml b/pyproject.toml index 66133bb..5014282 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -1,61 +1,111 @@ -[build-system] -requires = ["hatchling"] -build-backend = "hatchling.build" - -[project] -name = "blockrun-llm" -version = "0.1.0" -description = "BlockRun LLM Gateway SDK - Pay-per-request AI via x402 on Base" -readme = "README.md" -license = "MIT" -requires-python = ">=3.9" -authors = [ - { name = "BlockRun", email = "hello@blockrun.ai" } -] -keywords = ["llm", "ai", "x402", "base", "usdc", "micropayments", "openai", "claude", "gemini"] -classifiers = [ - "Development Status :: 4 - Beta", - "Intended Audience :: Developers", - "License :: OSI Approved :: MIT License", - "Programming Language :: Python :: 3", - "Programming Language :: Python :: 3.9", - "Programming Language :: Python :: 3.10", - "Programming Language :: Python :: 3.11", - "Programming Language :: Python :: 3.12", - "Topic :: Scientific/Engineering :: Artificial Intelligence", -] -dependencies = [ - "httpx>=0.25.0", - "eth-account>=0.11.0", - "pydantic>=2.0.0", - "python-dotenv>=1.0.0", -] - -[project.optional-dependencies] -dev = [ - "pytest>=7.0.0", - "pytest-asyncio>=0.21.0", - "black>=23.0.0", - "mypy>=1.0.0", - "ruff>=0.1.0", -] - -[project.urls] -Homepage = "https://blockrun.ai" -Documentation = "https://docs.blockrun.ai" -Repository = "https://github.com/blockrun/blockrun-llm" - -[tool.hatch.build.targets.wheel] -packages = ["blockrun_llm"] - -[tool.black] -line-length = 100 -target-version = ["py39"] - -[tool.ruff] -line-length = 100 -target-version = "py39" - -[tool.mypy] -python_version = "3.9" -strict = true +[build-system] +requires = ["hatchling"] +build-backend = "hatchling.build" + +[project] +name = "blockrun-llm" +version = "1.17.0" +description = "BlockRun SDK - Pay-per-request AI (LLM, Image, Video, Music, Voice) via x402 on Base and Solana" +readme = "README.md" +license = "MIT" +requires-python = ">=3.9" +authors = [ + { name = "BlockRun", email = "hello@blockrun.ai" } +] +keywords = ["llm", "ai", "x402", "base", "usdc", "micropayments", "openai", "claude", "gemini", "nvidia", "zai", "free-models", "image-generation", "dall-e"] +classifiers = [ + "Development Status :: 4 - Beta", + "Intended Audience :: Developers", + "License :: OSI Approved :: MIT License", + "Programming Language :: Python :: 3", + "Programming Language :: Python :: 3.9", + "Programming Language :: Python :: 3.10", + "Programming Language :: Python :: 3.11", + "Programming Language :: Python :: 3.12", + "Topic :: Scientific/Engineering :: Artificial Intelligence", +] +dependencies = [ + "httpx>=0.25.0", + "eth-account>=0.11.0", + "pydantic>=2.0.0", + "python-dotenv>=1.0.0", + "qrcode[pil]>=7.0", +] + +[project.optional-dependencies] +dev = [ + "pytest>=7.0.0", + "pytest-asyncio>=0.21.0", + "black==24.10.0", # Pin version for consistent formatting + "mypy>=1.0.0", + "ruff==0.16.0", # Pin version: an unpinned linter breaks CI with no code change +] +anthropic = [ + # Anthropic 1.x switched to httpx2; the x402 transport uses httpx. + # Keep the supported major until both payment modes are migrated together. + "anthropic>=0.40.0,<1", +] +solana = [ + "x402[svm]>=2.0.0", + # solana 0.40.0 (2026-06-27) removed solana.rpc.api, which x402's SVM + # signer imports — fresh installs get _HAS_X402=False. Lift when x402 + # supports the new layout. + "solana>=0.36,<0.40", +] + +[project.urls] +Homepage = "https://blockrun.ai" +Documentation = "https://github.com/BlockRunAI/awesome-blockrun/tree/main/docs" +Repository = "https://github.com/BlockRunAI/blockrun-llm" + +[tool.hatch.build.targets.wheel] +packages = ["blockrun_llm"] + +[tool.hatch.build.targets.sdist] +exclude = [ + "sweep-*.json", + "sweep-*.log", + ".claude/", + ".ruff_cache/", + "*.png", +] + +[tool.black] +line-length = 100 +target-version = ["py39"] + +[tool.ruff] +line-length = 100 +target-version = "py39" + +[tool.ruff.lint] +# Deferred, NOT blessed. Each of these is a behaviour change in a payments SDK +# and belongs in its own reviewable PR, not folded into a typing sweep: +# +# BLE001 157 blind `except Exception` — several are deliberate best-effort +# paths (telemetry, cleanup) where raising would be worse than +# swallowing. Which ones are deliberate has to be read case by case. +# S110/ 54 try/except/pass and try/except/continue — same question. +# S112 +# TRY004 8 raising something other than TypeError on a type check. +# DTZ005/ 4 naive datetime.now()/fromtimestamp() in cache.py, tx_log.py and +# DTZ006 wallet.py. Worth fixing — a transaction log without timezone is +# genuinely ambiguous — but it changes recorded values, so it needs +# its own change and its own migration thought. +# RUF012 2 mutable class defaults, both in examples/ and tests/. +ignore = ["BLE001", "S110", "S112", "TRY004", "DTZ005", "DTZ006", "RUF012"] + +[tool.mypy] +python_version = "3.9" +strict = true + +[tool.ruff.lint.per-file-ignores] +# pydantic EVALUATES annotations at runtime to build each model. Under +# `from __future__ import annotations` they are strings, and on Python 3.9 +# evaluating "str | None" is a TypeError — PEP 604 does not exist there. +# +# Parsing is not the constraint; evaluation is. compileall passes on 3.9 and +# the import still fails, which is exactly how this got shipped to CI once. +# +# So this file keeps typing.Optional/List and no future import. +"blockrun_llm/types.py" = ["FA100", "UP006", "UP007", "UP035", "UP045"] diff --git a/pytest.ini b/pytest.ini index 8648856..81f76ce 100644 --- a/pytest.ini +++ b/pytest.ini @@ -1,20 +1,16 @@ -[pytest] -testpaths = tests -python_files = test_*.py -python_classes = Test* -python_functions = test_* -addopts = - -v - --strict-markers - --tb=short - --cov=blockrun_llm - --cov-report=term-missing - --cov-report=html - --cov-fail-under=85 -markers = - integration: Integration tests requiring API access and funded wallet - unit: Unit tests (run by default) - asyncio: Async tests using pytest-asyncio - -# Async test configuration -asyncio_mode = auto +[pytest] +testpaths = tests +python_files = test_*.py +python_classes = Test* +python_functions = test_* +addopts = + -v + --strict-markers + --tb=short +markers = + integration: Integration tests requiring API access and funded wallet + unit: Unit tests (run by default) + asyncio: Async tests using pytest-asyncio + +# Async test configuration +asyncio_mode = auto diff --git a/scripts/sync-brand-numbers.mjs b/scripts/sync-brand-numbers.mjs new file mode 100644 index 0000000..bd4ab15 --- /dev/null +++ b/scripts/sync-brand-numbers.mjs @@ -0,0 +1,426 @@ +#!/usr/bin/env node +/** + * Sync marketing numbers from BlockRun's canonical brand artifact. + * + * This file is copied byte-for-byte into every public repo as + * scripts/sync-brand-numbers.mjs. It is a copy rather than an npm package on + * purpose: a package would mean 37 dependency bumps, and several consuming + * repos have no package.json at all. Zero dependencies, plain Node. + * + * node scripts/sync-brand-numbers.mjs rewrite markers in place + * node scripts/sync-brand-numbers.mjs --check exit 1 on drift, write nothing + * node scripts/sync-brand-numbers.mjs --refresh re-fetch the artifact first + * + * --check NEVER touches the network. PR CI must be deterministic and offline: + * if it fetched, a deploy in progress would fail every repo in the org at once. + * Freshness is the fan-out job's problem, not the pull request's. + * + * Markers look like: 66 + * and wrap the WHOLE token, so a badge URL, its alt text and the prose number + * can all regenerate from one key. + * + * THIS COPY IS AHEAD OF THE SOURCE. blockrun's `brand-script-sync` CI job + * diffs every consumer against brand/sync-brand-numbers.mjs and its printed + * remediation is "copy the source over the consumer" — twice that overwrote a + * fix made here (#84, #128). What this copy carries that the source does not, + * as of 2026-09-13: assertRenderable + escAttr (a value from the mirror is + * refused, and attribute-escaped, before it is written into a README that the + * brand-sync bot then pushes unattended with contents:write), keyOf() on the + * keys-in-use count, and the --check summary that does not say "up to date" + * under a list of stale fenced markers. Resync source <- consumer: land THIS + * file in blockrun/brand and fan it out; do not copy the source over it. + * test/brand-sync-script.test.ts fails on a copy without the guard, so a + * consumer <- source resync cannot pass this repo's required `test` check. + */ +import { execFileSync } from "node:child_process"; +import { existsSync, lstatSync, readFileSync, writeFileSync, readdirSync } from "node:fs"; +import { join, relative, extname } from "node:path"; + +const ROOT = process.cwd(); +const SNAPSHOT = join(ROOT, "brand-numbers.json"); +// ORIGIN is tried first because it IS the truth — the mirror can only ever be +// as fresh as the last time someone refreshed it. The mirror exists so a repo +// can still sync while blockrun.ai is down, not to front the origin. +// +// The mirror is awesome-blockrun's own brand-numbers.json: that repo consumes +// the artifact like every other, and its snapshot doubles as the org's copy. +// One file, one role per repo, nothing to keep in step by hand. +const ORIGIN = "https://blockrun.ai/brand/numbers.json"; +const MIRROR = + "https://raw.githubusercontent.com/BlockRunAI/awesome-blockrun/main/brand-numbers.json"; + +const argv = new Set(process.argv.slice(2)); +const check = argv.has("--check"); +const refresh = argv.has("--refresh"); + +const SKIP_DIRS = new Set([ + "node_modules", ".git", "dist", "build", "out", ".next", "coverage", + "vendor", "target", "__pycache__", ".venv", "venv", +]); +// .txt is here for llms.txt, which is a first-class marketing surface: it is +// what agents read to find out what BlockRun serves. Scanning other .txt files +// costs a read and changes nothing — only files with markers are ever written. +const TEXT_EXT = new Set([".md", ".mdx", ".txt"]); + +/* ── 1. numbers ──────────────────────────────────────────────────────────── */ + +async function loadNumbers() { + if (!refresh) { + try { + return JSON.parse(readFileSync(SNAPSHOT, "utf8")); + } catch { + fail( + `no brand-numbers.json in ${ROOT}\n` + + ` run with --refresh once to seed it from ${ORIGIN}`, + ); + } + } + for (const url of [ORIGIN, MIRROR]) { + try { + const res = await fetch(url, { signal: AbortSignal.timeout(10_000) }); + if (!res.ok) continue; + const json = await res.json(); + writeFileSync(SNAPSHOT, `${JSON.stringify(json, null, 2)}\n`); + return json; + } catch { + /* try the next source */ + } + } + fail(`could not refresh from ${MIRROR} or ${ORIGIN}`); +} + +/** Flatten nested numbers into dotted keys, ignoring $comment / rationale prose. */ +function flatten(obj, prefix = "") { + return Object.entries(obj).flatMap(([k, v]) => { + if (k.startsWith("$")) return []; + const key = `${prefix}${k}`; + if (v && typeof v === "object" && !Array.isArray(v)) return flatten(v, `${key}.`); + if (v === null) return []; + return [[key, v]]; + }); +} + +/* ── 2. renderers ────────────────────────────────────────────────────────── */ + +/** + * How a key becomes text. Default is the bare value. + * + * A marker may carry an `@modifier` — `` — which + * selects a renderer without changing which number is looked up. The modifier + * is what makes a key reusable: the same mcp.tools appears as a shields badge + * at the top of a README and as a bare "19 tools" in a table two screens down, + * and one marker still keeps the badge URL, its alt text and the label in step. + * + * Renderers are registered under the FULL marker name so a badge's label is + * written out rather than guessed from the key. + */ +/** + * What a brand value is allowed to be, checked at the moment it is USED. + * + * These values arrive over the network from blockrun.ai (or the + * awesome-blockrun mirror) and are written verbatim into README.md, + * CONTRIBUTING.md and skills/*\/SKILL.md, which `.github/workflows/brand-sync.yml` + * then commits and pushes to the default branch weekly, unattended, with + * `contents: write`. Rendering was `String(value)` and the badge renderer + * interpolated straight into `src="..."` and `alt="..."`, so a value carrying + * a quote or an angle bracket closed the attribute and injected markup into + * every consuming repo's README. Write access to one mirror repo was enough. + * + * Checked here rather than over the whole artifact on purpose: the payload + * legitimately carries prose fields we never render (`savings.baselineModel` + * is a string), and refusing those would break the sync on an unrelated + * addition upstream. + */ +const SAFE_TEXT = /^[\p{L}\p{N} .,%+/·—–-]{1,64}$/u; + +function assertRenderable(marker, value) { + const what = () => `${marker} = ${JSON.stringify(value)}`; + if (typeof value === "number") { + if (!Number.isFinite(value)) fail(`brand-numbers: refusing to render ${what()} — not a finite number`); + return value; + } + if (typeof value === "string") { + if (!SAFE_TEXT.test(value)) { + fail( + `brand-numbers: refusing to render ${what()} — a rendered value must be ` + + `a number or a short plain label. This value would be written verbatim ` + + `into README/CONTRIBUTING/SKILL.md and pushed by the brand-sync bot.`, + ); + } + return value; + } + fail(`brand-numbers: refusing to render ${what()} — expected a number or a string, got ${Array.isArray(value) ? "an array" : typeof value}`); +} + +/** Escape for an HTML attribute. Belt to assertRenderable's braces. */ +const escAttr = (v) => + String(v).replace(/&/g, "&").replace(//g, ">") + .replace(/"/g, """).replace(/'/g, "'"); + +const badge = (label) => (n) => + `${escAttr(n)} ${label}`; + +const RENDER = { + "mcp.tools@badge": badge("tools"), + "models.totalVisible@badge": badge("models"), + "models.chatVisible@badge": badge("models"), +}; +const render = (marker, value) => (RENDER[marker] ?? String)(assertRenderable(marker, value)); + +/** `mcp.tools@badge` looks up `mcp.tools`. Unmodified markers are unaffected. */ +const keyOf = (marker) => marker.split("@")[0]; + +/** + * `@live` opts a marker INTO rewriting inside a fenced block. + * + * The fence skip exists so a code block demonstrating the marker syntax is not + * itself rewritten. But a fence is also how you draw an ASCII diagram, and a + * number inside one is live marketing copy, not documentation. Franklin's README + * had three model counts inside box-drawing blocks: silently skipped by the + * rewriter AND by --check, so they would have gone stale while the same number + * updated five lines above, with CI reporting "up to date" throughout. That is + * the exact hand-maintained staleness this tool exists to remove. + * + * `live` occupies the modifier slot, so it cannot be combined with a renderer + * modifier like `@badge`. A badge inside a code fence is not a thing worth + * supporting; a bare number inside a diagram very much is. + */ +const isLive = (marker) => marker.split("@")[1] === "live"; + +/* ── 3. marker rewriting ─────────────────────────────────────────────────── */ + +const esc = (s) => s.replace(/[.*+?^${}()|[\]\\]/g, "\\$&"); +const OPEN_ANY = //g; +const CLOSE_ANY = //g; + +/** Byte ranges of fenced code blocks — markers inside them are documentation. */ +function fencedRanges(text) { + const ranges = []; + const fence = /^(\s*)(`{3,}|~{3,})[^\n]*$/gm; + let open = null; + for (let m; (m = fence.exec(text)); ) { + if (open === null) open = m.index; + else { + ranges.push([open, m.index + m[0].length]); + open = null; + } + } + // An unterminated fence used to be dropped, which silently reclassified the + // whole tail of the file as live prose and rewrote every example after it. + // A fence with no partner runs to EOF — that is how the renderers read it too. + if (open !== null) ranges.push([open, text.length]); + return ranges; +} + +function syncFile(file, numbers, problems, skipped) { + const before = readFileSync(file, "utf8"); + const rel = relative(ROOT, file); + const fenced = fencedRanges(before); + const inFence = (i) => fenced.some(([a, b]) => i >= a && i < b); + const known = new Map(numbers); + const used = new Set(); + + // Markers actually present, so a file is only ever rewritten for what it uses + // and an @modifier is carried through to the renderer verbatim. + const markers = new Set(); + // A marker naming a key that does not exist is an error, never a silent + // no-op: a typo'd marker would otherwise sit there looking synced forever. + for (const [re, shown] of [ + [OPEN_ANY, (n) => ``], + [CLOSE_ANY, (n) => ``], + ]) { + for (const m of before.matchAll(re)) { + // A fenced marker is documentation and is neither collected nor validated + // — unless it is @live, which is explicitly live content that happens to + // sit in a diagram, so it must be discovered and key-checked like any other. + if (!isLive(m[1]) && inFence(m.index)) continue; + markers.add(m[1]); + if (!known.has(keyOf(m[1]))) problems.push(`${rel}: unknown key ${shown(m[1])}`); + } + } + + let after = before; + for (const marker of markers) { + const key = keyOf(marker); + if (!known.has(key)) continue; + const value = known.get(key); + const live = isLive(marker); + const pair = new RegExp( + `()([\\s\\S]*?)()`, + "g", + ); + // Fence ranges must be recomputed against the string we are about to search. + // `replace` reports offsets into ITS OWN input, and `after` has already been + // rewritten by every earlier marker in this loop — a badge turns 2 characters + // into ~150 — so ranges measured on `before` drift further out of alignment + // with each pass and start judging the wrong side of a fence boundary. + const fencedNow = fencedRanges(after); + const inFenceNow = (i) => fencedNow.some(([a, b]) => i >= a && i < b); + after = after.replace(pair, (whole, open, inner, close, offset) => { + if (!live && inFenceNow(offset)) return whole; + // Nesting means the closing tag of an inner marker would be consumed by + // the outer one. Refuse rather than produce mangled output. + if (/`, "g"))] + .filter(counted).length; + const closes = [...before.matchAll(new RegExp(``, "g"))] + .filter(counted).length; + if (opens !== closes) problems.push(`${rel}: unbalanced marker br:${marker} (${opens} open, ${closes} close)`); + } + + // Skipping a fenced marker is right for a documentation example and wrong for + // a number inside a diagram, and only a human can tell those apart. Report the + // ones whose value actually disagrees with the artifact — those are numbers + // going stale under a green build. An example already showing the right value + // stays quiet, so repos that document the syntax get no noise. + // + // Scanned over `before` rather than recorded during rewriting because a marker + // that appears ONLY inside a fence is dropped at discovery and never reaches + // the replace loop at all. + const ANY_PAIR = /([\s\S]*?)/g; + for (const m of before.matchAll(ANY_PAIR)) { + const name = m[1]; + if (isLive(name) || !inFence(m.index)) continue; + const key = keyOf(name); + if (!known.has(key) || m[2] === render(name, known.get(key))) continue; + skipped.push(`${rel}:${before.slice(0, m.index).split("\n").length} br:${name}`); + } + + return { before, after, changed: before !== after, used }; +} + +/* ── 4. walk ─────────────────────────────────────────────────────────────── */ + +/** + * The set of paths git tracks, or null when this is not a git checkout. + * + * Without this the walker rewrote ANY .md/.txt it could reach, including files + * git ignores — a developer's scratch notes, a generated docs/ output — and + * --check then reported drift against files that are not in the repo at all. CI + * never saw it (a fresh checkout contains only tracked files), so it was purely + * a local mystery. Returning null on a non-git tree keeps the tool usable in a + * bare directory, which is how it is often run the first time. + */ +function trackedFiles() { + try { + const out = execFileSync("git", ["ls-files", "-z"], { + cwd: ROOT, + encoding: "utf8", + stdio: ["ignore", "pipe", "ignore"], + }); + const set = new Set(out.split("\0").filter(Boolean).map((rel) => join(ROOT, rel))); + return set.size ? set : null; + } catch { + return null; + } +} + +const TRACKED = trackedFiles(); + +function* walk(dir) { + for (const name of readdirSync(dir)) { + if (SKIP_DIRS.has(name)) continue; + const p = join(dir, name); + // lstat, not stat: a symlinked directory is reached by its real path or not + // at all. blockrun's docs/ -> awesome-blockrun/docs is exactly the case that + // matters — following it would edit a submodule's files behind the skip + // below, and a link pointing at an ancestor would recurse forever. + const s = lstatSync(p); + if (s.isSymbolicLink()) continue; + if (s.isDirectory()) { + // A nested repo is a submodule or vendored checkout: it carries its own + // brand-numbers.json and syncs itself. Rewriting its markers from THIS + // repo's snapshot would dirty a submodule nobody asked us to touch, and + // would report drift that belongs to another repo's CI. + if (existsSync(join(p, ".git"))) continue; + yield* walk(p); + } else if (TEXT_EXT.has(extname(name)) && (TRACKED === null || TRACKED.has(p))) yield p; + } +} + +function fail(msg) { + console.error(`brand-numbers: ${msg}`); + process.exit(1); +} + +/* ── 5. run ──────────────────────────────────────────────────────────────── */ + +const raw = await loadNumbers(); +const numbers = flatten(raw); +const problems = []; +const drifted = []; +const skipped = []; +const everUsed = new Set(); + +for (const file of walk(ROOT)) { + const { before, after, changed, used } = syncFile(file, numbers, problems, skipped); + // keyOf: mcp.tools and mcp.tools@badge are ONE key in use, not two. + used.forEach((k) => everUsed.add(keyOf(k))); + if (!changed) continue; + drifted.push({ file: relative(ROOT, file), before, after }); + if (!check) writeFileSync(file, after); +} + +// Non-fatal on purpose: this ships to 37 repos at once, and a hard failure would +// break every one of them that documents the marker syntax. Visible, not fatal. +if (skipped.length) { + console.error( + `brand-numbers: ${skipped.length} marker(s) skipped inside code fences ` + + `(add @live to sync one, e.g. ):`, + ); + for (const s of skipped) console.error(` ${s}`); + console.error(""); +} + +if (problems.length) { + for (const p of problems) console.error(` ${p}`); + fail(`${problems.length} marker problem(s)`); +} + +if (check) { + if (drifted.length === 0) { + // Do not say "up to date" straight after listing markers known to be + // stale. The skip stays non-fatal for the reason above, but a CI log that + // prints the stale ones and then declares everything current is a log + // nobody reads twice. + console.log( + skipped.length + ? `brand-numbers: no drift outside code fences (${everUsed.size} keys in use), ` + + `but ${skipped.length} fenced marker(s) listed above are stale — add @live to sync them` + : `brand-numbers: up to date (${everUsed.size} keys in use)`, + ); + process.exit(0); + } + console.error("brand-numbers: these files disagree with brand-numbers.json\n"); + for (const { file, before, after } of drifted) { + const b = before.split("\n"); + const a = after.split("\n"); + for (let i = 0; i < Math.max(b.length, a.length); i++) { + if (b[i] !== a[i]) { + console.error(` ${file}:${i + 1}`); + console.error(` - ${(b[i] ?? "").trim()}`); + console.error(` + ${(a[i] ?? "").trim()}`); + } + } + } + console.error( + "\n fix with: node scripts/sync-brand-numbers.mjs && git commit -am 'chore: sync brand numbers'", + ); + process.exit(1); +} + +console.log( + drifted.length + ? `brand-numbers: updated ${drifted.length} file(s)` + : `brand-numbers: already up to date (${everUsed.size} keys in use)`, +); diff --git a/tests/__init__.py b/tests/__init__.py index ec7d357..005b3b8 100644 --- a/tests/__init__.py +++ b/tests/__init__.py @@ -1 +1 @@ -"""Tests for BlockRun LLM SDK.""" +"""Tests for BlockRun LLM SDK.""" diff --git a/tests/conftest.py b/tests/conftest.py new file mode 100644 index 0000000..5df3843 --- /dev/null +++ b/tests/conftest.py @@ -0,0 +1,18 @@ +"""Test-wide fixtures. + +``BLOCKRUN_API_KEY`` is cleared for every test. Without this, every test that +asserts "no credential configured" — and every test that expects a wallet +client — fails on the machine of anyone who actually has a key exported, which +is every developer working on the API-key rail. Tests that want the variable +set it explicitly with ``monkeypatch.setenv``, which overrides this. +""" + +import pytest + +from blockrun_llm.apikey import ENV_API_KEY, ENV_API_KEY_URL + + +@pytest.fixture(autouse=True) +def _clear_api_key_env(monkeypatch): + monkeypatch.delenv(ENV_API_KEY, raising=False) + monkeypatch.delenv(ENV_API_KEY_URL, raising=False) diff --git a/tests/helpers.py b/tests/helpers.py index 2dc5d91..990a209 100644 --- a/tests/helpers.py +++ b/tests/helpers.py @@ -1,151 +1,154 @@ -""" -Test utilities and mock builders for BlockRun LLM SDK tests. -""" - -import json -import base64 -from typing import Dict, Any, Optional -from eth_account import Account - -# Test private key (DO NOT use in production) -# This is a well-known test key from Hardhat/Foundry -TEST_PRIVATE_KEY = "0xac0974bec39a17e36ba4a6b4d238ff944bacb478cbed5efcae784d7bf4f2ff80" - -# Test account derived from TEST_PRIVATE_KEY -# Address: 0xf39Fd6e51aad88F6F4ce6aB8827279cffFb92266 -TEST_ACCOUNT = Account.from_key(TEST_PRIVATE_KEY) - -# Test recipient address for payment mocks -TEST_RECIPIENT = "0x70997970C51812dc3A010C7d01b50e0d17dc79C8" - - -def build_payment_required_response( - amount: str = "1000000", - recipient: str = TEST_RECIPIENT, - network: str = "eip155:8453", - resource: Optional[Dict[str, str]] = None, -) -> str: - """Build a mock 402 Payment Required response.""" - payment_required = { - "x402Version": 2, - "accepts": [ - { - "scheme": "exact", - "network": network, - "amount": amount, - "asset": "0x833589fCD6eDb6E08f4c7C32D4f71b54bdA02913", - "payTo": recipient, - "maxTimeoutSeconds": 300, - "extra": {"name": "USD Coin", "version": "2"}, - } - ], - "resource": resource - or { - "url": "https://api.blockrun.ai/v1/chat/completions", - "description": "BlockRun AI API call", - }, - } - - return base64.b64encode(json.dumps(payment_required).encode()).decode() - - -def build_chat_response( - content: str = "This is a test response.", - model: str = "gpt-4o", - prompt_tokens: int = 10, - completion_tokens: int = 20, -) -> Dict[str, Any]: - """Build a mock successful chat response.""" - return { - "id": "chatcmpl-test123", - "object": "chat.completion", - "created": 1234567890, - "model": model, - "choices": [ - { - "index": 0, - "message": {"role": "assistant", "content": content}, - "finish_reason": "stop", - } - ], - "usage": { - "prompt_tokens": prompt_tokens, - "completion_tokens": completion_tokens, - "total_tokens": prompt_tokens + completion_tokens, - }, - } - - -def build_error_response( - error: str = "Test error message", - code: str = "test_error", - include_sensitive: bool = True, -) -> Dict[str, Any]: - """Build a mock error response.""" - response = {"error": error, "code": code} - - if include_sensitive: - # These should be filtered out by sanitization - response.update( - { - "internal_stack": "/var/app/handler.py:123", - "api_key": "secret_key_should_be_filtered", - "database_url": "postgres://user:pass@host/db", - } - ) - - return response - - -def build_models_response() -> Dict[str, Any]: - """Build a mock models list response.""" - return { - "data": [ - { - "id": "openai/gpt-4o", - "provider": "openai", - "name": "GPT-4o", - "inputPrice": 2.5, - "outputPrice": 10.0, - }, - { - "id": "anthropic/claude-sonnet-4.5", - "provider": "anthropic", - "name": "Claude Sonnet 4.5", - "inputPrice": 3.0, - "outputPrice": 15.0, - }, - { - "id": "google/gemini-2.5-flash", - "provider": "google", - "name": "Gemini 2.5 Flash", - "inputPrice": 0.15, - "outputPrice": 0.6, - }, - ] - } - - -class MockResponse: - """Mock HTTP response for testing.""" - - def __init__( - self, - status_code: int, - json_data: Optional[Dict[str, Any]] = None, - text_data: Optional[str] = None, - headers: Optional[Dict[str, str]] = None, - ): - self.status_code = status_code - self._json_data = json_data - self._text_data = text_data or (json.dumps(json_data) if json_data else "") - self.headers = headers or {} - - def json(self) -> Dict[str, Any]: - if self._json_data is None: - raise ValueError("No JSON data available") - return self._json_data - - @property - def text(self) -> str: - return self._text_data +""" +Test utilities and mock builders for BlockRun LLM SDK tests. +""" + +from __future__ import annotations + +import base64 +import json +from typing import Any + +from eth_account import Account + +# Test private key (DO NOT use in production) +# This is a well-known test key from Hardhat/Foundry +TEST_PRIVATE_KEY = "0xac0974bec39a17e36ba4a6b4d238ff944bacb478cbed5efcae784d7bf4f2ff80" + +# Test account derived from TEST_PRIVATE_KEY +# Address: 0xf39Fd6e51aad88F6F4ce6aB8827279cffFb92266 +TEST_ACCOUNT = Account.from_key(TEST_PRIVATE_KEY) + +# Test recipient address for payment mocks +TEST_RECIPIENT = "0x70997970C51812dc3A010C7d01b50e0d17dc79C8" + + +def build_payment_required_response( + amount: str = "1000000", + recipient: str = TEST_RECIPIENT, + network: str = "eip155:8453", + resource: dict[str, str] | None = None, +) -> str: + """Build a mock 402 Payment Required response.""" + payment_required = { + "x402Version": 2, + "accepts": [ + { + "scheme": "exact", + "network": network, + "amount": amount, + "asset": "0x833589fCD6eDb6E08f4c7C32D4f71b54bdA02913", + "payTo": recipient, + "maxTimeoutSeconds": 300, + "extra": {"name": "USD Coin", "version": "2"}, + } + ], + "resource": resource + or { + "url": "https://api.blockrun.ai/v1/chat/completions", + "description": "BlockRun AI API call", + }, + } + + return base64.b64encode(json.dumps(payment_required).encode()).decode() + + +def build_chat_response( + content: str = "This is a test response.", + model: str = "gpt-5.2", + prompt_tokens: int = 10, + completion_tokens: int = 20, +) -> dict[str, Any]: + """Build a mock successful chat response.""" + return { + "id": "chatcmpl-test123", + "object": "chat.completion", + "created": 1234567890, + "model": model, + "choices": [ + { + "index": 0, + "message": {"role": "assistant", "content": content}, + "finish_reason": "stop", + } + ], + "usage": { + "prompt_tokens": prompt_tokens, + "completion_tokens": completion_tokens, + "total_tokens": prompt_tokens + completion_tokens, + }, + } + + +def build_error_response( + error: str = "Test error message", + code: str = "test_error", + include_sensitive: bool = True, +) -> dict[str, Any]: + """Build a mock error response.""" + response = {"error": error, "code": code} + + if include_sensitive: + # These should be filtered out by sanitization + response.update( + { + "internal_stack": "/var/app/handler.py:123", + "api_key": "secret_key_should_be_filtered", + "database_url": "postgres://user:pass@host/db", + } + ) + + return response + + +def build_models_response() -> dict[str, Any]: + """Build a mock models list response.""" + return { + "data": [ + { + "id": "openai/gpt-5.2", + "provider": "openai", + "name": "GPT-5.2", + "inputPrice": 2.5, + "outputPrice": 10.0, + }, + { + "id": "anthropic/claude-sonnet-4.6", + "provider": "anthropic", + "name": "Claude Sonnet 4.6", + "inputPrice": 3.0, + "outputPrice": 15.0, + }, + { + "id": "google/gemini-2.5-flash", + "provider": "google", + "name": "Gemini 2.5 Flash", + "inputPrice": 0.15, + "outputPrice": 0.6, + }, + ] + } + + +class MockResponse: + """Mock HTTP response for testing.""" + + def __init__( + self, + status_code: int, + json_data: dict[str, Any] | None = None, + text_data: str | None = None, + headers: dict[str, str] | None = None, + ): + self.status_code = status_code + self._json_data = json_data + self._text_data = text_data or (json.dumps(json_data) if json_data else "") + self.headers = headers or {} + + def json(self) -> dict[str, Any]: + if self._json_data is None: + raise ValueError("No JSON data available") + return self._json_data + + @property + def text(self) -> str: + return self._text_data diff --git a/tests/integration/EXA_E2E_TEST_NOTE.md b/tests/integration/EXA_E2E_TEST_NOTE.md new file mode 100644 index 0000000..14d9498 --- /dev/null +++ b/tests/integration/EXA_E2E_TEST_NOTE.md @@ -0,0 +1,123 @@ +# Exa Web Search — E2E Integration Test Note + +**Feature:** Exa neural web search via sol.blockrun.ai +**Payment:** Solana USDC (x402) +**Estimated cost per full run:** ~$0.04 +**Date added:** 2026-03-31 + +--- + +## What's Being Tested + +| Test | Endpoint | Expected Cost | +|---|---|---| +| `test_exa_search` | `POST /api/v1/exa/search` | $0.01 | +| `test_exa_find_similar` | `POST /api/v1/exa/find-similar` | $0.01 | +| `test_exa_contents` | `POST /api/v1/exa/contents` | $0.002/URL | +| `test_exa_answer` | `POST /api/v1/exa/answer` | $0.01 | +| `test_exa_generic_proxy` | `POST /api/v1/exa/search` (via `exa()`) | $0.01 | +| `test_exa_spending_tracked` | Session tracking across all calls | — | + +--- + +## Setup + +### 1. Install the SDK + +```bash +pip install "blockrun-llm[solana]" +``` + +Or from source: + +```bash +git clone https://github.com/BlockRunAI/blockrun-llm +cd blockrun-llm +pip install -e ".[solana,dev]" +``` + +### 2. Prepare a Solana wallet with USDC + +- You need a Solana mainnet wallet with at least **$0.10 USDC** (covers multiple runs) +- The private key must be **bs58-encoded** (64-byte keypair, standard Solana format) +- USDC mint: `EPjFWdd5AufqSSqeM2qN1xzybapC8G4wEGGkZwyTDt1v` + +### 3. Set environment variable + +```bash +export SOLANA_WALLET_KEY="your-bs58-private-key-here" +``` + +--- + +## Run the Tests + +```bash +# Run only Exa E2E tests +pytest tests/integration -k TestSolanaExa -v + +# Run all integration tests (Base + Solana) +pytest tests/integration -v +``` + +Expected output: + +``` +tests/integration/test_production_api.py::TestSolanaExa::test_exa_search PASSED + ✓ exa_search: 3 results, cost=$0.0100 +tests/integration/test_production_api.py::TestSolanaExa::test_exa_find_similar PASSED + ✓ exa_find_similar: 3 results +tests/integration/test_production_api.py::TestSolanaExa::test_exa_contents PASSED + ✓ exa_contents: response received +tests/integration/test_production_api.py::TestSolanaExa::test_exa_answer PASSED + ✓ exa_answer: response received +tests/integration/test_production_api.py::TestSolanaExa::test_exa_generic_proxy PASSED + ✓ exa() generic: 2 results +tests/integration/test_production_api.py::TestSolanaExa::test_exa_spending_tracked PASSED + ✓ Spending: $0.0400 over 5 calls +``` + +--- + +## Manual API Smoke Test (no wallet needed) + +Verify endpoints are live and pricing is correct: + +```bash +# search — expect $0.0100 +curl -s -X POST https://sol.blockrun.ai/api/v1/exa/search \ + -H "Content-Type: application/json" \ + -d '{"query":"test"}' | python3 -m json.tool | grep -E '"amount"|"network"|"endpoint"' + +# contents with 2 URLs — expect $0.0040 ($0.002 × 2) +curl -s -X POST https://sol.blockrun.ai/api/v1/exa/contents \ + -H "Content-Type: application/json" \ + -d '{"urls":["https://a.com","https://b.com"]}' | python3 -m json.tool | grep '"amount"' + +# discovery +curl -s https://sol.blockrun.ai/api/.well-known/x402 | python3 -m json.tool | grep exa +``` + +All should return HTTP 402 with correct `price` and `network: solana`. + +--- + +## Pass Criteria + +- All 6 `TestSolanaExa` tests pass +- `exa_search` cost is exactly $0.01 (±$0.0001) +- `exa_contents` cost scales correctly with number of URLs +- Session `total_usd` and `calls` are tracked accurately +- No `APIError` or `PaymentError` raised on valid requests + +## Fail Criteria + +- HTTP 503 → `EXA_API_KEY` not configured in Cloud Run (contact DevOps) +- HTTP 402 after payment → wallet has insufficient USDC balance +- `AssertionError` on result structure → Exa API response format changed + +--- + +## Contact + +Questions → @bc1max on Telegram diff --git a/tests/integration/__init__.py b/tests/integration/__init__.py index 98e40eb..9b84820 100644 --- a/tests/integration/__init__.py +++ b/tests/integration/__init__.py @@ -1 +1 @@ -"""Integration tests for BlockRun LLM SDK.""" +"""Integration tests for BlockRun LLM SDK.""" diff --git a/tests/integration/conftest.py b/tests/integration/conftest.py index 759ff2d..2e2aee5 100644 --- a/tests/integration/conftest.py +++ b/tests/integration/conftest.py @@ -1,27 +1,27 @@ -"""Pytest configuration for integration tests.""" - -import os -import pytest - - -def pytest_configure(config): - """Configure pytest with custom markers.""" - config.addinivalue_line( - "markers", - "integration: Integration tests requiring funded wallet and API access" - ) - - -@pytest.fixture(scope="session") -def wallet_private_key(): - """Get wallet private key from environment variable. - - Returns None if not set, which will cause integration tests to be skipped. - """ - return os.environ.get("BLOCKRUN_WALLET_KEY") - - -@pytest.fixture(scope="session") -def production_api_url(): - """Get production API URL.""" - return "https://api.blockrun.ai" +"""Pytest configuration for integration tests.""" + +import os + +import pytest + + +def pytest_configure(config): + """Configure pytest with custom markers.""" + config.addinivalue_line( + "markers", "integration: Integration tests requiring funded wallet and API access" + ) + + +@pytest.fixture(scope="session") +def wallet_private_key(): + """Get wallet private key from environment variable. + + Returns None if not set, which will cause integration tests to be skipped. + """ + return os.environ.get("BASE_CHAIN_WALLET_KEY") + + +@pytest.fixture(scope="session") +def production_api_url(): + """Get production API URL.""" + return "https://blockrun.ai/api" diff --git a/tests/integration/test_production_api.py b/tests/integration/test_production_api.py index c65f877..1ca60c5 100644 --- a/tests/integration/test_production_api.py +++ b/tests/integration/test_production_api.py @@ -1,247 +1,316 @@ -"""Integration tests for BlockRun LLM SDK against production API. - -Requirements: -- BLOCKRUN_WALLET_KEY environment variable with funded Base wallet -- Minimum $1 USDC on Base chain -- Estimated cost per test run: ~$0.05 - -Run with: pytest tests/integration -Skip if no wallet: Tests will be skipped if BLOCKRUN_WALLET_KEY not set -""" - -import os -import pytest -import time -from blockrun_llm import LLMClient, AsyncLLMClient - -WALLET_KEY = os.environ.get("BLOCKRUN_WALLET_KEY") -PRODUCTION_API = "https://api.blockrun.ai" - -# Skip all tests if no wallet key configured -pytestmark = pytest.mark.skipif( - not WALLET_KEY, reason="BLOCKRUN_WALLET_KEY environment variable not set" -) - - -class TestProductionAPISync: - """Integration tests for synchronous LLMClient against production API.""" - - @pytest.fixture(scope="class") - def client(self): - """Create LLMClient instance for testing.""" - if not WALLET_KEY: - pytest.skip("BLOCKRUN_WALLET_KEY not set") - - client = LLMClient(private_key=WALLET_KEY, api_url=PRODUCTION_API) - - print("\n🧪 Running sync integration tests against production API") - print(f" Wallet: {client.get_wallet_address()}") - print(f" API: {PRODUCTION_API}") - print(f" Estimated cost: ~$0.05\n") - - return client - - def test_list_models(self, client): - """Should list available models from production API.""" - models = client.list_models() - - assert models is not None - assert isinstance(models, list) - assert len(models) > 0 - - # Verify model structure - first_model = models[0] - assert "id" in first_model - assert "provider" in first_model - assert "inputPrice" in first_model - assert "outputPrice" in first_model - - print(f" ✓ Found {len(models)} models") - - # Respect rate limits - time.sleep(2) - - def test_simple_chat_request(self, client): - """Should complete a simple chat request.""" - # Use cheapest model for testing - response = client.chat( - "gemini-2.0-flash-exp", - [{"role": "user", "content": "Say 'test passed' and nothing else"}], - ) - - assert response is not None - assert isinstance(response, str) - assert "test passed" in response.lower() - - print(f" ✓ Chat response: {response[:50]}...") - - time.sleep(2) - - def test_chat_completion_with_usage_stats(self, client): - """Should return chat completion with usage stats.""" - completion = client.chat_completion( - "gemini-2.0-flash-exp", - [{"role": "user", "content": "Count to 5"}], - max_tokens=50, - ) - - assert completion is not None - assert "choices" in completion - assert len(completion["choices"]) > 0 - assert "message" in completion["choices"][0] - assert "content" in completion["choices"][0]["message"] - assert completion["choices"][0]["message"]["content"] - - # Verify usage stats - assert "usage" in completion - assert completion["usage"]["prompt_tokens"] > 0 - assert completion["usage"]["completion_tokens"] > 0 - assert completion["usage"]["total_tokens"] > 0 - - print(f" ✓ Completion with usage: {completion['usage']}") - - time.sleep(2) - - def test_payment_flow_end_to_end(self, client): - """Should handle 402 payment flow end-to-end. - - This test verifies the full x402 payment protocol: - 1. Request to API - 2. Receive 402 with payment required - 3. Create payment payload with EIP-712 signature - 4. Retry with payment receipt - 5. Receive successful response - """ - response = client.chat( - "gemini-2.0-flash-exp", [{"role": "user", "content": "What is 2+2?"}] - ) - - # If we got a response, the payment flow succeeded - assert response is not None - assert isinstance(response, str) - assert response - - print(f" ✓ Payment flow successful, response received") - - time.sleep(2) - - - -class TestProductionAPIAsync: - """Integration tests for asynchronous AsyncLLMClient against production API.""" - - @pytest.fixture(scope="class") - async def async_client(self): - """Create AsyncLLMClient instance for testing.""" - if not WALLET_KEY: - pytest.skip("BLOCKRUN_WALLET_KEY not set") - - client = AsyncLLMClient(private_key=WALLET_KEY, api_url=PRODUCTION_API) - - print("\n🧪 Running async integration tests against production API") - print(f" Wallet: {client.get_wallet_address()}") - print(f" API: {PRODUCTION_API}") - print(f" Estimated cost: ~$0.05\n") - - return client - - @pytest.mark.asyncio - async def test_async_list_models(self, async_client): - """Should list available models asynchronously.""" - models = await async_client.list_models() - - assert models is not None - assert isinstance(models, list) - assert len(models) > 0 - - print(f" ✓ Async: Found {len(models)} models") - - await asyncio.sleep(2) - - @pytest.mark.asyncio - async def test_async_simple_chat(self, async_client): - """Should complete a simple chat request asynchronously.""" - response = await async_client.chat( - "gemini-2.0-flash-exp", - [{"role": "user", "content": "Say 'async test passed' and nothing else"}], - ) - - assert response is not None - assert isinstance(response, str) - assert "test passed" in response.lower() - - print(f" ✓ Async chat response: {response[:50]}...") - - await asyncio.sleep(2) - - @pytest.mark.asyncio - async def test_async_chat_completion(self, async_client): - """Should return chat completion with usage stats asynchronously.""" - completion = await async_client.chat_completion( - "gemini-2.0-flash-exp", - [{"role": "user", "content": "Count to 5"}], - max_tokens=50, - ) - - assert completion is not None - assert "choices" in completion - assert len(completion["choices"]) > 0 - assert "usage" in completion - assert completion["usage"]["total_tokens"] > 0 - - print(f" ✓ Async completion with usage: {completion['usage']}") - - await asyncio.sleep(2) - - - -class TestProductionAPIErrorHandling: - """Integration tests for error handling against production API.""" - - @pytest.fixture(scope="class") - def client(self): - """Create LLMClient instance for testing.""" - if not WALLET_KEY: - pytest.skip("BLOCKRUN_WALLET_KEY not set") - - return LLMClient(private_key=WALLET_KEY, api_url=PRODUCTION_API) - - def test_invalid_model_error(self, client): - """Should handle invalid model error gracefully.""" - from blockrun_llm import APIError - - with pytest.raises(APIError): - client.chat( - "invalid-model-that-does-not-exist", - [{"role": "user", "content": "test"}], - ) - - print(f" ✓ Invalid model error handled correctly") - - time.sleep(2) - - def test_error_response_sanitization(self, client): - """Should sanitize error responses.""" - from blockrun_llm import APIError - - try: - client.chat( - "invalid-model", [{"role": "user", "content": "test"}] - ) - pytest.fail("Should have raised APIError") - except APIError as e: - # Error should be sanitized (no internal stack traces, API keys, etc.) - assert e.message is not None - assert "/var/" not in str(e.message) - assert "internal" not in str(e.message).lower() or "internal" in str( - e.message - ).lower() # Allow "internal" in error message but not internal paths - assert "stack" not in str(e.message).lower() - - print(f" ✓ Error response properly sanitized") - - time.sleep(2) - - -# Import asyncio for async tests -import asyncio +"""Integration tests for BlockRun LLM SDK against production API. + +Requirements: +- BASE_CHAIN_WALLET_KEY environment variable with funded Base wallet +- Minimum $1 USDC on Base chain +- Estimated cost per test run: ~$0.05 + +Run with: pytest tests/integration +Skip if no wallet: Tests will be skipped if BASE_CHAIN_WALLET_KEY not set +""" + +import asyncio +import os +import time + +import pytest + +from blockrun_llm import AsyncLLMClient, LLMClient + +WALLET_KEY = os.environ.get("BASE_CHAIN_WALLET_KEY") +PRODUCTION_API = "https://blockrun.ai/api" + +# Skip all tests if no wallet key configured +pytestmark = pytest.mark.skipif( + not WALLET_KEY, reason="BASE_CHAIN_WALLET_KEY environment variable not set" +) + + +class TestProductionAPISync: + """Integration tests for synchronous LLMClient against production API.""" + + @pytest.fixture(scope="class") + def client(self): + """Create LLMClient instance for testing.""" + if not WALLET_KEY: + pytest.skip("BASE_CHAIN_WALLET_KEY not set") + + client = LLMClient(private_key=WALLET_KEY, api_url=PRODUCTION_API) + + print("\n🧪 Running sync integration tests against production API") + print(f" Wallet: {client.get_wallet_address()}") + print(f" API: {PRODUCTION_API}") + print(" Estimated cost: ~$0.05\n") + + return client + + def test_list_models(self, client): + """Should list available models from production API.""" + models = client.list_models() + + assert models is not None + assert isinstance(models, list) + assert len(models) > 0 + + # Verify model structure + first_model = models[0] + assert "id" in first_model + assert "provider" in first_model + assert "inputPrice" in first_model + assert "outputPrice" in first_model + + print(f" ✓ Found {len(models)} models") + + # Respect rate limits + time.sleep(2) + + def test_simple_chat_request(self, client): + """Should complete a simple chat request.""" + # Use cheapest model for testing + response = client.chat( + "google/gemini-2.5-flash-lite", + [{"role": "user", "content": "Say 'test passed' and nothing else"}], + ) + + assert response is not None + assert isinstance(response, str) + assert "test passed" in response.lower() + + print(f" ✓ Chat response: {response[:50]}...") + + time.sleep(2) + + def test_chat_completion_with_usage_stats(self, client): + """Should return chat completion with usage stats.""" + completion = client.chat_completion( + "google/gemini-2.5-flash-lite", + [{"role": "user", "content": "Count to 5"}], + max_tokens=50, + ) + + assert completion is not None + assert "choices" in completion + assert len(completion["choices"]) > 0 + assert "message" in completion["choices"][0] + assert "content" in completion["choices"][0]["message"] + assert completion["choices"][0]["message"]["content"] + + # Verify usage stats + assert "usage" in completion + assert completion["usage"]["prompt_tokens"] > 0 + assert completion["usage"]["completion_tokens"] > 0 + assert completion["usage"]["total_tokens"] > 0 + + print(f" ✓ Completion with usage: {completion['usage']}") + + time.sleep(2) + + def test_payment_flow_end_to_end(self, client): + """Should handle 402 payment flow end-to-end. + + This test verifies the full x402 payment protocol: + 1. Request to API + 2. Receive 402 with payment required + 3. Create payment payload with EIP-712 signature + 4. Retry with payment receipt + 5. Receive successful response + """ + response = client.chat( + "google/gemini-2.5-flash-lite", [{"role": "user", "content": "What is 2+2?"}] + ) + + # If we got a response, the payment flow succeeded + assert response is not None + assert isinstance(response, str) + assert response + + print(" ✓ Payment flow successful, response received") + + time.sleep(2) + + +class TestProductionAPIAsync: + """Integration tests for asynchronous AsyncLLMClient against production API.""" + + @pytest.fixture(scope="class") + async def async_client(self): + """Create AsyncLLMClient instance for testing.""" + if not WALLET_KEY: + pytest.skip("BASE_CHAIN_WALLET_KEY not set") + + client = AsyncLLMClient(private_key=WALLET_KEY, api_url=PRODUCTION_API) + + print("\n🧪 Running async integration tests against production API") + print(f" Wallet: {client.get_wallet_address()}") + print(f" API: {PRODUCTION_API}") + print(" Estimated cost: ~$0.05\n") + + return client + + @pytest.mark.asyncio + async def test_async_list_models(self, async_client): + """Should list available models asynchronously.""" + models = await async_client.list_models() + + assert models is not None + assert isinstance(models, list) + assert len(models) > 0 + + print(f" ✓ Async: Found {len(models)} models") + + await asyncio.sleep(2) + + @pytest.mark.asyncio + async def test_async_simple_chat(self, async_client): + """Should complete a simple chat request asynchronously.""" + response = await async_client.chat( + "google/gemini-2.5-flash-lite", + [{"role": "user", "content": "Say 'async test passed' and nothing else"}], + ) + + assert response is not None + assert isinstance(response, str) + assert "test passed" in response.lower() + + print(f" ✓ Async chat response: {response[:50]}...") + + await asyncio.sleep(2) + + @pytest.mark.asyncio + async def test_async_chat_completion(self, async_client): + """Should return chat completion with usage stats asynchronously.""" + completion = await async_client.chat_completion( + "google/gemini-2.5-flash-lite", + [{"role": "user", "content": "Count to 5"}], + max_tokens=50, + ) + + assert completion is not None + assert "choices" in completion + assert len(completion["choices"]) > 0 + assert "usage" in completion + assert completion["usage"]["total_tokens"] > 0 + + print(f" ✓ Async completion with usage: {completion['usage']}") + + await asyncio.sleep(2) + + +class TestProductionAPIErrorHandling: + """Integration tests for error handling against production API.""" + + @pytest.fixture(scope="class") + def client(self): + """Create LLMClient instance for testing.""" + if not WALLET_KEY: + pytest.skip("BASE_CHAIN_WALLET_KEY not set") + + return LLMClient(private_key=WALLET_KEY, api_url=PRODUCTION_API) + + def test_invalid_model_error(self, client): + """Should handle invalid model error gracefully.""" + from blockrun_llm import APIError + + with pytest.raises(APIError): + client.chat( + "invalid-model-that-does-not-exist", + [{"role": "user", "content": "test"}], + ) + + print(" ✓ Invalid model error handled correctly") + + time.sleep(2) + + def test_error_response_sanitization(self, client): + """Should sanitize error responses.""" + from blockrun_llm import APIError + + try: + client.chat("invalid-model", [{"role": "user", "content": "test"}]) + pytest.fail("Should have raised APIError") + except APIError as e: + # Error should be sanitized (no internal stack traces, API keys, etc.) + assert e.message is not None + assert "/var/" not in str(e.message) + assert ( + "internal" not in str(e.message).lower() or "internal" in str(e.message).lower() + ) # Allow "internal" in error message but not internal paths + assert "stack" not in str(e.message).lower() + + print(" ✓ Error response properly sanitized") + + time.sleep(2) + + +# ============================================================================= +# Solana + Exa Integration Tests +# ============================================================================= + +SOLANA_WALLET_KEY = os.environ.get("SOLANA_WALLET_KEY") +SOLANA_API = "https://sol.blockrun.ai/api" + + +class TestSolanaExa: + """Integration tests for Exa web search via SolanaLLMClient.""" + + @pytest.fixture(scope="class") + def client(self): + if not SOLANA_WALLET_KEY: + pytest.skip("SOLANA_WALLET_KEY not set") + from blockrun_llm import SolanaLLMClient + + c = SolanaLLMClient(private_key=SOLANA_WALLET_KEY, api_url=SOLANA_API) + print("\n🧪 Running Solana/Exa integration tests against sol.blockrun.ai") + print(f" Wallet: {c.get_wallet_address()}") + print(" Estimated cost: ~$0.04\n") + return c + + def test_exa_search(self, client): + """exa_search returns results with title/url fields.""" + result = client.exa_search("latest AI safety research", numResults=3) + assert "results" in result, f"Expected 'results' key, got: {list(result.keys())}" + assert len(result["results"]) > 0 + first = result["results"][0] + assert "url" in first or "title" in first + cost = client.get_spending()["total_usd"] + assert 0.009 <= cost <= 0.011, f"Expected ~$0.01 cost, got {cost}" + print(f" ✓ exa_search: {len(result['results'])} results, cost=${cost:.4f}") + time.sleep(1) + + def test_exa_find_similar(self, client): + """exa_find_similar returns semantically similar pages.""" + result = client.exa_find_similar("https://openai.com/research/gpt-4", numResults=3) + assert "results" in result + assert len(result["results"]) > 0 + print(f" ✓ exa_find_similar: {len(result['results'])} results") + time.sleep(1) + + def test_exa_contents(self, client): + """exa_contents extracts text from a URL, priced per URL.""" + result = client.exa_contents(["https://www.anthropic.com/research"]) + assert result is not None + assert isinstance(result, dict) + print(" ✓ exa_contents: response received") + time.sleep(1) + + def test_exa_answer(self, client): + """exa_answer returns an AI-generated answer from live web.""" + result = client.exa_answer("What is Anthropic Claude?") + assert result is not None + assert isinstance(result, dict) + print(" ✓ exa_answer: response received") + time.sleep(1) + + def test_exa_generic_proxy(self, client): + """exa() generic proxy works for any endpoint.""" + result = client.exa("search", {"query": "blockrun.ai", "numResults": 2}) + assert "results" in result + print(f" ✓ exa() generic: {len(result['results'])} results") + time.sleep(1) + + def test_exa_spending_tracked(self, client): + """Session spending is tracked across Exa calls.""" + spending = client.get_spending() + assert spending["total_usd"] > 0 + assert spending["calls"] >= 3 + print(f" ✓ Spending tracked: ${spending['total_usd']:.4f} over {spending['calls']} calls") diff --git a/tests/unit/__init__.py b/tests/unit/__init__.py index ceb7f13..0360524 100644 --- a/tests/unit/__init__.py +++ b/tests/unit/__init__.py @@ -1 +1 @@ -"""Unit tests for BlockRun LLM SDK.""" +"""Unit tests for BlockRun LLM SDK.""" diff --git a/tests/unit/router_core_decisions.snapshot.json b/tests/unit/router_core_decisions.snapshot.json new file mode 100644 index 0000000..61fb442 --- /dev/null +++ b/tests/unit/router_core_decisions.snapshot.json @@ -0,0 +1,3960 @@ +[ + { + "prompt": 0, + "profile": "auto", + "model": "google/gemini-2.5-flash", + "tier": "SIMPLE", + "confidence": 0.7685247834990178, + "method": "portfolio", + "reasoning": "score=-0.10 | short (15 tokens), simple (what is, capital of) | v3 task=chat agentRisk=standard deepWebResearch=false terminalCode=false terminalSafety=false candidates=9", + "costEstimate": 0.0001459, + "baselineCost": 0.006475, + "savings": 0.9774671814671815, + "agenticScore": 0, + "candidates": [ + "google/gemini-2.5-flash", + "google/gemini-3-flash-preview", + "google/gemini-3.5-flash-lite", + "deepseek/deepseek-chat", + "google/gemini-3.1-flash-lite", + "openai/gpt-5.6-luna", + "openai/gpt-5.4-nano", + "google/gemini-2.5-flash-lite", + "nvidia/nemotron-3.5-lightning" + ], + "candidateScores": [ + { + "model": "google/gemini-2.5-flash", + "score": 0.7870260539502253, + "quality": 0.86, + "cost": 0.5, + "speed": 0.18322934214607486, + "reliability": 1 + } + ], + "taskType": "chat", + "routerVersion": "v3-portfolio" + }, + { + "prompt": 1, + "profile": "agentic", + "model": "anthropic/claude-sonnet-5", + "tier": "SIMPLE", + "confidence": 0.7231218051243898, + "method": "portfolio", + "reasoning": "score=-0.08 | short (1 tokens) | agentic (tools) | v3 task=tool_agent agentRisk=standard deepWebResearch=false terminalCode=false terminalSafety=false candidates=13", + "costEstimate": 0.0032001, + "baselineCost": 0.032005, + "savings": 0.9000124980471801, + "agenticScore": 0, + "candidates": [ + "anthropic/claude-sonnet-5", + "openai/gpt-4o-mini", + "openai/gpt-5.6-luna", + "zai/glm-5.3-flash", + "anthropic/claude-haiku-4.5", + "google/gemini-2.5-flash", + "anthropic/claude-opus-5", + "openai/gpt-5-mini", + "openai/gpt-4.1", + "google/gemini-3.5-flash", + "openai/gpt-5.3-codex", + "moonshot/kimi-k3", + "deepseek/deepseek-v4-pro" + ], + "candidateScores": [ + { + "model": "anthropic/claude-sonnet-5", + "score": 0.8469181450377915, + "quality": 1, + "cost": 0.5, + "speed": 0.09883064339702209, + "reliability": 1 + } + ], + "taskType": "tool_agent", + "routerVersion": "v3-portfolio" + }, + { + "prompt": 2, + "profile": "auto", + "model": "google/gemini-2.5-flash", + "tier": "SIMPLE", + "confidence": 0.7231218051243898, + "method": "portfolio", + "reasoning": "score=-0.08 | short (16 tokens) | v3 task=vision agentRisk=standard deepWebResearch=false terminalCode=false terminalSafety=false candidates=3", + "costEstimate": 0.0009680000000000001, + "baselineCost": 0.057679999999999995, + "savings": 0.9832177531206658, + "agenticScore": 0, + "candidates": [ + "google/gemini-2.5-flash", + "google/gemini-3-flash-preview", + "openai/gpt-5.6-luna" + ], + "candidateScores": [ + { + "model": "google/gemini-2.5-flash", + "score": 0.7024260539502254, + "quality": 0.68, + "cost": 0.5, + "speed": 0.18322934214607486, + "reliability": 1 + } + ], + "taskType": "vision", + "routerVersion": "v3-portfolio" + }, + { + "prompt": 3, + "profile": "auto", + "model": "google/gemini-3.5-flash", + "tier": "MEDIUM", + "confidence": 0.5, + "method": "portfolio", + "reasoning": "score=0.02 | short (31 tokens), code (function), creative (write a) | ambiguous -> default: MEDIUM | v3 task=code_edit agentRisk=standard deepWebResearch=false terminalCode=false terminalSafety=false candidates=11", + "costEstimate": 0.0224092, + "baselineCost": 0.083355, + "savings": 0.7311594985303821, + "agenticScore": 0, + "candidates": [ + "google/gemini-3.5-flash", + "google/gemini-3.6-flash", + "zai/glm-5.3-flash", + "openai/gpt-5.6-terra", + "google/gemini-3-flash-preview", + "deepseek/deepseek-chat", + "google/gemini-2.5-flash", + "minimax/minimax-m3", + "google/gemini-3.1-flash-lite", + "openai/gpt-5.6-luna", + "google/gemini-2.5-flash-lite" + ], + "candidateScores": [ + { + "model": "google/gemini-3.5-flash", + "score": 0.7030477432793295, + "quality": 0.68, + "cost": 0.5, + "speed": 0.19211061827613424, + "reliability": 1 + } + ], + "taskType": "code_edit", + "routerVersion": "v3-portfolio" + }, + { + "prompt": 4, + "profile": "auto", + "model": "deepseek/deepseek-v4-pro", + "tier": "REASONING", + "confidence": 0.973403006423134, + "method": "portfolio", + "reasoning": "score=0.10 | short (16 tokens), reasoning (prove, step by step) | v3 task=reasoning agentRisk=standard deepWebResearch=false terminalCode=false terminalSafety=false candidates=7", + "costEstimate": 0.0012016, + "baselineCost": 0.00648, + "savings": 0.8145679012345679, + "agenticScore": 0, + "candidates": [ + "deepseek/deepseek-v4-pro", + "google/gemini-3.5-flash", + "deepseek/deepseek-reasoner", + "xai/grok-4.3", + "qwen/qwen3.7-plus", + "openai/o4-mini", + "openai/o3" + ], + "candidateScores": [ + { + "model": "deepseek/deepseek-v4-pro", + "score": 0.9113686326916596, + "quality": 0.95, + "cost": 1, + "speed": 0.06955189559513505, + "reliability": 1 + }, + { + "model": "google/gemini-3.5-flash", + "score": 0.6758477432793295, + "quality": 0.92, + "cost": 0, + "speed": 0.19211061827613424, + "reliability": 1 + } + ], + "taskType": "reasoning", + "routerVersion": "v3-portfolio" + }, + { + "prompt": 5, + "profile": "auto", + "model": "google/gemini-3.5-flash", + "tier": "MEDIUM", + "confidence": 0.5, + "method": "portfolio", + "reasoning": "score=0.07 | short (30 tokens), code (class, return, ```) | ambiguous -> default: MEDIUM | v3 task=code_edit agentRisk=standard deepWebResearch=false terminalCode=false terminalSafety=false candidates=11", + "costEstimate": 0.008684000000000002, + "baselineCost": 0.03215, + "savings": 0.7298911353032658, + "agenticScore": 0, + "candidates": [ + "google/gemini-3.5-flash", + "google/gemini-3.6-flash", + "zai/glm-5.3-flash", + "openai/gpt-5.6-terra", + "google/gemini-3-flash-preview", + "deepseek/deepseek-chat", + "google/gemini-2.5-flash", + "minimax/minimax-m3", + "google/gemini-3.1-flash-lite", + "openai/gpt-5.6-luna", + "google/gemini-2.5-flash-lite" + ], + "candidateScores": [ + { + "model": "google/gemini-3.5-flash", + "score": 0.7030477432793295, + "quality": 0.68, + "cost": 0.5, + "speed": 0.19211061827613424, + "reliability": 1 + } + ], + "taskType": "code_edit", + "routerVersion": "v3-portfolio" + }, + { + "prompt": 6, + "profile": "agentic", + "model": "anthropic/claude-opus-4.8", + "tier": "SIMPLE", + "confidence": 0.7231218051243898, + "method": "portfolio", + "reasoning": "score=-0.08 | short (20 tokens) | agentic (tools) | v3 task=tool_agent_parallel agentRisk=high deepWebResearch=false terminalCode=false terminalSafety=false candidates=13", + "costEstimate": 0.020291200000000002, + "baselineCost": 0.0577, + "savings": 0.6483327556325823, + "agenticScore": 0, + "candidates": [ + "anthropic/claude-opus-4.8", + "openai/gpt-4o-mini", + "openai/gpt-5.6-luna", + "zai/glm-5.3-flash", + "anthropic/claude-haiku-4.5", + "google/gemini-2.5-flash", + "anthropic/claude-opus-5", + "anthropic/claude-sonnet-5", + "openai/gpt-5-mini", + "openai/gpt-4.1", + "xai/grok-4.5", + "google/gemini-3.5-flash", + "deepseek/deepseek-v4-pro" + ], + "candidateScores": [ + { + "model": "anthropic/claude-opus-4.8", + "score": 0.8468563440295361, + "quality": 1, + "cost": 0.5, + "speed": 0.09794777185051722, + "reliability": 1 + } + ], + "taskType": "tool_agent_parallel", + "routerVersion": "v3-portfolio" + }, + { + "prompt": 7, + "profile": "auto", + "model": "google/gemini-2.5-flash", + "tier": "SIMPLE", + "confidence": 0.7231218051243898, + "method": "portfolio", + "reasoning": "score=-0.08 | short (13 tokens) | v3 task=vision agentRisk=standard deepWebResearch=false terminalCode=false terminalSafety=false candidates=3", + "costEstimate": 0.0013689000000000002, + "baselineCost": 0.08326499999999999, + "savings": 0.983559718969555, + "agenticScore": 0, + "candidates": [ + "google/gemini-2.5-flash", + "google/gemini-3-flash-preview", + "openai/gpt-5.6-luna" + ], + "candidateScores": [ + { + "model": "google/gemini-2.5-flash", + "score": 0.7024260539502254, + "quality": 0.68, + "cost": 0.5, + "speed": 0.18322934214607486, + "reliability": 1 + } + ], + "taskType": "vision", + "routerVersion": "v3-portfolio" + }, + { + "prompt": 8, + "profile": "auto", + "model": "moonshot/kimi-k3", + "tier": "MEDIUM", + "confidence": 0.5, + "method": "portfolio", + "reasoning": "score=-0.07 | short (5 tokens), constraints (不超过) | ambiguous -> default: MEDIUM | v3 task=extraction agentRisk=standard deepWebResearch=false terminalCode=false terminalSafety=false candidates=13", + "costEstimate": 0.0017297000000000002, + "baselineCost": 0.006424999999999999, + "savings": 0.7307859922178989, + "agenticScore": 0, + "candidates": [ + "moonshot/kimi-k3", + "google/gemini-3.5-flash", + "google/gemini-3.6-flash", + "zai/glm-5.3-flash", + "openai/gpt-5.6-terra", + "google/gemini-3-flash-preview", + "deepseek/deepseek-chat", + "google/gemini-2.5-flash", + "minimax/minimax-m3", + "google/gemini-3.1-flash-lite", + "openai/gpt-5.6-luna", + "google/gemini-2.5-flash-lite", + "anthropic/claude-sonnet-5" + ], + "candidateScores": [ + { + "model": "moonshot/kimi-k3", + "score": 0.8419118013428357, + "quality": 1, + "cost": 0.5, + "speed": 0.02731144775479715, + "reliability": 1 + } + ], + "taskType": "extraction", + "routerVersion": "v3-portfolio" + }, + { + "prompt": 9, + "profile": "auto", + "model": "google/gemini-2.5-flash", + "tier": "MEDIUM", + "confidence": 0.5, + "method": "portfolio", + "reasoning": "score=-0.07 | short (16 tokens), imperative (设计), references (代码) | ambiguous -> default: MEDIUM | v3 task=chat agentRisk=standard deepWebResearch=false terminalCode=false terminalSafety=false candidates=11", + "costEstimate": 0.0005584000000000001, + "baselineCost": 0.03208, + "savings": 0.9825935162094763, + "agenticScore": 0, + "candidates": [ + "google/gemini-2.5-flash", + "google/gemini-3.5-flash", + "google/gemini-3.6-flash", + "zai/glm-5.3-flash", + "openai/gpt-5.6-terra", + "google/gemini-3-flash-preview", + "deepseek/deepseek-chat", + "minimax/minimax-m3", + "google/gemini-3.1-flash-lite", + "openai/gpt-5.6-luna", + "google/gemini-2.5-flash-lite" + ], + "candidateScores": [ + { + "model": "google/gemini-2.5-flash", + "score": 0.8270260539502253, + "quality": 0.86, + "cost": 1, + "speed": 0.18322934214607486, + "reliability": 1 + }, + { + "model": "google/gemini-3.5-flash", + "score": 0.6976477432793294, + "quality": 0.86, + "cost": 0, + "speed": 0.19211061827613424, + "reliability": 1 + } + ], + "taskType": "chat", + "routerVersion": "v3-portfolio" + }, + { + "prompt": 10, + "profile": "auto", + "model": "google/gemini-3.5-flash", + "tier": "MEDIUM", + "confidence": 0.5, + "method": "portfolio", + "reasoning": "score=-0.02 | short (23 tokens), code (def), simple (define), agentic-light (debug) | ambiguous -> default: MEDIUM | v3 task=debug agentRisk=standard deepWebResearch=false terminalCode=false terminalSafety=false candidates=11", + "costEstimate": 0.015519600000000001, + "baselineCost": 0.057714999999999995, + "savings": 0.7310993675820844, + "agenticScore": 0.2, + "candidates": [ + "google/gemini-3.5-flash", + "google/gemini-3.6-flash", + "zai/glm-5.3-flash", + "openai/gpt-5.6-terra", + "google/gemini-3-flash-preview", + "deepseek/deepseek-chat", + "google/gemini-2.5-flash", + "minimax/minimax-m3", + "google/gemini-3.1-flash-lite", + "openai/gpt-5.6-luna", + "google/gemini-2.5-flash-lite" + ], + "candidateScores": [ + { + "model": "google/gemini-3.5-flash", + "score": 0.7030477432793295, + "quality": 0.68, + "cost": 0.5, + "speed": 0.19211061827613424, + "reliability": 1 + } + ], + "taskType": "debug", + "routerVersion": "v3-portfolio" + }, + { + "prompt": 11, + "profile": "agentic", + "model": "anthropic/claude-sonnet-5", + "tier": "MEDIUM", + "confidence": 0.7373034537835593, + "method": "portfolio", + "reasoning": "score=0.09 | long (1217 tokens), references (following) | agentic (tools) | v3 task=tool_agent agentRisk=standard deepWebResearch=false terminalCode=false terminalSafety=false candidates=13", + "costEstimate": 0.0084417, + "baselineCost": 0.089285, + "savings": 0.905452203617629, + "agenticScore": 0, + "candidates": [ + "anthropic/claude-sonnet-5", + "openai/gpt-5-mini", + "google/gemini-3.5-flash", + "zai/glm-5.3-flash", + "openai/gpt-5.6-terra", + "openai/gpt-4o-mini", + "anthropic/claude-haiku-4.5", + "deepseek/deepseek-chat", + "moonshot/kimi-k3", + "anthropic/claude-opus-5", + "openai/gpt-4.1", + "openai/gpt-5.3-codex", + "deepseek/deepseek-v4-pro" + ], + "candidateScores": [ + { + "model": "anthropic/claude-sonnet-5", + "score": 0.8469181450377915, + "quality": 1, + "cost": 0.5, + "speed": 0.09883064339702209, + "reliability": 1 + } + ], + "taskType": "tool_agent", + "routerVersion": "v3-portfolio" + }, + { + "prompt": 12, + "profile": "auto", + "model": "google/gemini-3.5-flash", + "tier": "MEDIUM", + "confidence": 0.5, + "method": "portfolio", + "reasoning": "score=0.06 | short (32 tokens), code (let), multi-step, references (following) | ambiguous -> default: MEDIUM | v3 task=vision agentRisk=standard deepWebResearch=false terminalCode=false terminalSafety=false candidates=7", + "costEstimate": 0.0018304000000000003, + "baselineCost": 0.00656, + "savings": 0.7209756097560975, + "agenticScore": 0, + "candidates": [ + "google/gemini-3.5-flash", + "google/gemini-3.6-flash", + "zai/glm-5.3-flash", + "openai/gpt-5.6-terra", + "google/gemini-3-flash-preview", + "google/gemini-2.5-flash", + "openai/gpt-5.6-luna" + ], + "candidateScores": [ + { + "model": "google/gemini-3.5-flash", + "score": 0.7030477432793295, + "quality": 0.68, + "cost": 0.5, + "speed": 0.19211061827613424, + "reliability": 1 + } + ], + "taskType": "vision", + "routerVersion": "v3-portfolio" + }, + { + "prompt": 13, + "profile": "auto", + "model": "xai/grok-4.5", + "tier": "REASONING", + "confidence": 0.5, + "method": "portfolio", + "reasoning": "score=-0.02 | short (13 tokens), multi-step | ambiguous -> default: MEDIUM | v3 task=reasoning_math agentRisk=standard deepWebResearch=false terminalCode=false terminalSafety=false candidates=10", + "costEstimate": 0.0005406000000000001, + "baselineCost": 0.032065, + "savings": 0.9831404958677685, + "agenticScore": 0, + "candidates": [ + "xai/grok-4.5", + "anthropic/claude-sonnet-5", + "deepseek/deepseek-v4-pro", + "google/gemini-3.5-flash", + "moonshot/kimi-k3", + "deepseek/deepseek-reasoner", + "xai/grok-4.3", + "qwen/qwen3.7-plus", + "openai/o4-mini", + "openai/o3" + ], + "candidateScores": [ + { + "model": "xai/grok-4.5", + "score": 0.8761979445576786, + "quality": 0.93, + "cost": 1, + "speed": 0.05854206510969567, + "reliability": 1 + }, + { + "model": "anthropic/claude-sonnet-5", + "score": 0.7931612595363191, + "quality": 0.9, + "cost": 0.6707950805473757, + "speed": 0.09883064339702209, + "reliability": 1 + }, + { + "model": "deepseek/deepseek-v4-pro", + "score": 0.7683415237361771, + "quality": 0.9, + "cost": 0.3359605058028754, + "speed": 0.06955189559513505, + "reliability": 1 + }, + { + "model": "google/gemini-3.5-flash", + "score": 0.7509477432793293, + "quality": 1, + "cost": 0, + "speed": 0.19211061827613424, + "reliability": 1 + }, + { + "model": "moonshot/kimi-k3", + "score": 0.6551144689333432, + "quality": 0.9, + "cost": 0.001125931058375107, + "speed": 0.02731144775479715, + "reliability": 1 + } + ], + "taskType": "reasoning_math", + "routerVersion": "v3-portfolio" + }, + { + "prompt": 14, + "profile": "auto", + "model": "google/gemini-2.5-flash", + "tier": "MEDIUM", + "confidence": 0.5, + "method": "portfolio", + "reasoning": "score=-0.07 | short (25 tokens), format (json) | ambiguous -> default: MEDIUM | v3 task=extraction agentRisk=standard deepWebResearch=false terminalCode=false terminalSafety=false candidates=13", + "costEstimate": 0.0009941000000000001, + "baselineCost": 0.057725, + "savings": 0.9827786920744912, + "agenticScore": 0, + "candidates": [ + "google/gemini-2.5-flash", + "anthropic/claude-sonnet-5", + "google/gemini-3.5-flash", + "moonshot/kimi-k3", + "google/gemini-3.6-flash", + "zai/glm-5.3-flash", + "openai/gpt-5.6-terra", + "google/gemini-3-flash-preview", + "deepseek/deepseek-chat", + "minimax/minimax-m3", + "google/gemini-3.1-flash-lite", + "openai/gpt-5.6-luna", + "google/gemini-2.5-flash-lite" + ], + "candidateScores": [ + { + "model": "google/gemini-2.5-flash", + "score": 0.8791593872835586, + "quality": 0.9, + "cost": 1, + "speed": 0.18322934214607486, + "reliability": 1 + }, + { + "model": "anthropic/claude-sonnet-5", + "score": 0.7808574061523814, + "quality": 0.9, + "cost": 0.6718847839699437, + "speed": 0.09883064339702209, + "reliability": 1 + }, + { + "model": "google/gemini-3.5-flash", + "score": 0.7164477432793295, + "quality": 0.9, + "cost": 0, + "speed": 0.19211061827613424, + "reliability": 1 + }, + { + "model": "moonshot/kimi-k3", + "score": 0.6717952205744078, + "quality": 0.9, + "cost": 0.001204180916140718, + "speed": 0.02731144775479715, + "reliability": 1 + } + ], + "taskType": "extraction", + "routerVersion": "v3-portfolio" + }, + { + "prompt": 15, + "profile": "auto", + "model": "google/gemini-2.5-flash", + "tier": "MEDIUM", + "confidence": 0.5, + "method": "portfolio", + "reasoning": "score=-0.06 | short (29 tokens), imperative (build), agentic-light (pip) | ambiguous -> default: MEDIUM | v3 task=chat agentRisk=standard deepWebResearch=false terminalCode=false terminalSafety=false candidates=11", + "costEstimate": 0.0014153, + "baselineCost": 0.083345, + "savings": 0.983018777371168, + "agenticScore": 0.2, + "candidates": [ + "google/gemini-2.5-flash", + "google/gemini-3.5-flash", + "google/gemini-3.6-flash", + "zai/glm-5.3-flash", + "openai/gpt-5.6-terra", + "google/gemini-3-flash-preview", + "deepseek/deepseek-chat", + "minimax/minimax-m3", + "google/gemini-3.1-flash-lite", + "openai/gpt-5.6-luna", + "google/gemini-2.5-flash-lite" + ], + "candidateScores": [ + { + "model": "google/gemini-2.5-flash", + "score": 0.8270260539502253, + "quality": 0.86, + "cost": 1, + "speed": 0.18322934214607486, + "reliability": 1 + }, + { + "model": "google/gemini-3.5-flash", + "score": 0.6976477432793294, + "quality": 0.86, + "cost": 0, + "speed": 0.19211061827613424, + "reliability": 1 + } + ], + "taskType": "chat", + "routerVersion": "v3-portfolio" + }, + { + "prompt": 16, + "profile": "agentic", + "model": "anthropic/claude-sonnet-5", + "tier": "MEDIUM", + "confidence": 0.5, + "method": "portfolio", + "reasoning": "score=-0.06 | short (38 tokens), imperative (deploy), references (the api), agentic-light (check the, deploy) | ambiguous -> default: MEDIUM | agentic (tools) | v3 task=tool_agent agentRisk=standard deepWebResearch=false terminalCode=false terminalSafety=false candidates=13", + "costEstimate": 0.0006437999999999999, + "baselineCost": 0.0065899999999999995, + "savings": 0.9023065250379363, + "agenticScore": 0.2, + "candidates": [ + "anthropic/claude-sonnet-5", + "openai/gpt-5-mini", + "google/gemini-3.5-flash", + "zai/glm-5.3-flash", + "openai/gpt-5.6-terra", + "openai/gpt-4o-mini", + "anthropic/claude-haiku-4.5", + "deepseek/deepseek-chat", + "moonshot/kimi-k3", + "anthropic/claude-opus-5", + "openai/gpt-4.1", + "openai/gpt-5.3-codex", + "deepseek/deepseek-v4-pro" + ], + "candidateScores": [ + { + "model": "anthropic/claude-sonnet-5", + "score": 0.8469181450377915, + "quality": 1, + "cost": 0.5, + "speed": 0.09883064339702209, + "reliability": 1 + } + ], + "taskType": "tool_agent", + "routerVersion": "v3-portfolio" + }, + { + "prompt": 17, + "profile": "auto", + "model": "google/gemini-3.5-flash", + "tier": "MEDIUM", + "confidence": 0.5, + "method": "portfolio", + "reasoning": "score=-0.06 | short (9 tokens), creative (write a) | ambiguous -> default: MEDIUM | v3 task=vision agentRisk=standard deepWebResearch=false terminalCode=false terminalSafety=false candidates=7", + "costEstimate": 0.008608400000000002, + "baselineCost": 0.032045000000000004, + "savings": 0.7313652675924481, + "agenticScore": 0, + "candidates": [ + "google/gemini-3.5-flash", + "google/gemini-3.6-flash", + "zai/glm-5.3-flash", + "openai/gpt-5.6-terra", + "google/gemini-3-flash-preview", + "google/gemini-2.5-flash", + "openai/gpt-5.6-luna" + ], + "candidateScores": [ + { + "model": "google/gemini-3.5-flash", + "score": 0.7030477432793295, + "quality": 0.68, + "cost": 0.5, + "speed": 0.19211061827613424, + "reliability": 1 + } + ], + "taskType": "vision", + "routerVersion": "v3-portfolio" + }, + { + "prompt": 18, + "profile": "auto", + "model": "google/gemini-2.5-flash", + "tier": "MEDIUM", + "confidence": 0.7685247834990178, + "method": "portfolio", + "reasoning": "score=-0.10 | short (30 tokens), simple (translate) | upgraded to MEDIUM (structured output) | v3 task=extraction agentRisk=standard deepWebResearch=false terminalCode=false terminalSafety=false candidates=13", + "costEstimate": 0.0010086000000000001, + "baselineCost": 0.057749999999999996, + "savings": 0.9825350649350649, + "agenticScore": 0, + "candidates": [ + "google/gemini-2.5-flash", + "anthropic/claude-sonnet-5", + "google/gemini-3.5-flash", + "moonshot/kimi-k3", + "google/gemini-3.6-flash", + "zai/glm-5.3-flash", + "openai/gpt-5.6-terra", + "google/gemini-3-flash-preview", + "deepseek/deepseek-chat", + "minimax/minimax-m3", + "google/gemini-3.1-flash-lite", + "openai/gpt-5.6-luna", + "google/gemini-2.5-flash-lite" + ], + "candidateScores": [ + { + "model": "google/gemini-2.5-flash", + "score": 0.8872869559818712, + "quality": 0.9, + "cost": 1, + "speed": 0.13969081765691935, + "reliability": 1 + }, + { + "model": "anthropic/claude-sonnet-5", + "score": 0.7946345209396576, + "quality": 0.9, + "cost": 0.6729268997399596, + "speed": 0.1367178599097662, + "reliability": 1 + }, + { + "model": "google/gemini-3.5-flash", + "score": 0.7278627942097315, + "quality": 0.9, + "cost": 0, + "speed": 0.16575196139820988, + "reliability": 1 + }, + { + "model": "moonshot/kimi-k3", + "score": 0.6732711638661456, + "quality": 0.9, + "cost": 0.0014446691707599157, + "speed": 0.022296378324947415, + "reliability": 1 + } + ], + "taskType": "extraction", + "routerVersion": "v3-portfolio" + }, + { + "prompt": 19, + "profile": "auto", + "model": "xai/grok-4.5", + "tier": "REASONING", + "confidence": 0.5, + "method": "portfolio", + "reasoning": "score=-0.07 | short (33 tokens), constraints (budget, budget) | ambiguous -> default: MEDIUM | v3 task=reasoning_math agentRisk=standard deepWebResearch=false terminalCode=false terminalSafety=false candidates=10", + "costEstimate": 0.0014038, + "baselineCost": 0.083365, + "savings": 0.9831607988964194, + "agenticScore": 0, + "candidates": [ + "xai/grok-4.5", + "anthropic/claude-sonnet-5", + "deepseek/deepseek-v4-pro", + "google/gemini-3.5-flash", + "moonshot/kimi-k3", + "deepseek/deepseek-reasoner", + "xai/grok-4.3", + "qwen/qwen3.7-plus", + "openai/o4-mini", + "openai/o3" + ], + "candidateScores": [ + { + "model": "xai/grok-4.5", + "score": 0.8761979445576786, + "quality": 0.93, + "cost": 1, + "speed": 0.05854206510969567, + "reliability": 1 + }, + { + "model": "anthropic/claude-sonnet-5", + "score": 0.7931437096989948, + "quality": 0.9, + "cost": 0.6706975814511295, + "speed": 0.09883064339702209, + "reliability": 1 + }, + { + "model": "deepseek/deepseek-v4-pro", + "score": 0.7683303556578799, + "quality": 0.9, + "cost": 0.335898460923446, + "speed": 0.06955189559513505, + "reliability": 1 + }, + { + "model": "google/gemini-3.5-flash", + "score": 0.7509477432793293, + "quality": 1, + "cost": 0, + "speed": 0.19211061827613424, + "reliability": 1 + }, + { + "model": "moonshot/kimi-k3", + "score": 0.6551096826140731, + "quality": 0.9, + "cost": 0.001099340395762649, + "speed": 0.02731144775479715, + "reliability": 1 + } + ], + "taskType": "reasoning_math", + "routerVersion": "v3-portfolio" + }, + { + "prompt": 20, + "profile": "auto", + "model": "google/gemini-3.5-flash", + "tier": "MEDIUM", + "confidence": 0.5, + "method": "portfolio", + "reasoning": "score=0.01 | imperative (design) | ambiguous -> default: MEDIUM | v3 task=debug agentRisk=standard deepWebResearch=false terminalCode=false terminalSafety=false candidates=11", + "costEstimate": 0.0019060000000000004, + "baselineCost": 0.006664999999999999, + "savings": 0.7140285071267816, + "agenticScore": 0, + "candidates": [ + "google/gemini-3.5-flash", + "google/gemini-3.6-flash", + "zai/glm-5.3-flash", + "openai/gpt-5.6-terra", + "google/gemini-3-flash-preview", + "deepseek/deepseek-chat", + "google/gemini-2.5-flash", + "minimax/minimax-m3", + "google/gemini-3.1-flash-lite", + "openai/gpt-5.6-luna", + "google/gemini-2.5-flash-lite" + ], + "candidateScores": [ + { + "model": "google/gemini-3.5-flash", + "score": 0.7030477432793295, + "quality": 0.68, + "cost": 0.5, + "speed": 0.19211061827613424, + "reliability": 1 + } + ], + "taskType": "debug", + "routerVersion": "v3-portfolio" + }, + { + "prompt": 21, + "profile": "agentic", + "model": "anthropic/claude-sonnet-5", + "tier": "MEDIUM", + "confidence": 0.7231218051243898, + "method": "portfolio", + "reasoning": "score=0.08 | long (14423 tokens) | agentic (tools) | v3 task=tool_agent agentRisk=standard deepWebResearch=false terminalCode=false terminalSafety=false candidates=13", + "costEstimate": 0.0046423, + "baselineCost": 0.104115, + "savings": 0.9554118042549105, + "agenticScore": 0, + "candidates": [ + "anthropic/claude-sonnet-5", + "openai/gpt-5-mini", + "google/gemini-3.5-flash", + "zai/glm-5.3-flash", + "openai/gpt-5.6-terra", + "openai/gpt-4o-mini", + "anthropic/claude-haiku-4.5", + "deepseek/deepseek-chat", + "moonshot/kimi-k3", + "anthropic/claude-opus-5", + "openai/gpt-4.1", + "openai/gpt-5.3-codex", + "deepseek/deepseek-v4-pro" + ], + "candidateScores": [ + { + "model": "anthropic/claude-sonnet-5", + "score": 0.8469181450377915, + "quality": 1, + "cost": 0.5, + "speed": 0.09883064339702209, + "reliability": 1 + } + ], + "taskType": "tool_agent", + "routerVersion": "v3-portfolio" + }, + { + "prompt": 0, + "profile": "eco", + "model": "nvidia/nemotron-3.5-lightning", + "tier": "SIMPLE", + "confidence": 0.7685247834990178, + "method": "portfolio", + "reasoning": "score=-0.10 | short (15 tokens), simple (what is, capital of) | eco | v3 task=chat agentRisk=standard deepWebResearch=false terminalCode=false terminalSafety=false candidates=7", + "costEstimate": 0.0001459, + "baselineCost": 0.006475, + "savings": 0.9774671814671815, + "agenticScore": 0, + "candidates": [ + "nvidia/nemotron-3.5-lightning", + "nvidia/nemotron-3-nano-30b", + "google/gemini-2.5-flash-lite", + "zai/glm-5.3-flash", + "openai/gpt-5.6-luna", + "openai/gpt-5.4-nano", + "google/gemini-3.1-flash-lite" + ], + "candidateScores": [ + { + "model": "nvidia/nemotron-3.5-lightning", + "score": 0.6948000000000001, + "quality": 0.68, + "cost": 0.5, + "speed": 0.5, + "reliability": 1 + } + ], + "taskType": "chat", + "routerVersion": "v3-portfolio" + }, + { + "prompt": 1, + "profile": "eco", + "model": "anthropic/claude-sonnet-5", + "tier": "SIMPLE", + "confidence": 0.7231218051243898, + "method": "portfolio", + "reasoning": "score=-0.08 | short (1 tokens) | eco | v3 task=tool_agent agentRisk=standard deepWebResearch=false terminalCode=false terminalSafety=false candidates=14", + "costEstimate": 0.0032001, + "baselineCost": 0.032005, + "savings": 0.9000124980471801, + "agenticScore": 0, + "candidates": [ + "anthropic/claude-sonnet-5", + "openai/gpt-5-mini", + "openai/gpt-5.3-codex", + "deepseek/deepseek-v4-pro", + "google/gemini-3.5-flash", + "moonshot/kimi-k3", + "google/gemini-2.5-flash-lite", + "zai/glm-5.3-flash", + "openai/gpt-5.6-luna", + "openai/gpt-5.4-nano", + "google/gemini-3.1-flash-lite", + "anthropic/claude-opus-5", + "openai/gpt-4.1", + "openai/gpt-4o-mini" + ], + "candidateScores": [ + { + "model": "anthropic/claude-sonnet-5", + "score": 0.9098830643397023, + "quality": 1, + "cost": 1, + "speed": 0.09883064339702209, + "reliability": 1 + }, + { + "model": "openai/gpt-5-mini", + "score": 0.8516673859004991, + "quality": 0.84, + "cost": 0.9996096291476904, + "speed": 0.13376689739145697, + "reliability": 1 + }, + { + "model": "openai/gpt-5.3-codex", + "score": 0.8370981461564362, + "quality": 0.87, + "cost": 0.9997397527651268, + "speed": 0.039714153822005965, + "reliability": 0.79999 + }, + { + "model": "deepseek/deepseek-v4-pro", + "score": 0.6821734068659546, + "quality": 0.82, + "cost": 0.5000650618087183, + "speed": 0.06955189559513505, + "reliability": 1 + }, + { + "model": "google/gemini-3.5-flash", + "score": 0.5880110618276134, + "quality": 0.88, + "cost": 0, + "speed": 0.19211061827613424, + "reliability": 1 + }, + { + "model": "moonshot/kimi-k3", + "score": 0.552767579388362, + "quality": 0.85, + "cost": 0.00013012361743647283, + "speed": 0.02731144775479715, + "reliability": 1 + } + ], + "taskType": "tool_agent", + "routerVersion": "v3-portfolio" + }, + { + "prompt": 2, + "profile": "eco", + "model": "zai/glm-5.3-flash", + "tier": "SIMPLE", + "confidence": 0.7231218051243898, + "method": "portfolio", + "reasoning": "score=-0.08 | short (16 tokens) | eco | v3 task=vision agentRisk=standard deepWebResearch=false terminalCode=false terminalSafety=false candidates=2", + "costEstimate": 0.020288000000000004, + "baselineCost": 0.057679999999999995, + "savings": 0.648266296809986, + "agenticScore": 0, + "candidates": [ + "zai/glm-5.3-flash", + "openai/gpt-5.6-luna" + ], + "candidateScores": [ + { + "model": "zai/glm-5.3-flash", + "score": 0.6505854692782567, + "quality": 0.68, + "cost": 0.5, + "speed": 0.05785469278256664, + "reliability": 1 + } + ], + "taskType": "vision", + "routerVersion": "v3-portfolio" + }, + { + "prompt": 3, + "profile": "eco", + "model": "zai/glm-5.3-flash", + "tier": "MEDIUM", + "confidence": 0.5, + "method": "portfolio", + "reasoning": "score=0.02 | short (31 tokens), code (function), creative (write a) | ambiguous -> default: MEDIUM | eco | v3 task=code_edit agentRisk=standard deepWebResearch=false terminalCode=false terminalSafety=false candidates=7", + "costEstimate": 0.0293112, + "baselineCost": 0.083355, + "savings": 0.6483570271729351, + "agenticScore": 0, + "candidates": [ + "zai/glm-5.3-flash", + "deepseek/deepseek-chat", + "google/gemini-3.1-flash-lite", + "openai/gpt-5.6-luna", + "openai/gpt-5.4-nano", + "google/gemini-2.5-flash-lite", + "google/gemini-2.5-flash" + ], + "candidateScores": [ + { + "model": "zai/glm-5.3-flash", + "score": 0.6505854692782567, + "quality": 0.68, + "cost": 0.5, + "speed": 0.05785469278256664, + "reliability": 1 + } + ], + "taskType": "code_edit", + "routerVersion": "v3-portfolio" + }, + { + "prompt": 4, + "profile": "eco", + "model": "deepseek/deepseek-v4-pro", + "tier": "REASONING", + "confidence": 0.973403006423134, + "method": "portfolio", + "reasoning": "score=0.10 | short (16 tokens), reasoning (prove, step by step) | eco | v3 task=reasoning agentRisk=standard deepWebResearch=false terminalCode=false terminalSafety=false candidates=5", + "costEstimate": 0.0012016, + "baselineCost": 0.00648, + "savings": 0.8145679012345679, + "agenticScore": 0, + "candidates": [ + "deepseek/deepseek-v4-pro", + "deepseek/deepseek-reasoner", + "qwen/qwen3.7-plus", + "minimax/minimax-m3", + "zai/glm-5.3-flash" + ], + "candidateScores": [ + { + "model": "deepseek/deepseek-v4-pro", + "score": 0.7489551895595136, + "quality": 0.95, + "cost": 0.5, + "speed": 0.06955189559513505, + "reliability": 1 + } + ], + "taskType": "reasoning", + "routerVersion": "v3-portfolio" + }, + { + "prompt": 5, + "profile": "eco", + "model": "zai/glm-5.3-flash", + "tier": "MEDIUM", + "confidence": 0.5, + "method": "portfolio", + "reasoning": "score=0.07 | short (30 tokens), code (class, return, ```) | ambiguous -> default: MEDIUM | eco | v3 task=code_edit agentRisk=standard deepWebResearch=false terminalCode=false terminalSafety=false candidates=7", + "costEstimate": 0.011288000000000001, + "baselineCost": 0.03215, + "savings": 0.6488958009331259, + "agenticScore": 0, + "candidates": [ + "zai/glm-5.3-flash", + "deepseek/deepseek-chat", + "google/gemini-3.1-flash-lite", + "openai/gpt-5.6-luna", + "openai/gpt-5.4-nano", + "google/gemini-2.5-flash-lite", + "google/gemini-2.5-flash" + ], + "candidateScores": [ + { + "model": "zai/glm-5.3-flash", + "score": 0.6505854692782567, + "quality": 0.68, + "cost": 0.5, + "speed": 0.05785469278256664, + "reliability": 1 + } + ], + "taskType": "code_edit", + "routerVersion": "v3-portfolio" + }, + { + "prompt": 6, + "profile": "eco", + "model": "xai/grok-4.5", + "tier": "SIMPLE", + "confidence": 0.7231218051243898, + "method": "portfolio", + "reasoning": "score=-0.08 | short (20 tokens) | eco | v3 task=tool_agent_parallel agentRisk=high deepWebResearch=false terminalCode=false terminalSafety=false candidates=14", + "costEstimate": 0.0009656, + "baselineCost": 0.0577, + "savings": 0.9832651646447141, + "agenticScore": 0, + "candidates": [ + "xai/grok-4.5", + "anthropic/claude-sonnet-5", + "deepseek/deepseek-v4-pro", + "anthropic/claude-opus-4.8", + "google/gemini-3.5-flash", + "google/gemini-2.5-flash-lite", + "zai/glm-5.3-flash", + "openai/gpt-5.6-luna", + "openai/gpt-5.4-nano", + "google/gemini-3.1-flash-lite", + "anthropic/claude-opus-5", + "openai/gpt-5-mini", + "openai/gpt-4.1", + "openai/gpt-4o-mini" + ], + "candidateScores": [ + { + "model": "xai/grok-4.5", + "score": 0.8310542065109696, + "quality": 0.82, + "cost": 1, + "speed": 0.05854206510969567, + "reliability": 1 + }, + { + "model": "anthropic/claude-sonnet-5", + "score": 0.777790163731183, + "quality": 0.84, + "cost": 0.7518110692552884, + "speed": 0.09883064339702209, + "reliability": 1 + }, + { + "model": "deepseek/deepseek-v4-pro", + "score": 0.6677551895595135, + "quality": 0.78, + "cost": 0.5, + "speed": 0.06955189559513505, + "reliability": 1 + }, + { + "model": "anthropic/claude-opus-4.8", + "score": 0.6297947771850518, + "quality": 1, + "cost": 0, + "speed": 0.09794777185051722, + "reliability": 1 + }, + { + "model": "google/gemini-3.5-flash", + "score": 0.6215011226795404, + "quality": 0.8, + "cost": 0.24746450304259626, + "speed": 0.19211061827613424, + "reliability": 1 + } + ], + "taskType": "tool_agent_parallel", + "routerVersion": "v3-portfolio" + }, + { + "prompt": 7, + "profile": "eco", + "model": "zai/glm-5.3-flash", + "tier": "SIMPLE", + "confidence": 0.7231218051243898, + "method": "portfolio", + "reasoning": "score=-0.08 | short (13 tokens) | eco | v3 task=vision agentRisk=standard deepWebResearch=false terminalCode=false terminalSafety=false candidates=2", + "costEstimate": 0.0292968, + "baselineCost": 0.08326499999999999, + "savings": 0.6481498829039812, + "agenticScore": 0, + "candidates": [ + "zai/glm-5.3-flash", + "openai/gpt-5.6-luna" + ], + "candidateScores": [ + { + "model": "zai/glm-5.3-flash", + "score": 0.6505854692782567, + "quality": 0.68, + "cost": 0.5, + "speed": 0.05785469278256664, + "reliability": 1 + } + ], + "taskType": "vision", + "routerVersion": "v3-portfolio" + }, + { + "prompt": 8, + "profile": "eco", + "model": "google/gemini-2.5-flash", + "tier": "MEDIUM", + "confidence": 0.5, + "method": "portfolio", + "reasoning": "score=-0.07 | short (5 tokens), constraints (不超过) | ambiguous -> default: MEDIUM | eco | v3 task=extraction agentRisk=standard deepWebResearch=false terminalCode=false terminalSafety=false candidates=10", + "costEstimate": 0.0001169, + "baselineCost": 0.006424999999999999, + "savings": 0.9818054474708172, + "agenticScore": 0, + "candidates": [ + "google/gemini-2.5-flash", + "anthropic/claude-sonnet-5", + "moonshot/kimi-k3", + "google/gemini-3.5-flash", + "zai/glm-5.3-flash", + "deepseek/deepseek-chat", + "google/gemini-3.1-flash-lite", + "openai/gpt-5.6-luna", + "openai/gpt-5.4-nano", + "google/gemini-2.5-flash-lite" + ], + "candidateScores": [ + { + "model": "google/gemini-2.5-flash", + "score": 0.8751229342146075, + "quality": 0.88, + "cost": 1, + "speed": 0.18322934214607486, + "reliability": 1 + }, + { + "model": "anthropic/claude-sonnet-5", + "score": 0.7559771310352413, + "quality": 0.88, + "cost": 0.6760502381983543, + "speed": 0.09883064339702209, + "reliability": 1 + }, + { + "model": "moonshot/kimi-k3", + "score": 0.6166708011923991, + "quality": 1, + "cost": 0.002165439584235651, + "speed": 0.02731144775479715, + "reliability": 1 + }, + { + "model": "google/gemini-3.5-flash", + "score": 0.5826777284942801, + "quality": 0.88, + "cost": 0, + "speed": 0.19211061827613424, + "reliability": 1 + } + ], + "taskType": "extraction", + "routerVersion": "v3-portfolio" + }, + { + "prompt": 9, + "profile": "eco", + "model": "google/gemini-2.5-flash", + "tier": "MEDIUM", + "confidence": 0.5, + "method": "portfolio", + "reasoning": "score=-0.07 | short (16 tokens), imperative (设计), references (代码) | ambiguous -> default: MEDIUM | eco | v3 task=chat agentRisk=standard deepWebResearch=false terminalCode=false terminalSafety=false candidates=7", + "costEstimate": 0.0005584000000000001, + "baselineCost": 0.03208, + "savings": 0.9825935162094763, + "agenticScore": 0, + "candidates": [ + "google/gemini-2.5-flash", + "zai/glm-5.3-flash", + "deepseek/deepseek-chat", + "google/gemini-3.1-flash-lite", + "openai/gpt-5.6-luna", + "openai/gpt-5.4-nano", + "google/gemini-2.5-flash-lite" + ], + "candidateScores": [ + { + "model": "google/gemini-2.5-flash", + "score": 0.7279229342146076, + "quality": 0.86, + "cost": 0.5, + "speed": 0.18322934214607486, + "reliability": 1 + } + ], + "taskType": "chat", + "routerVersion": "v3-portfolio" + }, + { + "prompt": 10, + "profile": "eco", + "model": "zai/glm-5.3-flash", + "tier": "MEDIUM", + "confidence": 0.5, + "method": "portfolio", + "reasoning": "score=-0.02 | short (23 tokens), code (def), simple (define), agentic-light (debug) | ambiguous -> default: MEDIUM | eco | v3 task=debug agentRisk=standard deepWebResearch=false terminalCode=false terminalSafety=false candidates=7", + "costEstimate": 0.020293600000000002, + "baselineCost": 0.057714999999999995, + "savings": 0.6483825695226544, + "agenticScore": 0.2, + "candidates": [ + "zai/glm-5.3-flash", + "deepseek/deepseek-chat", + "google/gemini-3.1-flash-lite", + "openai/gpt-5.6-luna", + "openai/gpt-5.4-nano", + "google/gemini-2.5-flash-lite", + "google/gemini-2.5-flash" + ], + "candidateScores": [ + { + "model": "zai/glm-5.3-flash", + "score": 0.6505854692782567, + "quality": 0.68, + "cost": 0.5, + "speed": 0.05785469278256664, + "reliability": 1 + } + ], + "taskType": "debug", + "routerVersion": "v3-portfolio" + }, + { + "prompt": 11, + "profile": "eco", + "model": "anthropic/claude-sonnet-5", + "tier": "MEDIUM", + "confidence": 0.7373034537835593, + "method": "portfolio", + "reasoning": "score=0.09 | long (1217 tokens), references (following) | eco | v3 task=tool_agent agentRisk=standard deepWebResearch=false terminalCode=false terminalSafety=false candidates=16", + "costEstimate": 0.0084417, + "baselineCost": 0.089285, + "savings": 0.905452203617629, + "agenticScore": 0, + "candidates": [ + "anthropic/claude-sonnet-5", + "openai/gpt-5-mini", + "openai/gpt-5.3-codex", + "deepseek/deepseek-v4-pro", + "google/gemini-3.5-flash", + "moonshot/kimi-k3", + "zai/glm-5.3-flash", + "deepseek/deepseek-chat", + "google/gemini-3.1-flash-lite", + "openai/gpt-5.6-luna", + "openai/gpt-5.4-nano", + "google/gemini-2.5-flash-lite", + "google/gemini-2.5-flash", + "anthropic/claude-opus-5", + "openai/gpt-4.1", + "openai/gpt-4o-mini" + ], + "candidateScores": [ + { + "model": "anthropic/claude-sonnet-5", + "score": 0.9098830643397023, + "quality": 1, + "cost": 1, + "speed": 0.09883064339702209, + "reliability": 1 + }, + { + "model": "openai/gpt-5-mini", + "score": 0.8125382143236466, + "quality": 0.84, + "cost": 0.8598625878017887, + "speed": 0.13376689739145697, + "reliability": 1 + }, + { + "model": "openai/gpt-5.3-codex", + "score": 0.8110120317718679, + "quality": 0.87, + "cost": 0.9065750585345258, + "speed": 0.039714153822005965, + "reliability": 0.79999 + }, + { + "model": "deepseek/deepseek-v4-pro", + "score": 0.6886949354620967, + "quality": 0.82, + "cost": 0.5233562353663685, + "speed": 0.06955189559513505, + "reliability": 1 + }, + { + "model": "google/gemini-3.5-flash", + "score": 0.5880110618276134, + "quality": 0.88, + "cost": 0, + "speed": 0.19211061827613424, + "reliability": 1 + }, + { + "model": "moonshot/kimi-k3", + "score": 0.5658106365806462, + "quality": 0.85, + "cost": 0.04671247073273721, + "speed": 0.02731144775479715, + "reliability": 1 + } + ], + "taskType": "tool_agent", + "routerVersion": "v3-portfolio" + }, + { + "prompt": 12, + "profile": "eco", + "model": "zai/glm-5.3-flash", + "tier": "MEDIUM", + "confidence": 0.5, + "method": "portfolio", + "reasoning": "score=0.06 | short (32 tokens), code (let), multi-step, references (following) | ambiguous -> default: MEDIUM | eco | v3 task=vision agentRisk=standard deepWebResearch=false terminalCode=false terminalSafety=false candidates=3", + "costEstimate": 0.0022784000000000003, + "baselineCost": 0.00656, + "savings": 0.6526829268292683, + "agenticScore": 0, + "candidates": [ + "zai/glm-5.3-flash", + "openai/gpt-5.6-luna", + "google/gemini-2.5-flash" + ], + "candidateScores": [ + { + "model": "zai/glm-5.3-flash", + "score": 0.6505854692782567, + "quality": 0.68, + "cost": 0.5, + "speed": 0.05785469278256664, + "reliability": 1 + } + ], + "taskType": "vision", + "routerVersion": "v3-portfolio" + }, + { + "prompt": 13, + "profile": "eco", + "model": "xai/grok-4.5", + "tier": "REASONING", + "confidence": 0.5, + "method": "portfolio", + "reasoning": "score=-0.02 | short (13 tokens), multi-step | ambiguous -> default: MEDIUM | eco | v3 task=reasoning_math agentRisk=standard deepWebResearch=false terminalCode=false terminalSafety=false candidates=9", + "costEstimate": 0.0005406000000000001, + "baselineCost": 0.032065, + "savings": 0.9831404958677685, + "agenticScore": 0, + "candidates": [ + "xai/grok-4.5", + "anthropic/claude-sonnet-5", + "deepseek/deepseek-v4-pro", + "google/gemini-3.5-flash", + "moonshot/kimi-k3", + "deepseek/deepseek-reasoner", + "qwen/qwen3.7-plus", + "minimax/minimax-m3", + "zai/glm-5.3-flash" + ], + "candidateScores": [ + { + "model": "xai/grok-4.5", + "score": 0.8706542065109696, + "quality": 0.93, + "cost": 1, + "speed": 0.05854206510969567, + "reliability": 1 + }, + { + "model": "anthropic/claude-sonnet-5", + "score": 0.7667056868929675, + "quality": 0.9, + "cost": 0.6707950805473757, + "speed": 0.09883064339702209, + "reliability": 1 + }, + { + "model": "deepseek/deepseek-v4-pro", + "score": 0.6850241311843186, + "quality": 0.9, + "cost": 0.3359605058028754, + "speed": 0.06955189559513505, + "reliability": 1 + }, + { + "model": "google/gemini-3.5-flash", + "score": 0.6342110618276136, + "quality": 1, + "cost": 0, + "speed": 0.19211061827613424, + "reliability": 1 + }, + { + "model": "moonshot/kimi-k3", + "score": 0.5670464054718248, + "quality": 0.9, + "cost": 0.001125931058375107, + "speed": 0.02731144775479715, + "reliability": 1 + } + ], + "taskType": "reasoning_math", + "routerVersion": "v3-portfolio" + }, + { + "prompt": 14, + "profile": "eco", + "model": "google/gemini-2.5-flash", + "tier": "MEDIUM", + "confidence": 0.5, + "method": "portfolio", + "reasoning": "score=-0.07 | short (25 tokens), format (json) | ambiguous -> default: MEDIUM | eco | v3 task=extraction agentRisk=standard deepWebResearch=false terminalCode=false terminalSafety=false candidates=10", + "costEstimate": 0.0009941000000000001, + "baselineCost": 0.057725, + "savings": 0.9827786920744912, + "agenticScore": 0, + "candidates": [ + "google/gemini-2.5-flash", + "anthropic/claude-sonnet-5", + "google/gemini-3.5-flash", + "moonshot/kimi-k3", + "zai/glm-5.3-flash", + "deepseek/deepseek-chat", + "google/gemini-3.1-flash-lite", + "openai/gpt-5.6-luna", + "openai/gpt-5.4-nano", + "google/gemini-2.5-flash-lite" + ], + "candidateScores": [ + { + "model": "google/gemini-2.5-flash", + "score": 0.8823229342146076, + "quality": 0.9, + "cost": 1, + "speed": 0.18322934214607486, + "reliability": 1 + }, + { + "model": "anthropic/claude-sonnet-5", + "score": 0.7620108038512865, + "quality": 0.9, + "cost": 0.6718847839699437, + "speed": 0.09883064339702209, + "reliability": 1 + }, + { + "model": "google/gemini-3.5-flash", + "score": 0.5898777284942802, + "quality": 0.9, + "cost": 0, + "speed": 0.19211061827613424, + "reliability": 1 + }, + { + "model": "moonshot/kimi-k3", + "score": 0.5804016487653325, + "quality": 0.9, + "cost": 0.001204180916140718, + "speed": 0.02731144775479715, + "reliability": 1 + } + ], + "taskType": "extraction", + "routerVersion": "v3-portfolio" + }, + { + "prompt": 15, + "profile": "eco", + "model": "google/gemini-2.5-flash", + "tier": "MEDIUM", + "confidence": 0.5, + "method": "portfolio", + "reasoning": "score=-0.06 | short (29 tokens), imperative (build), agentic-light (pip) | ambiguous -> default: MEDIUM | eco | v3 task=chat agentRisk=standard deepWebResearch=false terminalCode=false terminalSafety=false candidates=7", + "costEstimate": 0.0014153, + "baselineCost": 0.083345, + "savings": 0.983018777371168, + "agenticScore": 0.2, + "candidates": [ + "google/gemini-2.5-flash", + "zai/glm-5.3-flash", + "deepseek/deepseek-chat", + "google/gemini-3.1-flash-lite", + "openai/gpt-5.6-luna", + "openai/gpt-5.4-nano", + "google/gemini-2.5-flash-lite" + ], + "candidateScores": [ + { + "model": "google/gemini-2.5-flash", + "score": 0.7279229342146076, + "quality": 0.86, + "cost": 0.5, + "speed": 0.18322934214607486, + "reliability": 1 + } + ], + "taskType": "chat", + "routerVersion": "v3-portfolio" + }, + { + "prompt": 16, + "profile": "eco", + "model": "anthropic/claude-sonnet-5", + "tier": "MEDIUM", + "confidence": 0.5, + "method": "portfolio", + "reasoning": "score=-0.06 | short (38 tokens), imperative (deploy), references (the api), agentic-light (check the, deploy) | ambiguous -> default: MEDIUM | eco | v3 task=tool_agent agentRisk=standard deepWebResearch=false terminalCode=false terminalSafety=false candidates=16", + "costEstimate": 0.0006437999999999999, + "baselineCost": 0.0065899999999999995, + "savings": 0.9023065250379363, + "agenticScore": 0.2, + "candidates": [ + "anthropic/claude-sonnet-5", + "openai/gpt-5-mini", + "openai/gpt-5.3-codex", + "deepseek/deepseek-v4-pro", + "google/gemini-3.5-flash", + "moonshot/kimi-k3", + "zai/glm-5.3-flash", + "deepseek/deepseek-chat", + "google/gemini-3.1-flash-lite", + "openai/gpt-5.6-luna", + "openai/gpt-5.4-nano", + "google/gemini-2.5-flash-lite", + "google/gemini-2.5-flash", + "anthropic/claude-opus-5", + "openai/gpt-4.1", + "openai/gpt-4o-mini" + ], + "candidateScores": [ + { + "model": "anthropic/claude-sonnet-5", + "score": 0.9098830643397023, + "quality": 1, + "cost": 1, + "speed": 0.09883064339702209, + "reliability": 1 + }, + { + "model": "openai/gpt-5-mini", + "score": 0.8332830628561794, + "quality": 0.84, + "cost": 0.9339513325608342, + "speed": 0.13376689739145697, + "reliability": 1 + }, + { + "model": "openai/gpt-5.3-codex", + "score": 0.8248419307935564, + "quality": 0.87, + "cost": 0.9559675550405561, + "speed": 0.039714153822005965, + "reliability": 0.79999 + }, + { + "model": "deepseek/deepseek-v4-pro", + "score": 0.6852374607066745, + "quality": 0.82, + "cost": 0.5110081112398608, + "speed": 0.06955189559513505, + "reliability": 1 + }, + { + "model": "google/gemini-3.5-flash", + "score": 0.5880110618276134, + "quality": 0.88, + "cost": 0, + "speed": 0.19211061827613424, + "reliability": 1 + }, + { + "model": "moonshot/kimi-k3", + "score": 0.5588956870698019, + "quality": 0.85, + "cost": 0.022016222479721792, + "speed": 0.02731144775479715, + "reliability": 1 + } + ], + "taskType": "tool_agent", + "routerVersion": "v3-portfolio" + }, + { + "prompt": 17, + "profile": "eco", + "model": "zai/glm-5.3-flash", + "tier": "MEDIUM", + "confidence": 0.5, + "method": "portfolio", + "reasoning": "score=-0.06 | short (9 tokens), creative (write a) | ambiguous -> default: MEDIUM | eco | v3 task=vision agentRisk=standard deepWebResearch=false terminalCode=false terminalSafety=false candidates=3", + "costEstimate": 0.011271200000000002, + "baselineCost": 0.032045000000000004, + "savings": 0.6482696208456857, + "agenticScore": 0, + "candidates": [ + "zai/glm-5.3-flash", + "openai/gpt-5.6-luna", + "google/gemini-2.5-flash" + ], + "candidateScores": [ + { + "model": "zai/glm-5.3-flash", + "score": 0.6505854692782567, + "quality": 0.68, + "cost": 0.5, + "speed": 0.05785469278256664, + "reliability": 1 + } + ], + "taskType": "vision", + "routerVersion": "v3-portfolio" + }, + { + "prompt": 18, + "profile": "eco", + "model": "google/gemini-2.5-flash", + "tier": "MEDIUM", + "confidence": 0.7685247834990178, + "method": "portfolio", + "reasoning": "score=-0.10 | short (30 tokens), simple (translate) | upgraded to MEDIUM (structured output) | eco | v3 task=extraction agentRisk=standard deepWebResearch=false terminalCode=false terminalSafety=false candidates=10", + "costEstimate": 0.0010086000000000001, + "baselineCost": 0.057749999999999996, + "savings": 0.9825350649350649, + "agenticScore": 0, + "candidates": [ + "google/gemini-2.5-flash", + "anthropic/claude-sonnet-5", + "google/gemini-3.5-flash", + "moonshot/kimi-k3", + "zai/glm-5.3-flash", + "deepseek/deepseek-chat", + "google/gemini-3.1-flash-lite", + "openai/gpt-5.6-luna", + "openai/gpt-5.4-nano", + "google/gemini-2.5-flash-lite" + ], + "candidateScores": [ + { + "model": "google/gemini-2.5-flash", + "score": 0.8891443471782455, + "quality": 0.9, + "cost": 1, + "speed": 0.13969081765691935, + "reliability": 1 + }, + { + "model": "anthropic/claude-sonnet-5", + "score": 0.7770287467109466, + "quality": 0.9, + "cost": 0.6729268997399596, + "speed": 0.1367178599097662, + "reliability": 1 + }, + { + "model": "google/gemini-3.5-flash", + "score": 0.6005020197183445, + "quality": 0.9, + "cost": 0, + "speed": 0.16575196139820988, + "reliability": 1 + }, + { + "model": "moonshot/kimi-k3", + "score": 0.5817511887996366, + "quality": 0.9, + "cost": 0.0014446691707599157, + "speed": 0.022296378324947415, + "reliability": 1 + } + ], + "taskType": "extraction", + "routerVersion": "v3-portfolio" + }, + { + "prompt": 19, + "profile": "eco", + "model": "xai/grok-4.5", + "tier": "REASONING", + "confidence": 0.5, + "method": "portfolio", + "reasoning": "score=-0.07 | short (33 tokens), constraints (budget, budget) | ambiguous -> default: MEDIUM | eco | v3 task=reasoning_math agentRisk=standard deepWebResearch=false terminalCode=false terminalSafety=false candidates=9", + "costEstimate": 0.0014038, + "baselineCost": 0.083365, + "savings": 0.9831607988964194, + "agenticScore": 0, + "candidates": [ + "xai/grok-4.5", + "anthropic/claude-sonnet-5", + "deepseek/deepseek-v4-pro", + "google/gemini-3.5-flash", + "moonshot/kimi-k3", + "deepseek/deepseek-reasoner", + "qwen/qwen3.7-plus", + "minimax/minimax-m3", + "zai/glm-5.3-flash" + ], + "candidateScores": [ + { + "model": "xai/grok-4.5", + "score": 0.8706542065109696, + "quality": 0.93, + "cost": 1, + "speed": 0.05854206510969567, + "reliability": 1 + }, + { + "model": "anthropic/claude-sonnet-5", + "score": 0.7666783871460185, + "quality": 0.9, + "cost": 0.6706975814511295, + "speed": 0.09883064339702209, + "reliability": 1 + }, + { + "model": "deepseek/deepseek-v4-pro", + "score": 0.6850067586180785, + "quality": 0.9, + "cost": 0.335898460923446, + "speed": 0.06955189559513505, + "reliability": 1 + }, + { + "model": "google/gemini-3.5-flash", + "score": 0.6342110618276136, + "quality": 1, + "cost": 0, + "speed": 0.19211061827613424, + "reliability": 1 + }, + { + "model": "moonshot/kimi-k3", + "score": 0.5670389600862934, + "quality": 0.9, + "cost": 0.001099340395762649, + "speed": 0.02731144775479715, + "reliability": 1 + } + ], + "taskType": "reasoning_math", + "routerVersion": "v3-portfolio" + }, + { + "prompt": 20, + "profile": "eco", + "model": "zai/glm-5.3-flash", + "tier": "MEDIUM", + "confidence": 0.5, + "method": "portfolio", + "reasoning": "score=0.01 | imperative (design) | ambiguous -> default: MEDIUM | eco | v3 task=debug agentRisk=standard deepWebResearch=false terminalCode=false terminalSafety=false candidates=7", + "costEstimate": 0.0022952000000000003, + "baselineCost": 0.006664999999999999, + "savings": 0.6556339084771191, + "agenticScore": 0, + "candidates": [ + "zai/glm-5.3-flash", + "deepseek/deepseek-chat", + "google/gemini-3.1-flash-lite", + "openai/gpt-5.6-luna", + "openai/gpt-5.4-nano", + "google/gemini-2.5-flash-lite", + "google/gemini-2.5-flash" + ], + "candidateScores": [ + { + "model": "zai/glm-5.3-flash", + "score": 0.6505854692782567, + "quality": 0.68, + "cost": 0.5, + "speed": 0.05785469278256664, + "reliability": 1 + } + ], + "taskType": "debug", + "routerVersion": "v3-portfolio" + }, + { + "prompt": 21, + "profile": "eco", + "model": "anthropic/claude-sonnet-5", + "tier": "MEDIUM", + "confidence": 0.7231218051243898, + "method": "portfolio", + "reasoning": "score=0.08 | long (14423 tokens) | eco | v3 task=tool_agent agentRisk=standard deepWebResearch=false terminalCode=false terminalSafety=false candidates=16", + "costEstimate": 0.0046423, + "baselineCost": 0.104115, + "savings": 0.9554118042549105, + "agenticScore": 0, + "candidates": [ + "anthropic/claude-sonnet-5", + "openai/gpt-5.3-codex", + "deepseek/deepseek-v4-pro", + "openai/gpt-5-mini", + "moonshot/kimi-k3", + "google/gemini-3.5-flash", + "zai/glm-5.3-flash", + "deepseek/deepseek-chat", + "google/gemini-3.1-flash-lite", + "openai/gpt-5.6-luna", + "openai/gpt-5.4-nano", + "google/gemini-2.5-flash-lite", + "google/gemini-2.5-flash", + "anthropic/claude-opus-5", + "openai/gpt-4.1", + "openai/gpt-4o-mini" + ], + "candidateScores": [ + { + "model": "anthropic/claude-sonnet-5", + "score": 0.9098830643397023, + "quality": 1, + "cost": 1, + "speed": 0.09883064339702209, + "reliability": 1 + }, + { + "model": "openai/gpt-5.3-codex", + "score": 0.735950638165583, + "quality": 0.87, + "cost": 0.6384986527977944, + "speed": 0.039714153822005965, + "reliability": 0.79999 + }, + { + "model": "deepseek/deepseek-v4-pro", + "score": 0.7074602838636679, + "quality": 0.82, + "cost": 0.5903753368005515, + "speed": 0.06955189559513505, + "reliability": 1 + }, + { + "model": "openai/gpt-5-mini", + "score": 0.6999461239142194, + "quality": 0.84, + "cost": 0.45774797919669175, + "speed": 0.13376689739145697, + "reliability": 1 + }, + { + "model": "moonshot/kimi-k3", + "score": 0.6033413333837886, + "quality": 0.85, + "cost": 0.18075067360110297, + "speed": 0.02731144775479715, + "reliability": 1 + }, + { + "model": "google/gemini-3.5-flash", + "score": 0.5880110618276134, + "quality": 0.88, + "cost": 0, + "speed": 0.19211061827613424, + "reliability": 1 + } + ], + "taskType": "tool_agent", + "routerVersion": "v3-portfolio" + }, + { + "prompt": 0, + "profile": "auto", + "model": "google/gemini-2.5-flash", + "tier": "SIMPLE", + "confidence": 0.7685247834990178, + "method": "portfolio", + "reasoning": "score=-0.10 | short (15 tokens), simple (what is, capital of) | v3 task=chat agentRisk=standard deepWebResearch=false terminalCode=false terminalSafety=false candidates=9", + "costEstimate": 0.0001459, + "baselineCost": 0.006475, + "savings": 0.9774671814671815, + "agenticScore": 0, + "candidates": [ + "google/gemini-2.5-flash", + "google/gemini-3-flash-preview", + "google/gemini-3.5-flash-lite", + "deepseek/deepseek-chat", + "google/gemini-3.1-flash-lite", + "openai/gpt-5.6-luna", + "openai/gpt-5.4-nano", + "google/gemini-2.5-flash-lite", + "nvidia/nemotron-3.5-lightning" + ], + "candidateScores": [ + { + "model": "google/gemini-2.5-flash", + "score": 0.7870260539502253, + "quality": 0.86, + "cost": 0.5, + "speed": 0.18322934214607486, + "reliability": 1 + } + ], + "taskType": "chat", + "routerVersion": "v3-portfolio" + }, + { + "prompt": 1, + "profile": "agentic", + "model": "anthropic/claude-sonnet-5", + "tier": "SIMPLE", + "confidence": 0.7231218051243898, + "method": "portfolio", + "reasoning": "score=-0.08 | short (1 tokens) | agentic (tools) | v3 task=tool_agent agentRisk=standard deepWebResearch=false terminalCode=false terminalSafety=false candidates=13", + "costEstimate": 0.0032001, + "baselineCost": 0.032005, + "savings": 0.9000124980471801, + "agenticScore": 0, + "candidates": [ + "anthropic/claude-sonnet-5", + "openai/gpt-4o-mini", + "openai/gpt-5.6-luna", + "zai/glm-5.3-flash", + "anthropic/claude-haiku-4.5", + "google/gemini-2.5-flash", + "anthropic/claude-opus-5", + "openai/gpt-5-mini", + "openai/gpt-4.1", + "google/gemini-3.5-flash", + "openai/gpt-5.3-codex", + "moonshot/kimi-k3", + "deepseek/deepseek-v4-pro" + ], + "candidateScores": [ + { + "model": "anthropic/claude-sonnet-5", + "score": 0.8469181450377915, + "quality": 1, + "cost": 0.5, + "speed": 0.09883064339702209, + "reliability": 1 + } + ], + "taskType": "tool_agent", + "routerVersion": "v3-portfolio" + }, + { + "prompt": 2, + "profile": "auto", + "model": "google/gemini-2.5-flash", + "tier": "SIMPLE", + "confidence": 0.7231218051243898, + "method": "portfolio", + "reasoning": "score=-0.08 | short (16 tokens) | v3 task=vision agentRisk=standard deepWebResearch=false terminalCode=false terminalSafety=false candidates=3", + "costEstimate": 0.0009680000000000001, + "baselineCost": 0.057679999999999995, + "savings": 0.9832177531206658, + "agenticScore": 0, + "candidates": [ + "google/gemini-2.5-flash", + "google/gemini-3-flash-preview", + "openai/gpt-5.6-luna" + ], + "candidateScores": [ + { + "model": "google/gemini-2.5-flash", + "score": 0.7024260539502254, + "quality": 0.68, + "cost": 0.5, + "speed": 0.18322934214607486, + "reliability": 1 + } + ], + "taskType": "vision", + "routerVersion": "v3-portfolio" + }, + { + "prompt": 3, + "profile": "auto", + "model": "google/gemini-3.5-flash", + "tier": "MEDIUM", + "confidence": 0.5, + "method": "portfolio", + "reasoning": "score=0.02 | short (31 tokens), code (function), creative (write a) | ambiguous -> default: MEDIUM | v3 task=code_edit agentRisk=standard deepWebResearch=false terminalCode=false terminalSafety=false candidates=11", + "costEstimate": 0.0224092, + "baselineCost": 0.083355, + "savings": 0.7311594985303821, + "agenticScore": 0, + "candidates": [ + "google/gemini-3.5-flash", + "google/gemini-3.6-flash", + "zai/glm-5.3-flash", + "openai/gpt-5.6-terra", + "google/gemini-3-flash-preview", + "deepseek/deepseek-chat", + "google/gemini-2.5-flash", + "minimax/minimax-m3", + "google/gemini-3.1-flash-lite", + "openai/gpt-5.6-luna", + "google/gemini-2.5-flash-lite" + ], + "candidateScores": [ + { + "model": "google/gemini-3.5-flash", + "score": 0.7030477432793295, + "quality": 0.68, + "cost": 0.5, + "speed": 0.19211061827613424, + "reliability": 1 + } + ], + "taskType": "code_edit", + "routerVersion": "v3-portfolio" + }, + { + "prompt": 4, + "profile": "auto", + "model": "deepseek/deepseek-v4-pro", + "tier": "REASONING", + "confidence": 0.973403006423134, + "method": "portfolio", + "reasoning": "score=0.10 | short (16 tokens), reasoning (prove, step by step) | v3 task=reasoning agentRisk=standard deepWebResearch=false terminalCode=false terminalSafety=false candidates=7", + "costEstimate": 0.0012016, + "baselineCost": 0.00648, + "savings": 0.8145679012345679, + "agenticScore": 0, + "candidates": [ + "deepseek/deepseek-v4-pro", + "google/gemini-3.5-flash", + "deepseek/deepseek-reasoner", + "xai/grok-4.3", + "qwen/qwen3.7-plus", + "openai/o4-mini", + "openai/o3" + ], + "candidateScores": [ + { + "model": "deepseek/deepseek-v4-pro", + "score": 0.9113686326916596, + "quality": 0.95, + "cost": 1, + "speed": 0.06955189559513505, + "reliability": 1 + }, + { + "model": "google/gemini-3.5-flash", + "score": 0.6758477432793295, + "quality": 0.92, + "cost": 0, + "speed": 0.19211061827613424, + "reliability": 1 + } + ], + "taskType": "reasoning", + "routerVersion": "v3-portfolio" + }, + { + "prompt": 5, + "profile": "auto", + "model": "google/gemini-3.5-flash", + "tier": "MEDIUM", + "confidence": 0.5, + "method": "portfolio", + "reasoning": "score=0.07 | short (30 tokens), code (class, return, ```) | ambiguous -> default: MEDIUM | v3 task=code_edit agentRisk=standard deepWebResearch=false terminalCode=false terminalSafety=false candidates=11", + "costEstimate": 0.008684000000000002, + "baselineCost": 0.03215, + "savings": 0.7298911353032658, + "agenticScore": 0, + "candidates": [ + "google/gemini-3.5-flash", + "google/gemini-3.6-flash", + "zai/glm-5.3-flash", + "openai/gpt-5.6-terra", + "google/gemini-3-flash-preview", + "deepseek/deepseek-chat", + "google/gemini-2.5-flash", + "minimax/minimax-m3", + "google/gemini-3.1-flash-lite", + "openai/gpt-5.6-luna", + "google/gemini-2.5-flash-lite" + ], + "candidateScores": [ + { + "model": "google/gemini-3.5-flash", + "score": 0.7030477432793295, + "quality": 0.68, + "cost": 0.5, + "speed": 0.19211061827613424, + "reliability": 1 + } + ], + "taskType": "code_edit", + "routerVersion": "v3-portfolio" + }, + { + "prompt": 6, + "profile": "agentic", + "model": "anthropic/claude-opus-4.8", + "tier": "SIMPLE", + "confidence": 0.7231218051243898, + "method": "portfolio", + "reasoning": "score=-0.08 | short (20 tokens) | agentic (tools) | v3 task=tool_agent_parallel agentRisk=high deepWebResearch=false terminalCode=false terminalSafety=false candidates=13", + "costEstimate": 0.020291200000000002, + "baselineCost": 0.0577, + "savings": 0.6483327556325823, + "agenticScore": 0, + "candidates": [ + "anthropic/claude-opus-4.8", + "openai/gpt-4o-mini", + "openai/gpt-5.6-luna", + "zai/glm-5.3-flash", + "anthropic/claude-haiku-4.5", + "google/gemini-2.5-flash", + "anthropic/claude-opus-5", + "anthropic/claude-sonnet-5", + "openai/gpt-5-mini", + "openai/gpt-4.1", + "xai/grok-4.5", + "google/gemini-3.5-flash", + "deepseek/deepseek-v4-pro" + ], + "candidateScores": [ + { + "model": "anthropic/claude-opus-4.8", + "score": 0.8468563440295361, + "quality": 1, + "cost": 0.5, + "speed": 0.09794777185051722, + "reliability": 1 + } + ], + "taskType": "tool_agent_parallel", + "routerVersion": "v3-portfolio" + }, + { + "prompt": 7, + "profile": "auto", + "model": "google/gemini-2.5-flash", + "tier": "SIMPLE", + "confidence": 0.7231218051243898, + "method": "portfolio", + "reasoning": "score=-0.08 | short (13 tokens) | v3 task=vision agentRisk=standard deepWebResearch=false terminalCode=false terminalSafety=false candidates=3", + "costEstimate": 0.0013689000000000002, + "baselineCost": 0.08326499999999999, + "savings": 0.983559718969555, + "agenticScore": 0, + "candidates": [ + "google/gemini-2.5-flash", + "google/gemini-3-flash-preview", + "openai/gpt-5.6-luna" + ], + "candidateScores": [ + { + "model": "google/gemini-2.5-flash", + "score": 0.7024260539502254, + "quality": 0.68, + "cost": 0.5, + "speed": 0.18322934214607486, + "reliability": 1 + } + ], + "taskType": "vision", + "routerVersion": "v3-portfolio" + }, + { + "prompt": 8, + "profile": "auto", + "model": "moonshot/kimi-k3", + "tier": "MEDIUM", + "confidence": 0.5, + "method": "portfolio", + "reasoning": "score=-0.07 | short (5 tokens), constraints (不超过) | ambiguous -> default: MEDIUM | v3 task=extraction agentRisk=standard deepWebResearch=false terminalCode=false terminalSafety=false candidates=13", + "costEstimate": 0.0017297000000000002, + "baselineCost": 0.006424999999999999, + "savings": 0.7307859922178989, + "agenticScore": 0, + "candidates": [ + "moonshot/kimi-k3", + "google/gemini-3.5-flash", + "google/gemini-3.6-flash", + "zai/glm-5.3-flash", + "openai/gpt-5.6-terra", + "google/gemini-3-flash-preview", + "deepseek/deepseek-chat", + "google/gemini-2.5-flash", + "minimax/minimax-m3", + "google/gemini-3.1-flash-lite", + "openai/gpt-5.6-luna", + "google/gemini-2.5-flash-lite", + "anthropic/claude-sonnet-5" + ], + "candidateScores": [ + { + "model": "moonshot/kimi-k3", + "score": 0.8419118013428357, + "quality": 1, + "cost": 0.5, + "speed": 0.02731144775479715, + "reliability": 1 + } + ], + "taskType": "extraction", + "routerVersion": "v3-portfolio" + }, + { + "prompt": 9, + "profile": "auto", + "model": "google/gemini-2.5-flash", + "tier": "MEDIUM", + "confidence": 0.5, + "method": "portfolio", + "reasoning": "score=-0.07 | short (16 tokens), imperative (设计), references (代码) | ambiguous -> default: MEDIUM | v3 task=chat agentRisk=standard deepWebResearch=false terminalCode=false terminalSafety=false candidates=11", + "costEstimate": 0.0005584000000000001, + "baselineCost": 0.03208, + "savings": 0.9825935162094763, + "agenticScore": 0, + "candidates": [ + "google/gemini-2.5-flash", + "google/gemini-3.5-flash", + "google/gemini-3.6-flash", + "zai/glm-5.3-flash", + "openai/gpt-5.6-terra", + "google/gemini-3-flash-preview", + "deepseek/deepseek-chat", + "minimax/minimax-m3", + "google/gemini-3.1-flash-lite", + "openai/gpt-5.6-luna", + "google/gemini-2.5-flash-lite" + ], + "candidateScores": [ + { + "model": "google/gemini-2.5-flash", + "score": 0.8270260539502253, + "quality": 0.86, + "cost": 1, + "speed": 0.18322934214607486, + "reliability": 1 + }, + { + "model": "google/gemini-3.5-flash", + "score": 0.6976477432793294, + "quality": 0.86, + "cost": 0, + "speed": 0.19211061827613424, + "reliability": 1 + } + ], + "taskType": "chat", + "routerVersion": "v3-portfolio" + }, + { + "prompt": 10, + "profile": "auto", + "model": "google/gemini-3.5-flash", + "tier": "MEDIUM", + "confidence": 0.5, + "method": "portfolio", + "reasoning": "score=-0.02 | short (23 tokens), code (def), simple (define), agentic-light (debug) | ambiguous -> default: MEDIUM | v3 task=debug agentRisk=standard deepWebResearch=false terminalCode=false terminalSafety=false candidates=11", + "costEstimate": 0.015519600000000001, + "baselineCost": 0.057714999999999995, + "savings": 0.7310993675820844, + "agenticScore": 0.2, + "candidates": [ + "google/gemini-3.5-flash", + "google/gemini-3.6-flash", + "zai/glm-5.3-flash", + "openai/gpt-5.6-terra", + "google/gemini-3-flash-preview", + "deepseek/deepseek-chat", + "google/gemini-2.5-flash", + "minimax/minimax-m3", + "google/gemini-3.1-flash-lite", + "openai/gpt-5.6-luna", + "google/gemini-2.5-flash-lite" + ], + "candidateScores": [ + { + "model": "google/gemini-3.5-flash", + "score": 0.7030477432793295, + "quality": 0.68, + "cost": 0.5, + "speed": 0.19211061827613424, + "reliability": 1 + } + ], + "taskType": "debug", + "routerVersion": "v3-portfolio" + }, + { + "prompt": 11, + "profile": "agentic", + "model": "anthropic/claude-sonnet-5", + "tier": "MEDIUM", + "confidence": 0.7373034537835593, + "method": "portfolio", + "reasoning": "score=0.09 | long (1217 tokens), references (following) | agentic (tools) | v3 task=tool_agent agentRisk=standard deepWebResearch=false terminalCode=false terminalSafety=false candidates=13", + "costEstimate": 0.0084417, + "baselineCost": 0.089285, + "savings": 0.905452203617629, + "agenticScore": 0, + "candidates": [ + "anthropic/claude-sonnet-5", + "openai/gpt-5-mini", + "google/gemini-3.5-flash", + "zai/glm-5.3-flash", + "openai/gpt-5.6-terra", + "openai/gpt-4o-mini", + "anthropic/claude-haiku-4.5", + "deepseek/deepseek-chat", + "moonshot/kimi-k3", + "anthropic/claude-opus-5", + "openai/gpt-4.1", + "openai/gpt-5.3-codex", + "deepseek/deepseek-v4-pro" + ], + "candidateScores": [ + { + "model": "anthropic/claude-sonnet-5", + "score": 0.8469181450377915, + "quality": 1, + "cost": 0.5, + "speed": 0.09883064339702209, + "reliability": 1 + } + ], + "taskType": "tool_agent", + "routerVersion": "v3-portfolio" + }, + { + "prompt": 12, + "profile": "auto", + "model": "google/gemini-3.5-flash", + "tier": "MEDIUM", + "confidence": 0.5, + "method": "portfolio", + "reasoning": "score=0.06 | short (32 tokens), code (let), multi-step, references (following) | ambiguous -> default: MEDIUM | v3 task=vision agentRisk=standard deepWebResearch=false terminalCode=false terminalSafety=false candidates=7", + "costEstimate": 0.0018304000000000003, + "baselineCost": 0.00656, + "savings": 0.7209756097560975, + "agenticScore": 0, + "candidates": [ + "google/gemini-3.5-flash", + "google/gemini-3.6-flash", + "zai/glm-5.3-flash", + "openai/gpt-5.6-terra", + "google/gemini-3-flash-preview", + "google/gemini-2.5-flash", + "openai/gpt-5.6-luna" + ], + "candidateScores": [ + { + "model": "google/gemini-3.5-flash", + "score": 0.7030477432793295, + "quality": 0.68, + "cost": 0.5, + "speed": 0.19211061827613424, + "reliability": 1 + } + ], + "taskType": "vision", + "routerVersion": "v3-portfolio" + }, + { + "prompt": 13, + "profile": "auto", + "model": "xai/grok-4.5", + "tier": "REASONING", + "confidence": 0.5, + "method": "portfolio", + "reasoning": "score=-0.02 | short (13 tokens), multi-step | ambiguous -> default: MEDIUM | v3 task=reasoning_math agentRisk=standard deepWebResearch=false terminalCode=false terminalSafety=false candidates=10", + "costEstimate": 0.0005406000000000001, + "baselineCost": 0.032065, + "savings": 0.9831404958677685, + "agenticScore": 0, + "candidates": [ + "xai/grok-4.5", + "anthropic/claude-sonnet-5", + "deepseek/deepseek-v4-pro", + "google/gemini-3.5-flash", + "moonshot/kimi-k3", + "deepseek/deepseek-reasoner", + "xai/grok-4.3", + "qwen/qwen3.7-plus", + "openai/o4-mini", + "openai/o3" + ], + "candidateScores": [ + { + "model": "xai/grok-4.5", + "score": 0.8761979445576786, + "quality": 0.93, + "cost": 1, + "speed": 0.05854206510969567, + "reliability": 1 + }, + { + "model": "anthropic/claude-sonnet-5", + "score": 0.7931612595363191, + "quality": 0.9, + "cost": 0.6707950805473757, + "speed": 0.09883064339702209, + "reliability": 1 + }, + { + "model": "deepseek/deepseek-v4-pro", + "score": 0.7683415237361771, + "quality": 0.9, + "cost": 0.3359605058028754, + "speed": 0.06955189559513505, + "reliability": 1 + }, + { + "model": "google/gemini-3.5-flash", + "score": 0.7509477432793293, + "quality": 1, + "cost": 0, + "speed": 0.19211061827613424, + "reliability": 1 + }, + { + "model": "moonshot/kimi-k3", + "score": 0.6551144689333432, + "quality": 0.9, + "cost": 0.001125931058375107, + "speed": 0.02731144775479715, + "reliability": 1 + } + ], + "taskType": "reasoning_math", + "routerVersion": "v3-portfolio" + }, + { + "prompt": 14, + "profile": "auto", + "model": "google/gemini-2.5-flash", + "tier": "MEDIUM", + "confidence": 0.5, + "method": "portfolio", + "reasoning": "score=-0.07 | short (25 tokens), format (json) | ambiguous -> default: MEDIUM | v3 task=extraction agentRisk=standard deepWebResearch=false terminalCode=false terminalSafety=false candidates=13", + "costEstimate": 0.0009941000000000001, + "baselineCost": 0.057725, + "savings": 0.9827786920744912, + "agenticScore": 0, + "candidates": [ + "google/gemini-2.5-flash", + "anthropic/claude-sonnet-5", + "google/gemini-3.5-flash", + "moonshot/kimi-k3", + "google/gemini-3.6-flash", + "zai/glm-5.3-flash", + "openai/gpt-5.6-terra", + "google/gemini-3-flash-preview", + "deepseek/deepseek-chat", + "minimax/minimax-m3", + "google/gemini-3.1-flash-lite", + "openai/gpt-5.6-luna", + "google/gemini-2.5-flash-lite" + ], + "candidateScores": [ + { + "model": "google/gemini-2.5-flash", + "score": 0.8791593872835586, + "quality": 0.9, + "cost": 1, + "speed": 0.18322934214607486, + "reliability": 1 + }, + { + "model": "anthropic/claude-sonnet-5", + "score": 0.7808574061523814, + "quality": 0.9, + "cost": 0.6718847839699437, + "speed": 0.09883064339702209, + "reliability": 1 + }, + { + "model": "google/gemini-3.5-flash", + "score": 0.7164477432793295, + "quality": 0.9, + "cost": 0, + "speed": 0.19211061827613424, + "reliability": 1 + }, + { + "model": "moonshot/kimi-k3", + "score": 0.6717952205744078, + "quality": 0.9, + "cost": 0.001204180916140718, + "speed": 0.02731144775479715, + "reliability": 1 + } + ], + "taskType": "extraction", + "routerVersion": "v3-portfolio" + }, + { + "prompt": 15, + "profile": "auto", + "model": "google/gemini-2.5-flash", + "tier": "MEDIUM", + "confidence": 0.5, + "method": "portfolio", + "reasoning": "score=-0.06 | short (29 tokens), imperative (build), agentic-light (pip) | ambiguous -> default: MEDIUM | v3 task=chat agentRisk=standard deepWebResearch=false terminalCode=false terminalSafety=false candidates=11", + "costEstimate": 0.0014153, + "baselineCost": 0.083345, + "savings": 0.983018777371168, + "agenticScore": 0.2, + "candidates": [ + "google/gemini-2.5-flash", + "google/gemini-3.5-flash", + "google/gemini-3.6-flash", + "zai/glm-5.3-flash", + "openai/gpt-5.6-terra", + "google/gemini-3-flash-preview", + "deepseek/deepseek-chat", + "minimax/minimax-m3", + "google/gemini-3.1-flash-lite", + "openai/gpt-5.6-luna", + "google/gemini-2.5-flash-lite" + ], + "candidateScores": [ + { + "model": "google/gemini-2.5-flash", + "score": 0.8270260539502253, + "quality": 0.86, + "cost": 1, + "speed": 0.18322934214607486, + "reliability": 1 + }, + { + "model": "google/gemini-3.5-flash", + "score": 0.6976477432793294, + "quality": 0.86, + "cost": 0, + "speed": 0.19211061827613424, + "reliability": 1 + } + ], + "taskType": "chat", + "routerVersion": "v3-portfolio" + }, + { + "prompt": 16, + "profile": "agentic", + "model": "anthropic/claude-sonnet-5", + "tier": "MEDIUM", + "confidence": 0.5, + "method": "portfolio", + "reasoning": "score=-0.06 | short (38 tokens), imperative (deploy), references (the api), agentic-light (check the, deploy) | ambiguous -> default: MEDIUM | agentic (tools) | v3 task=tool_agent agentRisk=standard deepWebResearch=false terminalCode=false terminalSafety=false candidates=13", + "costEstimate": 0.0006437999999999999, + "baselineCost": 0.0065899999999999995, + "savings": 0.9023065250379363, + "agenticScore": 0.2, + "candidates": [ + "anthropic/claude-sonnet-5", + "openai/gpt-5-mini", + "google/gemini-3.5-flash", + "zai/glm-5.3-flash", + "openai/gpt-5.6-terra", + "openai/gpt-4o-mini", + "anthropic/claude-haiku-4.5", + "deepseek/deepseek-chat", + "moonshot/kimi-k3", + "anthropic/claude-opus-5", + "openai/gpt-4.1", + "openai/gpt-5.3-codex", + "deepseek/deepseek-v4-pro" + ], + "candidateScores": [ + { + "model": "anthropic/claude-sonnet-5", + "score": 0.8469181450377915, + "quality": 1, + "cost": 0.5, + "speed": 0.09883064339702209, + "reliability": 1 + } + ], + "taskType": "tool_agent", + "routerVersion": "v3-portfolio" + }, + { + "prompt": 17, + "profile": "auto", + "model": "google/gemini-3.5-flash", + "tier": "MEDIUM", + "confidence": 0.5, + "method": "portfolio", + "reasoning": "score=-0.06 | short (9 tokens), creative (write a) | ambiguous -> default: MEDIUM | v3 task=vision agentRisk=standard deepWebResearch=false terminalCode=false terminalSafety=false candidates=7", + "costEstimate": 0.008608400000000002, + "baselineCost": 0.032045000000000004, + "savings": 0.7313652675924481, + "agenticScore": 0, + "candidates": [ + "google/gemini-3.5-flash", + "google/gemini-3.6-flash", + "zai/glm-5.3-flash", + "openai/gpt-5.6-terra", + "google/gemini-3-flash-preview", + "google/gemini-2.5-flash", + "openai/gpt-5.6-luna" + ], + "candidateScores": [ + { + "model": "google/gemini-3.5-flash", + "score": 0.7030477432793295, + "quality": 0.68, + "cost": 0.5, + "speed": 0.19211061827613424, + "reliability": 1 + } + ], + "taskType": "vision", + "routerVersion": "v3-portfolio" + }, + { + "prompt": 18, + "profile": "auto", + "model": "google/gemini-2.5-flash", + "tier": "MEDIUM", + "confidence": 0.7685247834990178, + "method": "portfolio", + "reasoning": "score=-0.10 | short (30 tokens), simple (translate) | upgraded to MEDIUM (structured output) | v3 task=extraction agentRisk=standard deepWebResearch=false terminalCode=false terminalSafety=false candidates=13", + "costEstimate": 0.0010086000000000001, + "baselineCost": 0.057749999999999996, + "savings": 0.9825350649350649, + "agenticScore": 0, + "candidates": [ + "google/gemini-2.5-flash", + "anthropic/claude-sonnet-5", + "google/gemini-3.5-flash", + "moonshot/kimi-k3", + "google/gemini-3.6-flash", + "zai/glm-5.3-flash", + "openai/gpt-5.6-terra", + "google/gemini-3-flash-preview", + "deepseek/deepseek-chat", + "minimax/minimax-m3", + "google/gemini-3.1-flash-lite", + "openai/gpt-5.6-luna", + "google/gemini-2.5-flash-lite" + ], + "candidateScores": [ + { + "model": "google/gemini-2.5-flash", + "score": 0.8872869559818712, + "quality": 0.9, + "cost": 1, + "speed": 0.13969081765691935, + "reliability": 1 + }, + { + "model": "anthropic/claude-sonnet-5", + "score": 0.7946345209396576, + "quality": 0.9, + "cost": 0.6729268997399596, + "speed": 0.1367178599097662, + "reliability": 1 + }, + { + "model": "google/gemini-3.5-flash", + "score": 0.7278627942097315, + "quality": 0.9, + "cost": 0, + "speed": 0.16575196139820988, + "reliability": 1 + }, + { + "model": "moonshot/kimi-k3", + "score": 0.6732711638661456, + "quality": 0.9, + "cost": 0.0014446691707599157, + "speed": 0.022296378324947415, + "reliability": 1 + } + ], + "taskType": "extraction", + "routerVersion": "v3-portfolio" + }, + { + "prompt": 19, + "profile": "auto", + "model": "xai/grok-4.5", + "tier": "REASONING", + "confidence": 0.5, + "method": "portfolio", + "reasoning": "score=-0.07 | short (33 tokens), constraints (budget, budget) | ambiguous -> default: MEDIUM | v3 task=reasoning_math agentRisk=standard deepWebResearch=false terminalCode=false terminalSafety=false candidates=10", + "costEstimate": 0.0014038, + "baselineCost": 0.083365, + "savings": 0.9831607988964194, + "agenticScore": 0, + "candidates": [ + "xai/grok-4.5", + "anthropic/claude-sonnet-5", + "deepseek/deepseek-v4-pro", + "google/gemini-3.5-flash", + "moonshot/kimi-k3", + "deepseek/deepseek-reasoner", + "xai/grok-4.3", + "qwen/qwen3.7-plus", + "openai/o4-mini", + "openai/o3" + ], + "candidateScores": [ + { + "model": "xai/grok-4.5", + "score": 0.8761979445576786, + "quality": 0.93, + "cost": 1, + "speed": 0.05854206510969567, + "reliability": 1 + }, + { + "model": "anthropic/claude-sonnet-5", + "score": 0.7931437096989948, + "quality": 0.9, + "cost": 0.6706975814511295, + "speed": 0.09883064339702209, + "reliability": 1 + }, + { + "model": "deepseek/deepseek-v4-pro", + "score": 0.7683303556578799, + "quality": 0.9, + "cost": 0.335898460923446, + "speed": 0.06955189559513505, + "reliability": 1 + }, + { + "model": "google/gemini-3.5-flash", + "score": 0.7509477432793293, + "quality": 1, + "cost": 0, + "speed": 0.19211061827613424, + "reliability": 1 + }, + { + "model": "moonshot/kimi-k3", + "score": 0.6551096826140731, + "quality": 0.9, + "cost": 0.001099340395762649, + "speed": 0.02731144775479715, + "reliability": 1 + } + ], + "taskType": "reasoning_math", + "routerVersion": "v3-portfolio" + }, + { + "prompt": 20, + "profile": "auto", + "model": "google/gemini-3.5-flash", + "tier": "MEDIUM", + "confidence": 0.5, + "method": "portfolio", + "reasoning": "score=0.01 | imperative (design) | ambiguous -> default: MEDIUM | v3 task=debug agentRisk=standard deepWebResearch=false terminalCode=false terminalSafety=false candidates=11", + "costEstimate": 0.0019060000000000004, + "baselineCost": 0.006664999999999999, + "savings": 0.7140285071267816, + "agenticScore": 0, + "candidates": [ + "google/gemini-3.5-flash", + "google/gemini-3.6-flash", + "zai/glm-5.3-flash", + "openai/gpt-5.6-terra", + "google/gemini-3-flash-preview", + "deepseek/deepseek-chat", + "google/gemini-2.5-flash", + "minimax/minimax-m3", + "google/gemini-3.1-flash-lite", + "openai/gpt-5.6-luna", + "google/gemini-2.5-flash-lite" + ], + "candidateScores": [ + { + "model": "google/gemini-3.5-flash", + "score": 0.7030477432793295, + "quality": 0.68, + "cost": 0.5, + "speed": 0.19211061827613424, + "reliability": 1 + } + ], + "taskType": "debug", + "routerVersion": "v3-portfolio" + }, + { + "prompt": 21, + "profile": "agentic", + "model": "anthropic/claude-sonnet-5", + "tier": "MEDIUM", + "confidence": 0.7231218051243898, + "method": "portfolio", + "reasoning": "score=0.08 | long (14423 tokens) | agentic (tools) | v3 task=tool_agent agentRisk=standard deepWebResearch=false terminalCode=false terminalSafety=false candidates=13", + "costEstimate": 0.0046423, + "baselineCost": 0.104115, + "savings": 0.9554118042549105, + "agenticScore": 0, + "candidates": [ + "anthropic/claude-sonnet-5", + "openai/gpt-5-mini", + "google/gemini-3.5-flash", + "zai/glm-5.3-flash", + "openai/gpt-5.6-terra", + "openai/gpt-4o-mini", + "anthropic/claude-haiku-4.5", + "deepseek/deepseek-chat", + "moonshot/kimi-k3", + "anthropic/claude-opus-5", + "openai/gpt-4.1", + "openai/gpt-5.3-codex", + "deepseek/deepseek-v4-pro" + ], + "candidateScores": [ + { + "model": "anthropic/claude-sonnet-5", + "score": 0.8469181450377915, + "quality": 1, + "cost": 0.5, + "speed": 0.09883064339702209, + "reliability": 1 + } + ], + "taskType": "tool_agent", + "routerVersion": "v3-portfolio" + }, + { + "prompt": 0, + "profile": "premium", + "model": "google/gemini-2.5-flash", + "tier": "SIMPLE", + "confidence": 0.7685247834990178, + "method": "portfolio", + "reasoning": "score=-0.10 | short (15 tokens), simple (what is, capital of) | premium | v3 task=chat agentRisk=standard deepWebResearch=false terminalCode=false terminalSafety=false candidates=7", + "costEstimate": 0.0001459, + "baselineCost": 0.006475, + "savings": 0, + "agenticScore": 0, + "candidates": [ + "google/gemini-2.5-flash", + "google/gemini-3.5-flash", + "google/gemini-3.6-flash", + "anthropic/claude-haiku-4.5", + "zai/glm-5.3", + "google/gemini-3.5-flash-lite", + "deepseek/deepseek-chat" + ], + "candidateScores": [ + { + "model": "google/gemini-2.5-flash", + "score": 0.8497937605287644, + "quality": 0.86, + "cost": 1, + "speed": 0.18322934214607486, + "reliability": 1 + }, + { + "model": "google/gemini-3.5-flash", + "score": 0.790326637096568, + "quality": 0.86, + "cost": 0, + "speed": 0.19211061827613424, + "reliability": 1 + } + ], + "taskType": "chat", + "routerVersion": "v3-portfolio" + }, + { + "prompt": 1, + "profile": "premium", + "model": "anthropic/claude-sonnet-5", + "tier": "SIMPLE", + "confidence": 0.7231218051243898, + "method": "portfolio", + "reasoning": "score=-0.08 | short (1 tokens) | premium | v3 task=tool_agent agentRisk=standard deepWebResearch=false terminalCode=false terminalSafety=false candidates=15", + "costEstimate": 0.0032001, + "baselineCost": 0.032005, + "savings": 0, + "agenticScore": 0, + "candidates": [ + "anthropic/claude-sonnet-5", + "google/gemini-3.5-flash", + "google/gemini-3.6-flash", + "anthropic/claude-haiku-4.5", + "zai/glm-5.3", + "google/gemini-2.5-flash", + "google/gemini-3.5-flash-lite", + "deepseek/deepseek-chat", + "anthropic/claude-opus-5", + "openai/gpt-5-mini", + "openai/gpt-4.1", + "openai/gpt-4o-mini", + "openai/gpt-5.3-codex", + "moonshot/kimi-k3", + "deepseek/deepseek-v4-pro" + ], + "candidateScores": [ + { + "model": "anthropic/claude-sonnet-5", + "score": 0.9059298386038215, + "quality": 1, + "cost": 0.5, + "speed": 0.09883064339702209, + "reliability": 1 + } + ], + "taskType": "tool_agent", + "routerVersion": "v3-portfolio" + }, + { + "prompt": 2, + "profile": "premium", + "model": "google/gemini-3.5-flash", + "tier": "SIMPLE", + "confidence": 0.7231218051243898, + "method": "portfolio", + "reasoning": "score=-0.08 | short (16 tokens) | premium | v3 task=vision agentRisk=standard deepWebResearch=false terminalCode=false terminalSafety=false candidates=4", + "costEstimate": 0.015494400000000002, + "baselineCost": 0.057679999999999995, + "savings": 0, + "agenticScore": 0, + "candidates": [ + "google/gemini-3.5-flash", + "google/gemini-3.6-flash", + "anthropic/claude-haiku-4.5", + "google/gemini-2.5-flash" + ], + "candidateScores": [ + { + "model": "google/gemini-3.5-flash", + "score": 0.7259266370965682, + "quality": 0.68, + "cost": 0.5, + "speed": 0.19211061827613424, + "reliability": 1 + } + ], + "taskType": "vision", + "routerVersion": "v3-portfolio" + }, + { + "prompt": 3, + "profile": "premium", + "model": "openai/gpt-5.3-codex", + "tier": "MEDIUM", + "confidence": 0.5, + "method": "portfolio", + "reasoning": "score=0.02 | short (31 tokens), code (function), creative (write a) | ambiguous -> default: MEDIUM | premium | v3 task=code_edit agentRisk=standard deepWebResearch=false terminalCode=false terminalSafety=false candidates=10", + "costEstimate": 0.008366499999999999, + "baselineCost": 0.083355, + "savings": 0, + "agenticScore": 0, + "candidates": [ + "openai/gpt-5.3-codex", + "anthropic/claude-sonnet-5", + "moonshot/kimi-k3", + "zai/glm-5.3", + "google/gemini-3.6-flash", + "google/gemini-3.5-flash", + "google/gemini-2.5-pro", + "xai/grok-4.5", + "anthropic/claude-sonnet-4.6", + "openai/gpt-5.6-terra" + ], + "candidateScores": [ + { + "model": "openai/gpt-5.3-codex", + "score": 0.8903822492293204, + "quality": 1, + "cost": 0.5, + "speed": 0.039714153822005965, + "reliability": 0.79999 + } + ], + "taskType": "code_edit", + "routerVersion": "v3-portfolio" + }, + { + "prompt": 4, + "profile": "premium", + "model": "anthropic/claude-sonnet-5", + "tier": "REASONING", + "confidence": 0.973403006423134, + "method": "portfolio", + "reasoning": "score=0.10 | short (16 tokens), reasoning (prove, step by step) | premium | v3 task=reasoning agentRisk=standard deepWebResearch=false terminalCode=false terminalSafety=false candidates=10", + "costEstimate": 0.0006416, + "baselineCost": 0.00648, + "savings": 0, + "agenticScore": 0, + "candidates": [ + "anthropic/claude-sonnet-5", + "xai/grok-4.5", + "anthropic/claude-sonnet-4.6", + "deepseek/deepseek-v4-pro", + "anthropic/claude-opus-5", + "anthropic/claude-opus-4.8", + "anthropic/claude-opus-4.7", + "xai/grok-4.3", + "openai/o4-mini", + "openai/o3" + ], + "candidateScores": [ + { + "model": "anthropic/claude-sonnet-5", + "score": 0.9093298386038213, + "quality": 0.98, + "cost": 0.6875, + "speed": 0.09883064339702209, + "reliability": 1 + }, + { + "model": "xai/grok-4.5", + "score": 0.8953791905732483, + "quality": 0.94, + "cost": 1, + "speed": 0.05854206510969567, + "reliability": 1 + }, + { + "model": "anthropic/claude-sonnet-4.6", + "score": 0.847328760008195, + "quality": 0.98, + "cost": 0, + "speed": 0.09325711124769513, + "reliability": 1 + }, + { + "model": "deepseek/deepseek-v4-pro", + "score": 0.8423953359579301, + "quality": 0.95, + "cost": 0.3402777777777777, + "speed": 0.06955189559513505, + "reliability": 1 + } + ], + "taskType": "reasoning", + "routerVersion": "v3-portfolio" + }, + { + "prompt": 5, + "profile": "premium", + "model": "openai/gpt-5.3-codex", + "tier": "MEDIUM", + "confidence": 0.5, + "method": "portfolio", + "reasoning": "score=0.07 | short (30 tokens), code (class, return, ```) | ambiguous -> default: MEDIUM | premium | v3 task=code_edit agentRisk=standard deepWebResearch=false terminalCode=false terminalSafety=false candidates=10", + "costEstimate": 0.003245, + "baselineCost": 0.03215, + "savings": 0, + "agenticScore": 0, + "candidates": [ + "openai/gpt-5.3-codex", + "anthropic/claude-sonnet-5", + "moonshot/kimi-k3", + "zai/glm-5.3", + "google/gemini-3.6-flash", + "google/gemini-3.5-flash", + "google/gemini-2.5-pro", + "xai/grok-4.5", + "anthropic/claude-sonnet-4.6", + "openai/gpt-5.6-terra" + ], + "candidateScores": [ + { + "model": "openai/gpt-5.3-codex", + "score": 0.8903822492293204, + "quality": 1, + "cost": 0.5, + "speed": 0.039714153822005965, + "reliability": 0.79999 + } + ], + "taskType": "code_edit", + "routerVersion": "v3-portfolio" + }, + { + "prompt": 6, + "profile": "premium", + "model": "anthropic/claude-opus-4.8", + "tier": "SIMPLE", + "confidence": 0.7231218051243898, + "method": "portfolio", + "reasoning": "score=-0.08 | short (20 tokens) | premium | v3 task=tool_agent_parallel agentRisk=high deepWebResearch=false terminalCode=false terminalSafety=false candidates=15", + "costEstimate": 0.020291200000000002, + "baselineCost": 0.0577, + "savings": 0, + "agenticScore": 0, + "candidates": [ + "anthropic/claude-opus-4.8", + "google/gemini-3.5-flash", + "google/gemini-3.6-flash", + "anthropic/claude-haiku-4.5", + "zai/glm-5.3", + "google/gemini-2.5-flash", + "google/gemini-3.5-flash-lite", + "deepseek/deepseek-chat", + "anthropic/claude-opus-5", + "anthropic/claude-sonnet-5", + "openai/gpt-5-mini", + "openai/gpt-4.1", + "openai/gpt-4o-mini", + "xai/grok-4.5", + "deepseek/deepseek-v4-pro" + ], + "candidateScores": [ + { + "model": "anthropic/claude-opus-4.8", + "score": 0.905876866311031, + "quality": 1, + "cost": 0.5, + "speed": 0.09794777185051722, + "reliability": 1 + } + ], + "taskType": "tool_agent_parallel", + "routerVersion": "v3-portfolio" + }, + { + "prompt": 7, + "profile": "premium", + "model": "google/gemini-3.5-flash", + "tier": "SIMPLE", + "confidence": 0.7231218051243898, + "method": "portfolio", + "reasoning": "score=-0.08 | short (13 tokens) | premium | v3 task=vision agentRisk=standard deepWebResearch=false terminalCode=false terminalSafety=false candidates=4", + "costEstimate": 0.0223444, + "baselineCost": 0.08326499999999999, + "savings": 0, + "agenticScore": 0, + "candidates": [ + "google/gemini-3.5-flash", + "google/gemini-3.6-flash", + "anthropic/claude-haiku-4.5", + "google/gemini-2.5-flash" + ], + "candidateScores": [ + { + "model": "google/gemini-3.5-flash", + "score": 0.7259266370965682, + "quality": 0.68, + "cost": 0.5, + "speed": 0.19211061827613424, + "reliability": 1 + } + ], + "taskType": "vision", + "routerVersion": "v3-portfolio" + }, + { + "prompt": 8, + "profile": "premium", + "model": "moonshot/kimi-k3", + "tier": "MEDIUM", + "confidence": 0.5, + "method": "portfolio", + "reasoning": "score=-0.07 | short (5 tokens), constraints (不超过) | ambiguous -> default: MEDIUM | premium | v3 task=extraction agentRisk=standard deepWebResearch=false terminalCode=false terminalSafety=false candidates=10", + "costEstimate": 0.0017297000000000002, + "baselineCost": 0.006424999999999999, + "savings": 0, + "agenticScore": 0, + "candidates": [ + "moonshot/kimi-k3", + "openai/gpt-5.3-codex", + "anthropic/claude-sonnet-5", + "zai/glm-5.3", + "google/gemini-3.6-flash", + "google/gemini-3.5-flash", + "google/gemini-2.5-pro", + "xai/grok-4.5", + "anthropic/claude-sonnet-4.6", + "openai/gpt-5.6-terra" + ], + "candidateScores": [ + { + "model": "moonshot/kimi-k3", + "score": 0.9016386868652879, + "quality": 1, + "cost": 0.5, + "speed": 0.02731144775479715, + "reliability": 1 + } + ], + "taskType": "extraction", + "routerVersion": "v3-portfolio" + }, + { + "prompt": 9, + "profile": "premium", + "model": "moonshot/kimi-k3", + "tier": "MEDIUM", + "confidence": 0.5, + "method": "portfolio", + "reasoning": "score=-0.07 | short (16 tokens), imperative (设计), references (代码) | ambiguous -> default: MEDIUM | premium | v3 task=chat agentRisk=standard deepWebResearch=false terminalCode=false terminalSafety=false candidates=10", + "costEstimate": 0.008622400000000002, + "baselineCost": 0.03208, + "savings": 0, + "agenticScore": 0, + "candidates": [ + "moonshot/kimi-k3", + "google/gemini-3.5-flash", + "openai/gpt-5.3-codex", + "anthropic/claude-sonnet-5", + "zai/glm-5.3", + "google/gemini-3.6-flash", + "google/gemini-2.5-pro", + "xai/grok-4.5", + "anthropic/claude-sonnet-4.6", + "openai/gpt-5.6-terra" + ], + "candidateScores": [ + { + "model": "moonshot/kimi-k3", + "score": 0.8604386868652878, + "quality": 0.86, + "cost": 1, + "speed": 0.02731144775479715, + "reliability": 1 + }, + { + "model": "google/gemini-3.5-flash", + "score": 0.770326637096568, + "quality": 0.86, + "cost": 0, + "speed": 0.19211061827613424, + "reliability": 1 + } + ], + "taskType": "chat", + "routerVersion": "v3-portfolio" + }, + { + "prompt": 10, + "profile": "premium", + "model": "openai/gpt-5.3-codex", + "tier": "MEDIUM", + "confidence": 0.5, + "method": "portfolio", + "reasoning": "score=-0.02 | short (23 tokens), code (def), simple (define), agentic-light (debug) | ambiguous -> default: MEDIUM | premium | v3 task=debug agentRisk=standard deepWebResearch=false terminalCode=false terminalSafety=false candidates=10", + "costEstimate": 0.0057945, + "baselineCost": 0.057714999999999995, + "savings": 0, + "agenticScore": 0.2, + "candidates": [ + "openai/gpt-5.3-codex", + "anthropic/claude-sonnet-5", + "moonshot/kimi-k3", + "zai/glm-5.3", + "google/gemini-3.6-flash", + "google/gemini-3.5-flash", + "google/gemini-2.5-pro", + "xai/grok-4.5", + "anthropic/claude-sonnet-4.6", + "openai/gpt-5.6-terra" + ], + "candidateScores": [ + { + "model": "openai/gpt-5.3-codex", + "score": 0.8903822492293204, + "quality": 1, + "cost": 0.5, + "speed": 0.039714153822005965, + "reliability": 0.79999 + } + ], + "taskType": "debug", + "routerVersion": "v3-portfolio" + }, + { + "prompt": 11, + "profile": "premium", + "model": "anthropic/claude-sonnet-5", + "tier": "MEDIUM", + "confidence": 0.7373034537835593, + "method": "portfolio", + "reasoning": "score=0.09 | long (1217 tokens), references (following) | premium | v3 task=tool_agent agentRisk=standard deepWebResearch=false terminalCode=false terminalSafety=false candidates=15", + "costEstimate": 0.0084417, + "baselineCost": 0.089285, + "savings": 0, + "agenticScore": 0, + "candidates": [ + "anthropic/claude-sonnet-5", + "openai/gpt-5.3-codex", + "moonshot/kimi-k3", + "zai/glm-5.3", + "google/gemini-3.6-flash", + "google/gemini-3.5-flash", + "google/gemini-2.5-pro", + "xai/grok-4.5", + "anthropic/claude-sonnet-4.6", + "openai/gpt-5.6-terra", + "anthropic/claude-opus-5", + "openai/gpt-5-mini", + "openai/gpt-4.1", + "openai/gpt-4o-mini", + "deepseek/deepseek-v4-pro" + ], + "candidateScores": [ + { + "model": "anthropic/claude-sonnet-5", + "score": 0.9059298386038215, + "quality": 1, + "cost": 0.5, + "speed": 0.09883064339702209, + "reliability": 1 + } + ], + "taskType": "tool_agent", + "routerVersion": "v3-portfolio" + }, + { + "prompt": 12, + "profile": "premium", + "model": "anthropic/claude-sonnet-4.6", + "tier": "MEDIUM", + "confidence": 0.5, + "method": "portfolio", + "reasoning": "score=0.06 | short (32 tokens), code (let), multi-step, references (following) | ambiguous -> default: MEDIUM | premium | v3 task=vision agentRisk=standard deepWebResearch=false terminalCode=false terminalSafety=false candidates=8", + "costEstimate": 0.0017856000000000003, + "baselineCost": 0.00656, + "savings": 0, + "agenticScore": 0, + "candidates": [ + "anthropic/claude-sonnet-4.6", + "moonshot/kimi-k3", + "anthropic/claude-sonnet-5", + "google/gemini-3.6-flash", + "google/gemini-3.5-flash", + "google/gemini-2.5-pro", + "xai/grok-4.5", + "openai/gpt-5.6-terra" + ], + "candidateScores": [ + { + "model": "anthropic/claude-sonnet-4.6", + "score": 0.8675954266748616, + "quality": 0.9, + "cost": 1, + "speed": 0.09325711124769513, + "reliability": 1 + }, + { + "model": "moonshot/kimi-k3", + "score": 0.8036386868652878, + "quality": 0.9, + "cost": 0, + "speed": 0.02731144775479715, + "reliability": 1 + } + ], + "taskType": "vision", + "routerVersion": "v3-portfolio" + }, + { + "prompt": 13, + "profile": "premium", + "model": "google/gemini-3.5-flash", + "tier": "REASONING", + "confidence": 0.5, + "method": "portfolio", + "reasoning": "score=-0.02 | short (13 tokens), multi-step | ambiguous -> default: MEDIUM | premium | v3 task=reasoning_math agentRisk=standard deepWebResearch=false terminalCode=false terminalSafety=false candidates=12", + "costEstimate": 0.008622800000000002, + "baselineCost": 0.032065, + "savings": 0, + "agenticScore": 0, + "candidates": [ + "google/gemini-3.5-flash", + "anthropic/claude-sonnet-5", + "anthropic/claude-sonnet-4.6", + "anthropic/claude-opus-5", + "anthropic/claude-opus-4.8", + "anthropic/claude-opus-4.7", + "xai/grok-4.5", + "deepseek/deepseek-v4-pro", + "xai/grok-4.3", + "openai/o4-mini", + "openai/o3", + "moonshot/kimi-k3" + ], + "candidateScores": [ + { + "model": "google/gemini-3.5-flash", + "score": 0.9115266370965682, + "quality": 1, + "cost": 0.5, + "speed": 0.19211061827613424, + "reliability": 1 + } + ], + "taskType": "reasoning_math", + "routerVersion": "v3-portfolio" + }, + { + "prompt": 14, + "profile": "premium", + "model": "anthropic/claude-sonnet-5", + "tier": "MEDIUM", + "confidence": 0.5, + "method": "portfolio", + "reasoning": "score=-0.07 | short (25 tokens), format (json) | ambiguous -> default: MEDIUM | premium | v3 task=extraction agentRisk=standard deepWebResearch=false terminalCode=false terminalSafety=false candidates=10", + "costEstimate": 0.0057625, + "baselineCost": 0.057725, + "savings": 0, + "agenticScore": 0, + "candidates": [ + "anthropic/claude-sonnet-5", + "google/gemini-3.5-flash", + "moonshot/kimi-k3", + "anthropic/claude-sonnet-4.6", + "openai/gpt-5.3-codex", + "zai/glm-5.3", + "google/gemini-3.6-flash", + "google/gemini-2.5-pro", + "xai/grok-4.5", + "openai/gpt-5.6-terra" + ], + "candidateScores": [ + { + "model": "anthropic/claude-sonnet-5", + "score": 0.8879298386038212, + "quality": 0.9, + "cost": 1, + "speed": 0.09883064339702209, + "reliability": 1 + }, + { + "model": "google/gemini-3.5-flash", + "score": 0.8001933037632348, + "quality": 0.9, + "cost": 0, + "speed": 0.19211061827613424, + "reliability": 1 + }, + { + "model": "moonshot/kimi-k3", + "score": 0.7971153996523453, + "quality": 0.9, + "cost": 0.0017922431715533538, + "speed": 0.02731144775479715, + "reliability": 1 + }, + { + "model": "anthropic/claude-sonnet-4.6", + "score": 0.7878821855823102, + "quality": 0.9, + "cost": 0.0035844863431069296, + "speed": 0.09325711124769513, + "reliability": 1 + } + ], + "taskType": "extraction", + "routerVersion": "v3-portfolio" + }, + { + "prompt": 15, + "profile": "premium", + "model": "moonshot/kimi-k3", + "tier": "MEDIUM", + "confidence": 0.5, + "method": "portfolio", + "reasoning": "score=-0.06 | short (29 tokens), imperative (build), agentic-light (pip) | ambiguous -> default: MEDIUM | premium | v3 task=chat agentRisk=standard deepWebResearch=false terminalCode=false terminalSafety=false candidates=10", + "costEstimate": 0.0223817, + "baselineCost": 0.083345, + "savings": 0, + "agenticScore": 0.2, + "candidates": [ + "moonshot/kimi-k3", + "google/gemini-3.5-flash", + "openai/gpt-5.3-codex", + "anthropic/claude-sonnet-5", + "zai/glm-5.3", + "google/gemini-3.6-flash", + "google/gemini-2.5-pro", + "xai/grok-4.5", + "anthropic/claude-sonnet-4.6", + "openai/gpt-5.6-terra" + ], + "candidateScores": [ + { + "model": "moonshot/kimi-k3", + "score": 0.8604386868652878, + "quality": 0.86, + "cost": 1, + "speed": 0.02731144775479715, + "reliability": 1 + }, + { + "model": "google/gemini-3.5-flash", + "score": 0.770326637096568, + "quality": 0.86, + "cost": 0, + "speed": 0.19211061827613424, + "reliability": 1 + } + ], + "taskType": "chat", + "routerVersion": "v3-portfolio" + }, + { + "prompt": 16, + "profile": "premium", + "model": "anthropic/claude-sonnet-5", + "tier": "MEDIUM", + "confidence": 0.5, + "method": "portfolio", + "reasoning": "score=-0.06 | short (38 tokens), imperative (deploy), references (the api), agentic-light (check the, deploy) | ambiguous -> default: MEDIUM | premium | v3 task=tool_agent agentRisk=standard deepWebResearch=false terminalCode=false terminalSafety=false candidates=15", + "costEstimate": 0.0006437999999999999, + "baselineCost": 0.0065899999999999995, + "savings": 0, + "agenticScore": 0.2, + "candidates": [ + "anthropic/claude-sonnet-5", + "openai/gpt-5.3-codex", + "moonshot/kimi-k3", + "zai/glm-5.3", + "google/gemini-3.6-flash", + "google/gemini-3.5-flash", + "google/gemini-2.5-pro", + "xai/grok-4.5", + "anthropic/claude-sonnet-4.6", + "openai/gpt-5.6-terra", + "anthropic/claude-opus-5", + "openai/gpt-5-mini", + "openai/gpt-4.1", + "openai/gpt-4o-mini", + "deepseek/deepseek-v4-pro" + ], + "candidateScores": [ + { + "model": "anthropic/claude-sonnet-5", + "score": 0.9059298386038215, + "quality": 1, + "cost": 0.5, + "speed": 0.09883064339702209, + "reliability": 1 + } + ], + "taskType": "tool_agent", + "routerVersion": "v3-portfolio" + }, + { + "prompt": 17, + "profile": "premium", + "model": "anthropic/claude-sonnet-4.6", + "tier": "MEDIUM", + "confidence": 0.5, + "method": "portfolio", + "reasoning": "score=-0.06 | short (9 tokens), creative (write a) | ambiguous -> default: MEDIUM | premium | v3 task=vision agentRisk=standard deepWebResearch=false terminalCode=false terminalSafety=false candidates=8", + "costEstimate": 0.008595800000000002, + "baselineCost": 0.032045000000000004, + "savings": 0, + "agenticScore": 0, + "candidates": [ + "anthropic/claude-sonnet-4.6", + "moonshot/kimi-k3", + "anthropic/claude-sonnet-5", + "google/gemini-3.6-flash", + "google/gemini-3.5-flash", + "google/gemini-2.5-pro", + "xai/grok-4.5", + "openai/gpt-5.6-terra" + ], + "candidateScores": [ + { + "model": "anthropic/claude-sonnet-4.6", + "score": 0.8675954266748616, + "quality": 0.9, + "cost": 1, + "speed": 0.09325711124769513, + "reliability": 1 + }, + { + "model": "moonshot/kimi-k3", + "score": 0.8036386868652878, + "quality": 0.9, + "cost": 0, + "speed": 0.02731144775479715, + "reliability": 1 + } + ], + "taskType": "vision", + "routerVersion": "v3-portfolio" + }, + { + "prompt": 18, + "profile": "premium", + "model": "anthropic/claude-sonnet-5", + "tier": "MEDIUM", + "confidence": 0.7685247834990178, + "method": "portfolio", + "reasoning": "score=-0.10 | short (30 tokens), simple (translate) | upgraded to MEDIUM (structured output) | premium | v3 task=extraction agentRisk=standard deepWebResearch=false terminalCode=false terminalSafety=false candidates=10", + "costEstimate": 0.005763000000000001, + "baselineCost": 0.057749999999999996, + "savings": 0, + "agenticScore": 0, + "candidates": [ + "anthropic/claude-sonnet-5", + "google/gemini-3.5-flash", + "anthropic/claude-sonnet-4.6", + "moonshot/kimi-k3", + "openai/gpt-5.3-codex", + "zai/glm-5.3", + "google/gemini-3.6-flash", + "google/gemini-2.5-pro", + "xai/grok-4.5", + "openai/gpt-5.6-terra" + ], + "candidateScores": [ + { + "model": "anthropic/claude-sonnet-5", + "score": 0.9011405003873671, + "quality": 0.9, + "cost": 1, + "speed": 0.1367178599097662, + "reliability": 1 + }, + { + "model": "google/gemini-3.5-flash", + "score": 0.8118719412624161, + "quality": 0.9, + "cost": 0, + "speed": 0.16575196139820988, + "reliability": 1 + }, + { + "model": "anthropic/claude-sonnet-4.6", + "score": 0.8011542380866079, + "quality": 0.9, + "cost": 0.004293688278231067, + "speed": 0.13436245017392473, + "reliability": 1 + }, + { + "model": "moonshot/kimi-k3", + "score": 0.7986265738299553, + "quality": 0.9, + "cost": 0.002146844139115478, + "speed": 0.022296378324947415, + "reliability": 1 + } + ], + "taskType": "extraction", + "routerVersion": "v3-portfolio" + }, + { + "prompt": 19, + "profile": "premium", + "model": "google/gemini-3.5-flash", + "tier": "REASONING", + "confidence": 0.5, + "method": "portfolio", + "reasoning": "score=-0.07 | short (33 tokens), constraints (budget, budget) | ambiguous -> default: MEDIUM | premium | v3 task=reasoning_math agentRisk=standard deepWebResearch=false terminalCode=false terminalSafety=false candidates=12", + "costEstimate": 0.0224164, + "baselineCost": 0.083365, + "savings": 0, + "agenticScore": 0, + "candidates": [ + "google/gemini-3.5-flash", + "anthropic/claude-sonnet-5", + "anthropic/claude-sonnet-4.6", + "anthropic/claude-opus-5", + "anthropic/claude-opus-4.8", + "anthropic/claude-opus-4.7", + "xai/grok-4.5", + "deepseek/deepseek-v4-pro", + "xai/grok-4.3", + "openai/o4-mini", + "openai/o3", + "moonshot/kimi-k3" + ], + "candidateScores": [ + { + "model": "google/gemini-3.5-flash", + "score": 0.9115266370965682, + "quality": 1, + "cost": 0.5, + "speed": 0.19211061827613424, + "reliability": 1 + } + ], + "taskType": "reasoning_math", + "routerVersion": "v3-portfolio" + }, + { + "prompt": 20, + "profile": "premium", + "model": "openai/gpt-5.3-codex", + "tier": "MEDIUM", + "confidence": 0.5, + "method": "portfolio", + "reasoning": "score=0.01 | imperative (design) | ambiguous -> default: MEDIUM | premium | v3 task=debug agentRisk=standard deepWebResearch=false terminalCode=false terminalSafety=false candidates=10", + "costEstimate": 0.0007195, + "baselineCost": 0.006664999999999999, + "savings": 0, + "agenticScore": 0, + "candidates": [ + "openai/gpt-5.3-codex", + "anthropic/claude-sonnet-5", + "moonshot/kimi-k3", + "zai/glm-5.3", + "google/gemini-3.6-flash", + "google/gemini-3.5-flash", + "google/gemini-2.5-pro", + "xai/grok-4.5", + "anthropic/claude-sonnet-4.6", + "openai/gpt-5.6-terra" + ], + "candidateScores": [ + { + "model": "openai/gpt-5.3-codex", + "score": 0.8903822492293204, + "quality": 1, + "cost": 0.5, + "speed": 0.039714153822005965, + "reliability": 0.79999 + } + ], + "taskType": "debug", + "routerVersion": "v3-portfolio" + }, + { + "prompt": 21, + "profile": "premium", + "model": "anthropic/claude-sonnet-5", + "tier": "MEDIUM", + "confidence": 0.7231218051243898, + "method": "portfolio", + "reasoning": "score=0.08 | long (14423 tokens) | premium | v3 task=tool_agent agentRisk=standard deepWebResearch=false terminalCode=false terminalSafety=false candidates=15", + "costEstimate": 0.0046423, + "baselineCost": 0.104115, + "savings": 0, + "agenticScore": 0, + "candidates": [ + "anthropic/claude-sonnet-5", + "openai/gpt-5.3-codex", + "moonshot/kimi-k3", + "zai/glm-5.3", + "google/gemini-3.6-flash", + "google/gemini-3.5-flash", + "google/gemini-2.5-pro", + "xai/grok-4.5", + "anthropic/claude-sonnet-4.6", + "openai/gpt-5.6-terra", + "anthropic/claude-opus-5", + "openai/gpt-5-mini", + "openai/gpt-4.1", + "openai/gpt-4o-mini", + "deepseek/deepseek-v4-pro" + ], + "candidateScores": [ + { + "model": "anthropic/claude-sonnet-5", + "score": 0.9059298386038215, + "quality": 1, + "cost": 0.5, + "speed": 0.09883064339702209, + "reliability": 1 + } + ], + "taskType": "tool_agent", + "routerVersion": "v3-portfolio" + } +] diff --git a/tests/unit/test_account_rail_gaps.py b/tests/unit/test_account_rail_gaps.py new file mode 100644 index 0000000..8175e31 --- /dev/null +++ b/tests/unit/test_account_rail_gaps.py @@ -0,0 +1,146 @@ +"""The account rail's remaining edges: wallet-keyed listings and Solana polling. + +Every case here is a place where an API-key client used to reach code that +assumes a wallet exists. The failures were never wrong answers — they were +`AttributeError: 'NoneType' object has no attribute 'address'`, a 402 reported +as a generic HTTP error, and a poll loop pointed at a URL the gateway answers +with `wrong_host`. All three look like SDK bugs to the caller, which is exactly +what the account rail is not supposed to feel like. +""" + +from __future__ import annotations + +import httpx +import pytest + +from blockrun_llm import PortraitClient, RealFaceClient, solana_client +from blockrun_llm.apikey import DEFAULT_API_KEY_URL +from blockrun_llm.types import APIError, PaymentError + +KEY = "brk_live_account_rail_fixture" +WALLET = "0x" + "01" * 40 +# Throwaway base58 keypair (seed = bytes(range(32))), only ever used to reach +# the wallet-rail branch of _absolute_url. Never funded, never signs anything. +_SOLANA_KEY = ( + "1GMkH3brNXiNNs1tiFZHu4yZSRrzJwxi5wB9bHFtMikjwpAW9DMZzU2Pqakc5it8X3N5vPmqdN7KF4CCUpmKhq" +) + + +def _mock(client: PortraitClient | RealFaceClient, status: int) -> None: + client._client.close() + client._client = httpx.Client( + headers=client._client.headers, + transport=httpx.MockTransport(lambda r: httpx.Response(status, json={"error": "fixture"})), + ) + + +@pytest.mark.parametrize( + "cls,method", + [(PortraitClient, "list_portraits"), (RealFaceClient, "list_realfaces")], +) +class TestWalletKeyedListings: + """These endpoints are keyed by wallet address, which the account rail has + none of. Both classes have to say so the same way — the pair drifted once + already, with only one of them growing the 402 handling.""" + + def test_no_address_argument_names_the_helper(self, monkeypatch, cls, method): + monkeypatch.setenv("BLOCKRUN_API_KEY", KEY) + client = cls() + try: + with pytest.raises(ValueError, match=f"{method}\\(\\) is wallet-only"): + getattr(client, method)() + finally: + client.close() + + def test_402_is_a_credit_refusal_not_a_generic_http_error(self, monkeypatch, cls, method): + monkeypatch.setenv("BLOCKRUN_API_KEY", KEY) + client = cls() + _mock(client, 402) + try: + with pytest.raises(PaymentError, match="no credit left"): + getattr(client, method)(WALLET) + finally: + client.close() + + def test_other_failures_stay_api_errors(self, monkeypatch, cls, method): + monkeypatch.setenv("BLOCKRUN_API_KEY", KEY) + client = cls() + _mock(client, 500) + try: + with pytest.raises(APIError) as failure: + getattr(client, method)(WALLET) + assert failure.value.status_code == 500 + finally: + client.close() + + def test_wallet_rail_still_defaults_to_its_own_address(self, monkeypatch, cls, method): + monkeypatch.delenv("BLOCKRUN_API_KEY", raising=False) + seen = [] + client = cls(private_key="0x" + "ac" * 32) + client._client.close() + + def handler(request): + seen.append(str(request.url)) + return httpx.Response(500, json={"error": "fixture"}) + + client._client = httpx.Client(transport=httpx.MockTransport(handler)) + try: + with pytest.raises(APIError): + getattr(client, method)() + assert client.account.address in seen[0] + finally: + client.close() + + +@pytest.mark.parametrize( + "client_class", [solana_client.SolanaLLMClient, solana_client.AsyncSolanaLLMClient] +) +class TestSolanaAccountPolling: + def test_poll_url_drops_the_gateway_api_prefix(self, monkeypatch, client_class): + """The gateway mints /api/v1/... against its own host. api.blockrun.ai + serves that route at /v1/... and answers /api/v1/... with wrong_host, so + leaving the prefix on makes every slow media job poll a dead URL until + its budget runs out.""" + monkeypatch.setenv("BLOCKRUN_API_KEY", KEY) + client = client_class(private_key=KEY) + resolved = client._absolute_url("/api/v1/images/generations/job_1") + assert resolved == f"{DEFAULT_API_KEY_URL}/v1/images/generations/job_1" + + def test_poll_url_refuses_a_foreign_origin(self, monkeypatch, client_class): + """The Authorization header rides on the client's default headers, so a + gateway response pointing the poll loop elsewhere would hand the key to + that host.""" + monkeypatch.setenv("BLOCKRUN_API_KEY", KEY) + client = client_class(private_key=KEY) + for hostile in ("https://elsewhere.example/x", "//elsewhere.example/x"): + with pytest.raises(ValueError, match="origin"): + client._absolute_url(hostile) + + def test_wallet_rail_poll_url_is_unchanged(self, monkeypatch, client_class): + monkeypatch.delenv("BLOCKRUN_API_KEY", raising=False) + pytest.importorskip("x402") + client = client_class(private_key=_SOLANA_KEY) + assert ( + client._absolute_url("/api/v1/images/generations/job_1") + == "https://sol.blockrun.ai/api/v1/images/generations/job_1" + ) + + +def test_solana_account_402_on_the_media_probe_is_a_credit_refusal(monkeypatch): + """Before the guard the probe fell into the x402 branch, which on this rail + has no signer to reach for and — without the optional SDK installed — no + decoder either.""" + monkeypatch.setenv("BLOCKRUN_API_KEY", KEY) + client = solana_client.SolanaLLMClient(private_key=KEY) + client._client.close() + client._client = httpx.Client( + headers={"Authorization": f"Bearer {KEY}"}, + transport=httpx.MockTransport( + lambda r: httpx.Response(402, json={"error": "insufficient_credit"}) + ), + ) + try: + with pytest.raises(PaymentError, match="no credit left"): + client.image("a cat") + finally: + client.close() diff --git a/tests/unit/test_account_service_contracts.py b/tests/unit/test_account_service_contracts.py new file mode 100644 index 0000000..c20989e --- /dev/null +++ b/tests/unit/test_account_service_contracts.py @@ -0,0 +1,172 @@ +"""Public service entrypoints against an in-memory account gateway, never production.""" + +import json +from unittest.mock import patch + +import httpx +import pytest + +from blockrun_llm import ( + ImageClient, + MusicClient, + PhoneClient, + PortraitClient, + PriceClient, + RealFaceClient, + RpcClient, + SearchClient, + SpeechClient, + SurfClient, + VideoClient, + VoiceClient, +) +from blockrun_llm.types import APIError, PaymentError + +KEY = "brk_live_contract_fixture" +PHONE = "+12025550123" +# Each public method must reach its documented endpoint with account auth, +# preserve account failures, and never reinterpret a 402 as a wallet challenge. +CASES = [ + (ImageClient, "generate", ("test",), {}, "POST", "/v1/images/generations", {"prompt": "test"}), + ( + VideoClient, + "generate", + ("test",), + {"duration_seconds": 5}, + "POST", + "/v1/videos/generations", + {"duration_seconds": 5}, + ), + ( + MusicClient, + "generate", + ("test music",), + {}, + "POST", + "/v1/audio/generations", + {"prompt": "test music"}, + ), + (SpeechClient, "generate", ("hello",), {}, "POST", "/v1/audio/speech", {"input": "hello"}), + (SpeechClient, "sound_effect", ("rain",), {}, "POST", "/v1/audio/sound-effects", {}), + (SpeechClient, "list_voices", (), {}, "GET", "/v1/audio/voices", {}), + ( + VoiceClient, + "call", + (PHONE, "Read a test message"), + {}, + "POST", + "/v1/voice/call", + {"to": PHONE}, + ), + (VoiceClient, "get_status", ("fixture-call",), {}, "GET", "/v1/voice/call/fixture-call", {}), + (PhoneClient, "lookup", (PHONE,), {}, "POST", "/v1/phone/lookup", {"phoneNumber": PHONE}), + (PhoneClient, "lookup_fraud", (PHONE,), {}, "POST", "/v1/phone/lookup/fraud", {}), + ( + PhoneClient, + "buy_number", + (), + {"area_code": "202"}, + "POST", + "/v1/phone/numbers/buy", + {"areaCode": "202"}, + ), + (PhoneClient, "renew_number", (PHONE,), {}, "POST", "/v1/phone/numbers/renew", {}), + (PhoneClient, "list_numbers", (), {}, "POST", "/v1/phone/numbers/list", {}), + (PhoneClient, "release_number", (PHONE,), {}, "POST", "/v1/phone/numbers/release", {}), + ( + PortraitClient, + "enroll", + ("fixture", "https://example.com/test.png"), + {}, + "POST", + "/v1/portrait/enroll", + {}, + ), + (RealFaceClient, "init", ("fixture",), {}, "POST", "/v1/realface/init", {}), + (RealFaceClient, "status", ("legacy_rf_123",), {}, "GET", "/v1/realface/status", {}), + ( + RealFaceClient, + "enroll", + ("fixture", "https://example.com/test.png", "legacy_rf_123"), + {}, + "POST", + "/v1/realface/enroll", + {"group_id": "legacy_rf_123"}, + ), + (SearchClient, "search", ("test",), {}, "POST", "/v1/search", {"query": "test"}), + (SurfClient, "get", ("market/ranking",), {}, "GET", "/v1/surf/market/ranking", {}), + ( + SurfClient, + "post", + ("onchain/sql", {"query": "SELECT 1"}), + {}, + "POST", + "/v1/surf/onchain/sql", + {"query": "SELECT 1"}, + ), + (PriceClient, "price", ("crypto", "BTC-USD"), {}, "GET", "/v1/crypto/price/BTC-USD", {}), + (PriceClient, "price", ("fx", "EUR-USD"), {}, "GET", "/v1/fx/price/EUR-USD", {}), + (PriceClient, "price", ("commodity", "XAU-USD"), {}, "GET", "/v1/commodity/price/XAU-USD", {}), + ( + PriceClient, + "price", + ("stocks", "AAPL"), + {"market": "us"}, + "GET", + "/v1/stocks/us/price/AAPL", + {}, + ), + ( + PriceClient, + "history", + ("crypto", "BTC-USD"), + {"from_ts": 1, "to_ts": 2}, + "GET", + "/v1/crypto/history/BTC-USD", + {}, + ), + (RpcClient, "call", ("solana", "getSlot"), {}, "POST", "/v1/rpc/solana", {"method": "getSlot"}), + (RpcClient, "batch", ("base", [{"method": "eth_blockNumber"}]), {}, "POST", "/v1/rpc/base", {}), +] + + +@pytest.mark.parametrize("case", CASES, ids=[c[0].__name__ + "." + c[1] + c[5] for c in CASES]) +@pytest.mark.parametrize("status", [401, 402, 429]) +def test_public_service_account_error_contract(case, status, monkeypatch): + cls, method, args, kwargs, verb, path, expected_body = case + monkeypatch.setenv("BLOCKRUN_API_KEY", KEY) + monkeypatch.delenv("BLOCKRUN_API_BASE_URL", raising=False) + monkeypatch.setenv("BLOCKRUN_WALLET_KEY", "must-not-read-wallet") + seen = [] + + def handler(request): + seen.append(request) + assert request.url.host == "api.blockrun.ai" + assert request.url.path == path + assert request.method == verb + assert request.headers["authorization"] == f"Bearer {KEY}" + assert "payment-signature" not in request.headers + assert "x-payment" not in request.headers + if expected_body: + body = json.loads(request.content) + assert all(body[k] == v for k, v in expected_body.items()) + return httpx.Response( + status, + json={"error": {"code": "fixture_limit", "message": "fixture"}}, + headers={"retry-after": "17", "payment-required": "must-not-sign"}, + ) + + with patch("blockrun_llm.wallet.load_wallet", side_effect=AssertionError("wallet read")): + client = cls() + client._client.close() + client._client = httpx.Client( + headers=client._client.headers, transport=httpx.MockTransport(handler) + ) + try: + with pytest.raises(PaymentError if status == 402 else APIError) as failure: + getattr(client, method)(*args, **kwargs) + if status != 402: + assert failure.value.status_code == status + assert len(seen) == 1 + finally: + client.close() diff --git a/tests/unit/test_anthropic_account.py b/tests/unit/test_anthropic_account.py new file mode 100644 index 0000000..754db5d --- /dev/null +++ b/tests/unit/test_anthropic_account.py @@ -0,0 +1,129 @@ +"""Exercise the public Anthropic wrapper without a wallet or a real upstream.""" + +import json + +import httpx +import pytest + +from blockrun_llm import AnthropicClient + +pytest.importorskip("anthropic") + +import blockrun_llm.anthropic_client as ac + +KEY = "brk_live_synthetic_customer_test" +WALLET = "0x" + "01" * 32 + + +def test_anthropic_account_never_loads_wallet_or_replays_failed_post(monkeypatch): + monkeypatch.setenv("BLOCKRUN_API_KEY", KEY) + monkeypatch.setenv("BLOCKRUN_WALLET_KEY", "must-not-parse-this") + calls = [] + + def handle(_self, request): + calls.append(request) + return httpx.Response(500, json={"error": {"type": "api_error", "message": "failed"}}) + + monkeypatch.setattr(httpx.HTTPTransport, "handle_request", handle) + client = AnthropicClient() + try: + assert client.payment_mode == "apikey" + import anthropic + + with pytest.raises(anthropic.InternalServerError): + client.messages.create( + model="anthropic/claude-sonnet-4.6", + max_tokens=8, + messages=[{"role": "user", "content": "hi"}], + ) + assert len(calls) == 1 + assert str(calls[0].url) == "https://api.blockrun.ai/v1/messages" + assert calls[0].headers["x-api-key"] == KEY + assert "payment-signature" not in calls[0].headers + finally: + client.close() + + +def test_anthropic_explicit_wallet_overrides_env_key(monkeypatch): + monkeypatch.setenv("BLOCKRUN_API_KEY", KEY) + client = AnthropicClient(private_key=WALLET) + try: + assert client.payment_mode == "wallet" + assert str(client.base_url) == "https://blockrun.ai/api/" + finally: + client.close() + + +@pytest.mark.parametrize("mode", ["wallet", "apikey"]) +def test_no_rail_replays_a_settled_post(monkeypatch, mode): + """One caller request must sign at most one payment. + + The x402 transport signs a *fresh* payment for every 402 it sees, so an SDK + retry after the gateway already settled signs and settles again. At the + Anthropic default of 2 that is three on-chain transfers for one + messages.create(), which is why max_retries has to be pinned on both rails + and not just the account one. + """ + signed = [] + + class Base(httpx.BaseTransport): + def handle_request(self, request): + if request.headers.get("PAYMENT-SIGNATURE") is None and mode == "wallet": + body = {"x402Version": 2, "accepts": [{}]} + return httpx.Response( + 402, json=body, headers={"payment-required": json.dumps(body)} + ) + signed.append(request) + return httpx.Response(500, json={"error": {"type": "api_error", "message": "boom"}}) + + def close(self): + pass + + if mode == "wallet": + monkeypatch.delenv("BLOCKRUN_API_KEY", raising=False) + monkeypatch.setattr(ac, "parse_payment_required", json.loads) + monkeypatch.setattr( + ac, + "extract_payment_details", + lambda p: {"recipient": "0x1", "amount": "1000", "network": "eip155:8453"}, + ) + monkeypatch.setattr(ac, "create_payment_payload", lambda **kw: "signed-payload") + client = AnthropicClient(private_key=WALLET) + else: + monkeypatch.setenv("BLOCKRUN_API_KEY", KEY) + client = AnthropicClient() + + try: + assert client._client.max_retries == 0 + # Swap in the counting transport, keeping the x402 wrapper on the + # wallet rail so a retry would really re-sign. + inner = client._client._client + if mode == "wallet": + inner._transport = ac._BlockRunX402Transport( + account=ac.Account.from_key(WALLET), + api_url="https://blockrun.ai/api", + base_transport=Base(), + ) + else: + inner._transport = Base() + + import anthropic + + with pytest.raises(anthropic.InternalServerError): + client.messages.create( + model="anthropic/claude-sonnet-4.6", + max_tokens=8, + messages=[{"role": "user", "content": "hi"}], + ) + assert len(signed) == 1, f"{mode} rail submitted {len(signed)} payments for one request" + finally: + client.close() + + +def test_max_retries_stays_an_explicit_caller_choice(monkeypatch): + monkeypatch.delenv("BLOCKRUN_API_KEY", raising=False) + client = AnthropicClient(private_key=WALLET, max_retries=3) + try: + assert client._client.max_retries == 3 + finally: + client.close() diff --git a/tests/unit/test_apikey.py b/tests/unit/test_apikey.py new file mode 100644 index 0000000..0de790b --- /dev/null +++ b/tests/unit/test_apikey.py @@ -0,0 +1,316 @@ +"""The API-key rail: precedence, routing, and the things it must refuse. + +The precedence rule gets its own test class because it decides whether a call +spends prepaid credit or on-chain USDC, and that is not a difference anyone +wants to discover from an invoice. +""" + +from __future__ import annotations + +import httpx +import pytest + +from blockrun_llm import LLMClient +from blockrun_llm.apikey import ( + DEFAULT_API_KEY_URL, + ENV_API_KEY, + ENV_API_KEY_URL, + PAYMENT_MODE_API_KEY, + PAYMENT_MODE_WALLET, + api_key_base_url, + auth_headers, + is_api_key, + resolve_api_key, + resolve_poll_url, +) +from blockrun_llm.image import ImageClient +from blockrun_llm.types import PaymentError + +API_KEY = "brk_live_TESTKEYTESTKEYTESTKEY" +WALLET_KEY = "0xac0974bec39a17e36ba4a6b4d238ff944bacb478cbed5efcae784d7bf4f2ff80" +WALLET_ADDRESS = "0xf39Fd6e51aad88F6F4ce6aB8827279cffFb92266" + + +class TestIsAPIKey: + @pytest.mark.parametrize( + "value,expected", + [ + ("brk_live_abc", True), + (" brk_test_abc ", True), + (WALLET_KEY, False), + ("", False), + ("sk-abc", False), + (None, False), + ], + ) + def test_prefix(self, value, expected): + assert is_api_key(value) is expected + + +class TestPrecedence: + def test_explicit_key_wins_over_everything(self, monkeypatch): + monkeypatch.setenv(ENV_API_KEY, "brk_live_fromenv") + assert resolve_api_key(API_KEY) == API_KEY + + def test_explicit_wallet_key_opts_out_of_the_env_key(self, monkeypatch): + """An explicit wallet key is a deliberate choice of the x402 rail.""" + monkeypatch.setenv(ENV_API_KEY, API_KEY) + assert resolve_api_key(WALLET_KEY) is None + + def test_env_key_beats_the_wallet_env_vars(self, monkeypatch): + monkeypatch.setenv(ENV_API_KEY, API_KEY) + monkeypatch.setenv("BLOCKRUN_WALLET_KEY", WALLET_KEY) + assert resolve_api_key(None) == API_KEY + + def test_no_key_anywhere(self): + assert resolve_api_key(None) is None + + def test_a_non_brk_env_value_is_not_a_key(self, monkeypatch): + monkeypatch.setenv(ENV_API_KEY, "not-a-key") + with pytest.raises(ValueError, match="BLOCKRUN_API_KEY"): + resolve_api_key(None) + + +class TestClientConstruction: + def test_api_key_client(self): + client = LLMClient(private_key=API_KEY) + assert client.payment_mode == PAYMENT_MODE_API_KEY + assert client.api_url == DEFAULT_API_KEY_URL + assert client.account is None + assert client.get_wallet_address() == "" + + def test_wallet_client_is_unchanged(self): + """A wallet client must be exactly what it was before this feature.""" + client = LLMClient(private_key=WALLET_KEY) + assert client.payment_mode == PAYMENT_MODE_WALLET + assert client.api_url == LLMClient.DEFAULT_API_URL + assert client.get_wallet_address() == WALLET_ADDRESS + + def test_env_key_beats_wallet_env(self, monkeypatch): + monkeypatch.setenv(ENV_API_KEY, API_KEY) + monkeypatch.setenv("BLOCKRUN_WALLET_KEY", WALLET_KEY) + assert LLMClient().payment_mode == PAYMENT_MODE_API_KEY + + def test_x402_env_url_does_not_retarget_an_api_key_client(self, monkeypatch): + """BLOCKRUN_API_URL names an x402 gateway. Following it would send the + key to a host configured for a different rail.""" + monkeypatch.setenv("BLOCKRUN_API_URL", "https://private-x402.example.com/api") + assert LLMClient(private_key=API_KEY).api_url == DEFAULT_API_KEY_URL + + def test_api_key_url_override(self, monkeypatch): + monkeypatch.setenv(ENV_API_KEY_URL, "https://api.staging.example.com/") + assert api_key_base_url(None) == "https://api.staging.example.com" + + def test_every_request_carries_the_key(self): + client = LLMClient(private_key=API_KEY) + assert client._client.headers.get("authorization") == f"Bearer {API_KEY}" + + def test_wallet_client_sends_no_authorization(self): + client = LLMClient(private_key=WALLET_KEY) + assert "authorization" not in client._client.headers + + +class TestAuthHeaders: + def test_key(self): + assert auth_headers("brk_live_x") == {"Authorization": "Bearer brk_live_x"} + + def test_no_key_is_empty_so_call_sites_can_be_unconditional(self): + assert auth_headers(None) == {} + + +class TestPollURL: + """``poll_url`` is minted by the x402 gateway relative to ITS host, so it + arrives as ``/api/v1/...``. api.blockrun.ai serves that route at ``/v1/...`` + and answers ``/api/v1/...`` with ``wrong_host`` — an unstripped prefix is an + async job polling a 404 until its budget runs out.""" + + def test_account_rail_strips_the_api_prefix(self): + got = resolve_poll_url( + "/api/v1/images/generations/job_1", DEFAULT_API_KEY_URL, "brk_live_x" + ) + assert got == f"{DEFAULT_API_KEY_URL}/v1/images/generations/job_1" + + def test_wallet_rail_keeps_it(self): + got = resolve_poll_url("/api/v1/images/generations/job_1", "https://blockrun.ai/api", None) + assert got == "https://blockrun.ai/api/v1/images/generations/job_1" + + @pytest.mark.parametrize("url", ["https://elsewhere.example/x", "//elsewhere.example/x"]) + def test_account_rejects_foreign_poll_origin(self, url): + with pytest.raises(ValueError, match="origin"): + resolve_poll_url(url, DEFAULT_API_KEY_URL, "brk_x") + + def test_same_origin_signed_url_preserves_query(self): + url = DEFAULT_API_KEY_URL + "/v1/videos/generations/job?token=a%2Fb&signature=x" + assert resolve_poll_url(url, DEFAULT_API_KEY_URL, "brk_x") == url + + +class TestRequests: + """One request, key attached, no 402 round trip.""" + + def test_chat_sends_bearer_and_skips_the_402_dance(self, monkeypatch): + seen: dict = {} + + def handler(request: httpx.Request) -> httpx.Response: + seen["count"] = seen.get("count", 0) + 1 + seen["auth"] = request.headers.get("authorization") + seen["path"] = request.url.path + seen["payment_sig"] = request.headers.get("payment-signature") + return httpx.Response( + 200, + json={ + "id": "x", + "object": "chat.completion", + "created": 1, + "model": "openai/gpt-4o", + "choices": [ + { + "index": 0, + "message": {"role": "assistant", "content": "4"}, + "finish_reason": "stop", + } + ], + }, + ) + + client = LLMClient(private_key=API_KEY) + client._client = httpx.Client( + transport=httpx.MockTransport(handler), headers=auth_headers(API_KEY) + ) + + assert client.chat("openai/gpt-4o", "2+2?") == "4" + assert seen["count"] == 1, "the account rail must not make a 402 round trip" + assert seen["auth"] == f"Bearer {API_KEY}" + # No "/api" inserted: the endpoint constants are already /v1/... + assert seen["path"] == "/v1/chat/completions" + assert seen["payment_sig"] is None + + def test_402_is_a_credit_refusal_not_a_challenge(self): + """The old path would answer 'no wallet configured', which sends the + reader hunting a wallet problem they do not have.""" + + def handler(request: httpx.Request) -> httpx.Response: + return httpx.Response( + 402, + json={"error": {"type": "insufficient_quota", "code": "BALANCE_EXHAUSTED"}}, + ) + + client = LLMClient(private_key=API_KEY) + client._client = httpx.Client( + transport=httpx.MockTransport(handler), headers=auth_headers(API_KEY) + ) + + with pytest.raises(PaymentError) as exc: + client.chat("openai/gpt-4o", "hi") + message = str(exc.value) + assert "user.blockrun.ai" in message, "does not say where to top up" + assert "BALANCE_EXHAUSTED" in message, "drops the gateway's own reason" + assert "no wallet" not in message.lower(), "blames a wallet the caller does not have" + + def test_image_async_202_on_the_first_post(self): + """The wallet rail only ever sees a 202 after the signed retry, so + without the account-rail branch every slow model raised 'API error: 202'.""" + polls = {"n": 0} + + def handler(request: httpx.Request) -> httpx.Response: + assert request.headers.get("authorization") == f"Bearer {API_KEY}" + assert request.headers.get("payment-signature") is None + if request.method == "POST": + return httpx.Response( + 202, + json={ + "id": "img_1", + "status": "queued", + # Minted by the gateway, so it carries the /api prefix. + "poll_url": "/api/v1/images/generations/img_1", + }, + ) + polls["n"] += 1 + if polls["n"] < 2: + return httpx.Response(202, json={"status": "in_progress"}) + return httpx.Response( + 200, + json={ + "status": "completed", + "created": 1, + "data": [{"url": "https://cdn.example/i.png"}], + }, + ) + + client = ImageClient(private_key=API_KEY) + client._client = httpx.Client( + transport=httpx.MockTransport(handler), headers=auth_headers(API_KEY) + ) + client.IMAGE_POLL_INTERVAL_SECONDS = 0.0 + + resp = client.generate("a red cube") + assert resp.data[0].url == "https://cdn.example/i.png" + assert polls["n"] == 2 + + +class TestWalletOnlyHelpers: + """Returning 0 from get_balance would be indistinguishable from an empty + wallet, and an agent gating on it would stop calling a funded account.""" + + def test_get_balance_refuses(self): + client = LLMClient(private_key=API_KEY) + with pytest.raises(ValueError, match="user.blockrun.ai"): + client.get_balance() + + def test_onramp_refuses(self): + client = LLMClient(private_key=API_KEY) + with pytest.raises(ValueError, match="wallet-only"): + client.onramp(WALLET_ADDRESS) + + def test_get_wallet_address_is_empty(self): + assert LLMClient(private_key=API_KEY).get_wallet_address() == "" + + +class TestSetupAgentWallet: + def test_uses_the_key_without_minting_a_wallet(self, monkeypatch, tmp_path): + """A skill calls this unconditionally; with a key configured it must not + write a private key to disk for a wallet that will never sign.""" + monkeypatch.setenv(ENV_API_KEY, API_KEY) + monkeypatch.setenv("HOME", str(tmp_path)) + + from blockrun_llm import setup_agent_wallet + + client = setup_agent_wallet() + assert client.payment_mode == PAYMENT_MODE_API_KEY + assert client.get_wallet_address() == "" + assert not (tmp_path / ".blockrun" / ".session").exists() + + +@pytest.mark.parametrize("bad_key", ["not-a-key", "sk-openai-shaped", "0x" + "ab" * 32]) +def test_invalid_env_never_selects_a_wallet(monkeypatch, bad_key): + monkeypatch.setenv(ENV_API_KEY, bad_key) + monkeypatch.setenv("BLOCKRUN_WALLET_KEY", WALLET_KEY) + with pytest.raises(ValueError, match="BLOCKRUN_API_KEY"): + LLMClient() + # Explicit wallet selection remains available even with a broken env key. + with LLMClient(private_key=WALLET_KEY) as client: + assert client.payment_mode == PAYMENT_MODE_WALLET + + +@pytest.mark.parametrize("blank", ["", " ", "\t\n"]) +def test_blank_env_is_unset_not_invalid(monkeypatch, blank): + """`BLOCKRUN_API_KEY=` is how a .env file, `docker -e VAR` and an + unpopulated CI secret all say "not set". Raising there would break wallet + users who never opted into the account rail at all.""" + monkeypatch.setenv(ENV_API_KEY, blank) + monkeypatch.setenv("BLOCKRUN_WALLET_KEY", WALLET_KEY) + assert resolve_api_key(None) is None + with LLMClient() as client: + assert client.payment_mode == PAYMENT_MODE_WALLET + assert "authorization" not in client._client.headers + + +def test_rotating_env_only_affects_new_clients(monkeypatch): + monkeypatch.setenv(ENV_API_KEY, API_KEY) + with LLMClient() as first: + monkeypatch.setenv(ENV_API_KEY, "brk_test_second") + with LLMClient() as second: + assert first._client.headers["authorization"] == f"Bearer {API_KEY}" + assert second._client.headers["authorization"] == "Bearer brk_test_second" + monkeypatch.delenv(ENV_API_KEY) + with LLMClient(private_key=WALLET_KEY) as wallet: + assert "authorization" not in wallet._client.headers diff --git a/tests/unit/test_client.py b/tests/unit/test_client.py index 696967f..e807609 100644 --- a/tests/unit/test_client.py +++ b/tests/unit/test_client.py @@ -1,180 +1,183 @@ -"""Unit tests for LLMClient.""" - -import pytest -from unittest.mock import Mock, patch -from blockrun_llm import LLMClient, APIError, PaymentError -from ..helpers import ( - TEST_PRIVATE_KEY, - build_chat_response, - build_error_response, - build_models_response, - MockResponse, -) - - -class TestLLMClientInit: - def test_init_with_valid_key(self): - """Should create client with valid private key.""" - client = LLMClient(private_key=TEST_PRIVATE_KEY) - assert client is not None - assert client.get_wallet_address().startswith("0x") - - def test_init_missing_key(self): - """Should raise ValueError when private key is missing.""" - with pytest.raises(ValueError, match="Private key required"): - LLMClient(private_key=None) - - def test_init_invalid_key_format(self): - """Should raise ValueError for invalid key format.""" - with pytest.raises(ValueError, match="must start with 0x"): - LLMClient(private_key="invalid") - - def test_init_short_key(self): - """Should raise ValueError for short key.""" - with pytest.raises(ValueError, match="66 characters"): - LLMClient(private_key="0x123") - - def test_init_non_hex_key(self): - """Should raise ValueError for non-hex key.""" - with pytest.raises(ValueError, match="hexadecimal"): - LLMClient( - private_key="0xGGGG74bec39a17e36ba4a6b4d238ff944bacb478cbed5efcae784d7bf4f2ff80" - ) - - def test_default_api_url(self): - """Should use default API URL.""" - client = LLMClient(private_key=TEST_PRIVATE_KEY) - assert client.api_url == "https://api.blockrun.ai" - - def test_custom_api_url(self): - """Should accept custom API URL.""" - client = LLMClient( - private_key=TEST_PRIVATE_KEY, api_url="https://custom.example.com" - ) - assert client.api_url == "https://custom.example.com" - - def test_invalid_api_url_http(self): - """Should reject HTTP for non-localhost.""" - with pytest.raises(ValueError, match="HTTPS"): - LLMClient(private_key=TEST_PRIVATE_KEY, api_url="http://insecure.com") - - def test_allow_localhost_http(self): - """Should allow HTTP for localhost.""" - client = LLMClient( - private_key=TEST_PRIVATE_KEY, api_url="http://localhost:3000" - ) - assert client.api_url == "http://localhost:3000" - - -class TestLLMClientMethods: - def test_get_wallet_address(self): - """Should return valid Ethereum address.""" - client = LLMClient(private_key=TEST_PRIVATE_KEY) - address = client.get_wallet_address() - - assert address == "0xf39Fd6e51aad88F6F4ce6aB8827279cffFb92266" - assert address.startswith("0x") - assert len(address) == 42 - - @patch("blockrun_llm.client.httpx.Client") - def test_list_models(self, mock_client_class): - """Should list available models.""" - mock_client = Mock() - mock_client_class.return_value = mock_client - - mock_response = MockResponse(200, build_models_response()) - mock_client.get.return_value = mock_response - - client = LLMClient(private_key=TEST_PRIVATE_KEY) - models = client.list_models() - - assert len(models) == 3 - assert models[0]["id"] == "openai/gpt-4o" - assert models[0]["provider"] == "openai" - - @patch("blockrun_llm.client.httpx.Client") - def test_list_models_error(self, mock_client_class): - """Should raise APIError on failure.""" - mock_client = Mock() - mock_client_class.return_value = mock_client - - mock_response = MockResponse(500) - mock_client.get.return_value = mock_response - - client = LLMClient(private_key=TEST_PRIVATE_KEY) - - with pytest.raises(APIError): - client.list_models() - - -class TestErrorSanitization: - @patch("blockrun_llm.client.httpx.Client") - def test_sanitize_error_responses(self, mock_client_class): - """Should sanitize error responses.""" - mock_client = Mock() - mock_client_class.return_value = mock_client - - raw_error = build_error_response( - error="Invalid model", include_sensitive=True - ) - mock_response = MockResponse(400, raw_error) - mock_client.get.return_value = mock_response - - client = LLMClient(private_key=TEST_PRIVATE_KEY) - - try: - client.list_models() - pytest.fail("Should have raised APIError") - except APIError as e: - # Should only contain safe fields - assert e.response == {"message": "Invalid model", "code": "test_error"} - - # Should NOT contain sensitive fields - assert "internal_stack" not in e.response - assert "api_key" not in e.response - assert "database_url" not in e.response - - -class TestInputValidation: - @patch("blockrun_llm.client.httpx.Client") - def test_validate_model_parameter(self, mock_client_class): - """Should validate model parameter.""" - client = LLMClient(private_key=TEST_PRIVATE_KEY) - - with pytest.raises(ValueError, match="non-empty string"): - client.chat_completion("", [{"role": "user", "content": "test"}]) - - @patch("blockrun_llm.client.httpx.Client") - def test_validate_max_tokens(self, mock_client_class): - """Should validate max_tokens parameter.""" - client = LLMClient(private_key=TEST_PRIVATE_KEY) - - with pytest.raises(ValueError, match="positive"): - client.chat_completion( - "gpt-4o", [{"role": "user", "content": "test"}], max_tokens=-1 - ) - - with pytest.raises(ValueError, match="too large"): - client.chat_completion( - "gpt-4o", [{"role": "user", "content": "test"}], max_tokens=200000 - ) - - @patch("blockrun_llm.client.httpx.Client") - def test_validate_temperature(self, mock_client_class): - """Should validate temperature parameter.""" - client = LLMClient(private_key=TEST_PRIVATE_KEY) - - with pytest.raises(ValueError, match="between 0 and 2"): - client.chat_completion( - "gpt-4o", [{"role": "user", "content": "test"}], temperature=3.0 - ) - - @patch("blockrun_llm.client.httpx.Client") - def test_validate_top_p(self, mock_client_class): - """Should validate top_p parameter.""" - client = LLMClient(private_key=TEST_PRIVATE_KEY) - - with pytest.raises(ValueError, match="between 0 and 1"): - client.chat_completion( - "gpt-4o", [{"role": "user", "content": "test"}], top_p=1.5 - ) +"""Unit tests for LLMClient.""" + +from unittest.mock import Mock, patch + +import pytest + +from blockrun_llm import APIError, LLMClient + +from ..helpers import ( + TEST_PRIVATE_KEY, + MockResponse, + build_error_response, + build_models_response, +) + + +class TestLLMClientInit: + def test_init_with_valid_key(self): + """Should create client with valid private key.""" + client = LLMClient(private_key=TEST_PRIVATE_KEY) + assert client is not None + assert client.get_wallet_address().startswith("0x") + + def test_init_missing_key_raises_error(self, monkeypatch, tmp_path): + """Should raise ValueError when no wallet configured.""" + monkeypatch.delenv("BLOCKRUN_WALLET_KEY", raising=False) + monkeypatch.delenv("BASE_CHAIN_WALLET_KEY", raising=False) + # Mock load_wallet to return None (no session file) + monkeypatch.setattr("blockrun_llm.wallet.load_wallet", lambda: None) + # Should raise ValueError with helpful message + with pytest.raises(ValueError, match="No credential configured"): + LLMClient(private_key=None) + + def test_init_invalid_key_format(self): + """Should raise ValueError for invalid key format (after 0x normalization).""" + # "invalid" becomes "0xinvalid" after normalization, which is too short + with pytest.raises(ValueError, match="66 characters"): + LLMClient(private_key="invalid") + + def test_init_short_key(self): + """Should raise ValueError for short key.""" + with pytest.raises(ValueError, match="66 characters"): + LLMClient(private_key="0x123") + + def test_init_non_hex_key(self): + """Should raise ValueError for non-hex key.""" + with pytest.raises(ValueError, match="hexadecimal"): + LLMClient( + private_key="0xGGGG74bec39a17e36ba4a6b4d238ff944bacb478cbed5efcae784d7bf4f2ff80" + ) + + def test_default_api_url(self): + """Should use default API URL.""" + client = LLMClient(private_key=TEST_PRIVATE_KEY) + assert client.api_url == "https://blockrun.ai/api" + + def test_custom_api_url(self): + """Should accept custom API URL.""" + client = LLMClient(private_key=TEST_PRIVATE_KEY, api_url="https://custom.example.com") + assert client.api_url == "https://custom.example.com" + + def test_invalid_api_url_http(self): + """Should reject HTTP for non-localhost.""" + with pytest.raises(ValueError, match="HTTPS"): + LLMClient(private_key=TEST_PRIVATE_KEY, api_url="http://insecure.com") + + def test_allow_localhost_http(self): + """Should allow HTTP for localhost.""" + client = LLMClient(private_key=TEST_PRIVATE_KEY, api_url="http://localhost:3000") + assert client.api_url == "http://localhost:3000" + + +class TestLLMClientMethods: + def test_get_wallet_address(self): + """Should return valid Ethereum address.""" + client = LLMClient(private_key=TEST_PRIVATE_KEY) + address = client.get_wallet_address() + + assert address == "0xf39Fd6e51aad88F6F4ce6aB8827279cffFb92266" + assert address.startswith("0x") + assert len(address) == 42 + + @patch("blockrun_llm.client.httpx.Client") + def test_list_models(self, mock_client_class): + """Should list available models.""" + mock_client = Mock() + mock_client_class.return_value = mock_client + + mock_response = MockResponse(200, build_models_response()) + mock_client.get.return_value = mock_response + + client = LLMClient(private_key=TEST_PRIVATE_KEY) + models = client.list_models() + + assert len(models) == 3 + assert models[0]["id"] == "openai/gpt-5.2" + assert models[0]["provider"] == "openai" + + @patch("blockrun_llm.client.httpx.Client") + def test_list_models_error(self, mock_client_class): + """Should raise APIError on failure.""" + mock_client = Mock() + mock_client_class.return_value = mock_client + + mock_response = MockResponse(500) + mock_client.get.return_value = mock_response + + client = LLMClient(private_key=TEST_PRIVATE_KEY) + + with pytest.raises(APIError): + client.list_models() + + +class TestErrorSanitization: + @patch("blockrun_llm.client.httpx.Client") + def test_sanitize_error_responses(self, mock_client_class): + """Should sanitize error responses.""" + mock_client = Mock() + mock_client_class.return_value = mock_client + + raw_error = build_error_response(error="Invalid model", include_sensitive=True) + mock_response = MockResponse(400, raw_error) + mock_client.get.return_value = mock_response + + client = LLMClient(private_key=TEST_PRIVATE_KEY) + + try: + client.list_models() + pytest.fail("Should have raised APIError") + except APIError as e: + # Should only contain safe fields + assert e.response == {"message": "Invalid model", "code": "test_error"} + + # Should NOT contain sensitive fields + assert "internal_stack" not in e.response + assert "api_key" not in e.response + assert "database_url" not in e.response + + +class TestInputValidation: + @patch("blockrun_llm.client.httpx.Client") + def test_validate_model_parameter(self, mock_client_class): + """Should validate model parameter.""" + client = LLMClient(private_key=TEST_PRIVATE_KEY) + + with pytest.raises(ValueError, match="non-empty string"): + client.chat_completion("", [{"role": "user", "content": "test"}]) + + @patch("blockrun_llm.client.httpx.Client") + def test_validate_max_tokens(self, mock_client_class): + """Should validate max_tokens parameter.""" + client = LLMClient(private_key=TEST_PRIVATE_KEY) + + with pytest.raises(ValueError, match="positive"): + client.chat_completion("gpt-5.2", [{"role": "user", "content": "test"}], max_tokens=-1) + + # 200000 used to be rejected here. It no longer is, and must not be: + # gpt-5.2 advertises 128000 and zai/glm-5.2 serves 262144, so a bound + # below those made the SDK the binding constraint instead of the model. + # Only implausible values fail locally now; real ceilings go to the + # gateway, which rejects with the model's own number. + with pytest.raises(ValueError, match="implausibly large"): + client.chat_completion( + "gpt-5.2", [{"role": "user", "content": "test"}], max_tokens=2_000_000 + ) + + @patch("blockrun_llm.client.httpx.Client") + def test_validate_temperature(self, mock_client_class): + """Should validate temperature parameter.""" + client = LLMClient(private_key=TEST_PRIVATE_KEY) + + with pytest.raises(ValueError, match="between 0 and 2"): + client.chat_completion( + "gpt-5.2", [{"role": "user", "content": "test"}], temperature=3.0 + ) + + @patch("blockrun_llm.client.httpx.Client") + def test_validate_top_p(self, mock_client_class): + """Should validate top_p parameter.""" + client = LLMClient(private_key=TEST_PRIVATE_KEY) + + with pytest.raises(ValueError, match="between 0 and 1"): + client.chat_completion("gpt-5.2", [{"role": "user", "content": "test"}], top_p=1.5) diff --git a/tests/unit/test_cost_log.py b/tests/unit/test_cost_log.py new file mode 100644 index 0000000..40a41bd --- /dev/null +++ b/tests/unit/test_cost_log.py @@ -0,0 +1,195 @@ +"""Unit tests for the cost-log reader / exporter. + +Uses monkeypatch to redirect ``COST_LOG_PATH`` to a temp file so tests don't +touch the real ``~/.blockrun/cost_log.jsonl``. +""" + +from __future__ import annotations + +import json +import time + +import pytest + +from blockrun_llm import cache + + +def _write_log(path, rows): + with open(path, "w") as f: + f.writelines(json.dumps(row) + "\n" for row in rows) + + +@pytest.fixture +def temp_log(tmp_path, monkeypatch): + log = tmp_path / "cost_log.jsonl" + monkeypatch.setattr(cache, "COST_LOG_PATH", log) + return log + + +# --------------------------------------------------------------------------- +# Backwards compatibility +# --------------------------------------------------------------------------- + + +def test_legacy_by_endpoint_alias_still_exposed(temp_log): + """When grouping by endpoint, ``by_endpoint`` is still emitted as a + backwards-compat alias mapping endpoint -> total cost (float).""" + _write_log( + temp_log, + [ + {"ts": time.time(), "endpoint": "/v1/chat/completions", "cost_usd": 0.001}, + {"ts": time.time(), "endpoint": "/v1/chat/completions", "cost_usd": 0.002}, + {"ts": time.time(), "endpoint": "/v1/search", "cost_usd": 0.01}, + ], + ) + summary = cache.get_cost_log_summary() + assert summary["calls"] == 3 + assert summary["total_usd"] == pytest.approx(0.013) + # New shape always includes total_usd / calls / groups + by_endpoint alias + assert "groups" in summary + assert summary["by_endpoint"]["/v1/chat/completions"] == pytest.approx(0.003) + assert summary["by_endpoint"]["/v1/search"] == pytest.approx(0.01) + + +def test_legacy_3_field_rows_still_readable(temp_log): + """Older entries with only ``{ts, endpoint, cost_usd}`` must aggregate + cleanly alongside new entries that carry the full metadata.""" + now = time.time() + _write_log( + temp_log, + [ + {"ts": now, "endpoint": "/v1/chat/completions", "cost_usd": 0.001}, # old + { + "ts": now, + "endpoint": "/v1/chat/completions", + "cost_usd": 0.002, + "model": "openai/gpt-5.2", + "wallet": "0xabc", + "network": "base-mainnet", + "client_kind": "LLMClient", + }, + ], + ) + summary = cache.get_cost_log_summary(group_by="model") + # New schema: legacy row groups under "unknown" model, new row under id. + assert summary["total_usd"] == pytest.approx(0.003) + assert summary["calls"] == 2 + assert summary["groups"]["openai/gpt-5.2"]["cost_usd"] == pytest.approx(0.002) + assert summary["groups"]["unknown"]["cost_usd"] == pytest.approx(0.001) + + +# --------------------------------------------------------------------------- +# Filters + grouping +# --------------------------------------------------------------------------- + + +def test_group_by_model_aggregates_correctly(temp_log): + now = time.time() + _write_log( + temp_log, + [ + {"ts": now, "endpoint": "/v1/chat/completions", "cost_usd": 0.001, "model": "a"}, + {"ts": now, "endpoint": "/v1/chat/completions", "cost_usd": 0.002, "model": "a"}, + {"ts": now, "endpoint": "/v1/chat/completions", "cost_usd": 0.005, "model": "b"}, + ], + ) + summary = cache.get_cost_log_summary(group_by="model") + assert summary["groups"]["a"] == {"calls": 2, "cost_usd": pytest.approx(0.003)} + assert summary["groups"]["b"] == {"calls": 1, "cost_usd": pytest.approx(0.005)} + + +def test_wallet_filter_isolates_to_one_wallet(temp_log): + now = time.time() + _write_log( + temp_log, + [ + {"ts": now, "endpoint": "/v1/x", "cost_usd": 0.01, "wallet": "0xa"}, + {"ts": now, "endpoint": "/v1/x", "cost_usd": 0.02, "wallet": "0xb"}, + {"ts": now, "endpoint": "/v1/x", "cost_usd": 0.04, "wallet": "0xa"}, + ], + ) + summary = cache.get_cost_log_summary(wallet="0xa") + assert summary["calls"] == 2 + assert summary["total_usd"] == pytest.approx(0.05) + + +def test_date_range_filters_correctly(temp_log): + """``YYYY-MM-DD`` strings anchor to UTC midnight; pass distinct + from/to dates to bracket a window.""" + from datetime import datetime, timezone + + # Pick a base timestamp at UTC noon on a known date so the entries are + # clearly inside / outside the window. + base = datetime(2026, 5, 9, 12, 0, 0, tzinfo=timezone.utc).timestamp() + _write_log( + temp_log, + [ + {"ts": base - 86_400, "endpoint": "/x", "cost_usd": 0.01}, # 2026-05-08 + {"ts": base, "endpoint": "/x", "cost_usd": 0.02}, # 2026-05-09 12:00 + {"ts": base + 86_400, "endpoint": "/x", "cost_usd": 0.04}, # 2026-05-10 + ], + ) + summary = cache.get_cost_log_summary(from_date="2026-05-09", to_date="2026-05-10") + # Window is [2026-05-09 00:00 UTC, 2026-05-10 00:00 UTC] — only the + # 2026-05-09 noon entry should be inside. + assert summary["calls"] == 1 + assert summary["total_usd"] == pytest.approx(0.02) + + +def test_invalid_group_by_raises(temp_log): + _write_log(temp_log, []) + with pytest.raises(ValueError): + cache.get_cost_log_summary(group_by="bogus") + + +# --------------------------------------------------------------------------- +# Exports +# --------------------------------------------------------------------------- + + +def test_export_csv_has_header_and_rows(temp_log): + now = time.time() + _write_log( + temp_log, + [ + { + "ts": now, + "endpoint": "/v1/chat/completions", + "cost_usd": 0.001, + "model": "openai/gpt-5.2", + "wallet": "0xabc", + "network": "base-mainnet", + "client_kind": "LLMClient", + }, + ], + ) + csv_text = cache.export_cost_log_csv() + lines = csv_text.strip().split("\n") + assert lines[0].startswith("ts_iso,endpoint,model,wallet,network,client_kind,cost_usd") + assert "openai/gpt-5.2" in lines[1] + assert "base-mainnet" in lines[1] + + +def test_export_json_returns_list_of_dicts(temp_log): + now = time.time() + _write_log( + temp_log, + [ + {"ts": now, "endpoint": "/x", "cost_usd": 0.01, "model": "m"}, + ], + ) + payload = json.loads(cache.export_cost_log_json()) + assert isinstance(payload, list) + assert len(payload) == 1 + assert payload[0]["model"] == "m" + assert payload[0]["cost_usd"] == pytest.approx(0.01) + assert "ts_iso" in payload[0] + + +def test_export_csv_writes_to_path(temp_log, tmp_path): + now = time.time() + _write_log(temp_log, [{"ts": now, "endpoint": "/x", "cost_usd": 0.001}]) + out = tmp_path / "out.csv" + cache.export_cost_log_csv(out) + assert out.exists() + assert "ts_iso" in out.read_text() diff --git a/tests/unit/test_credential_host_routing.py b/tests/unit/test_credential_host_routing.py new file mode 100644 index 0000000..0bb59fe --- /dev/null +++ b/tests/unit/test_credential_host_routing.py @@ -0,0 +1,145 @@ +"""One rule, checked on every client: the credential decides the host. + + API key -> api.blockrun.ai (prepaid credit, bearer auth) + Solana key -> sol.blockrun.ai (x402 on SVM) + Base key -> blockrun.ai (x402 on EVM) + +Sending a credential to the wrong front door is not a 404 — an API key handed +to an x402 host is a key disclosed to a host that never needed it, and a wallet +pointed at the account rail signs nothing and gets a bearer-auth rejection. +The table is parametrized over every exported client so a new one cannot quietly +skip the rule. +""" + +from __future__ import annotations + +import pytest + +from blockrun_llm import ( + AnthropicClient, + AsyncLLMClient, + AsyncSolanaLLMClient, + ImageClient, + LLMClient, + MusicClient, + PhoneClient, + PortraitClient, + PriceClient, + RealFaceClient, + RpcClient, + SearchClient, + SolanaLLMClient, + SpeechClient, + SurfClient, + VideoClient, + VoiceClient, +) +from blockrun_llm.apikey import DEFAULT_API_KEY_URL, ENV_API_KEY, ENV_API_KEY_URL + +API_KEY = "brk_live_host_routing_fixture" +BASE_KEY = "0x" + "ac" * 32 +# Throwaway base58 keypair (seed = bytes(range(32))). Never funded. +SOLANA_KEY = ( + "1GMkH3brNXiNNs1tiFZHu4yZSRrzJwxi5wB9bHFtMikjwpAW9DMZzU2Pqakc5it8X3N5vPmqdN7KF4CCUpmKhq" +) + +SOLANA_CLIENTS = [SolanaLLMClient, AsyncSolanaLLMClient] +BASE_CLIENTS = [ + LLMClient, + AsyncLLMClient, + ImageClient, + VideoClient, + MusicClient, + SpeechClient, + VoiceClient, + PhoneClient, + PortraitClient, + RealFaceClient, + RpcClient, + SearchClient, + SurfClient, + PriceClient, + AnthropicClient, +] +ALL_CLIENTS = BASE_CLIENTS + SOLANA_CLIENTS + + +def host_of(client) -> str: + for attr in ("api_url", "_api_url"): + value = getattr(client, attr, None) + if value: + return str(value).rstrip("/") + raise AssertionError(f"{type(client).__name__} exposes no resolved host") + + +def build(cls, credential, **kwargs): + if cls is AnthropicClient: + pytest.importorskip("anthropic") + if cls in SOLANA_CLIENTS and credential is BASE_KEY: + pytest.skip("Solana clients take a Solana key, not a Base one") + return cls(credential, **kwargs) + + +@pytest.fixture(autouse=True) +def _clean_env(monkeypatch): + for var in ( + ENV_API_KEY, + ENV_API_KEY_URL, + "BLOCKRUN_API_URL", + "BLOCKRUN_WALLET_KEY", + "BASE_CHAIN_WALLET_KEY", + "SOLANA_WALLET_KEY", + ): + monkeypatch.delenv(var, raising=False) + + +@pytest.mark.parametrize("cls", ALL_CLIENTS, ids=lambda c: c.__name__) +def test_an_api_key_always_reaches_the_account_rail(cls): + client = build(cls, API_KEY) + assert host_of(client) == DEFAULT_API_KEY_URL + + +@pytest.mark.parametrize("cls", SOLANA_CLIENTS, ids=lambda c: c.__name__) +def test_a_solana_wallet_reaches_the_solana_gateway(cls): + pytest.importorskip("x402") + assert host_of(build(cls, SOLANA_KEY)) == "https://sol.blockrun.ai/api" + + +@pytest.mark.parametrize("cls", BASE_CLIENTS, ids=lambda c: c.__name__) +def test_a_base_wallet_reaches_the_base_gateway(cls): + assert host_of(build(cls, BASE_KEY)) == "https://blockrun.ai/api" + + +@pytest.mark.parametrize("cls", ALL_CLIENTS, ids=lambda c: c.__name__) +def test_an_x402_gateway_override_never_captures_an_api_key(cls, monkeypatch): + """BLOCKRUN_API_URL names an x402 host. A developer pointing it at a private + deployment must not have an API-key client follow it there and hand over the + key — which is why the account rail has its own variable.""" + monkeypatch.setenv("BLOCKRUN_API_URL", "https://x402-deployment.internal/api") + assert host_of(build(cls, API_KEY)) == DEFAULT_API_KEY_URL + + +@pytest.mark.parametrize("cls", ALL_CLIENTS, ids=lambda c: c.__name__) +def test_the_account_rail_has_its_own_override(cls, monkeypatch): + monkeypatch.setenv(ENV_API_KEY_URL, "https://staging.blockrun.ai") + assert host_of(build(cls, API_KEY)) == "https://staging.blockrun.ai" + + +@pytest.mark.parametrize("cls", ALL_CLIENTS, ids=lambda c: c.__name__) +def test_an_explicit_api_url_wins_on_every_client(cls): + """The Solana constructors default api_url to the SVM gateway rather than + None, so an account-rail client there has to tell "the caller typed this" + apart from "nobody passed anything" — otherwise it either ignores the + argument or sends the key to sol.blockrun.ai.""" + assert host_of(build(cls, API_KEY, api_url="https://custom.example")) == ( + "https://custom.example" + ) + + +@pytest.mark.parametrize("cls", SOLANA_CLIENTS, ids=lambda c: c.__name__) +def test_the_solana_default_never_leaks_onto_the_account_rail(cls): + """Passing the default explicitly is indistinguishable from not passing it, + and it must resolve to the account rail either way.""" + assert host_of(build(cls, API_KEY, api_url="https://sol.blockrun.ai/api")) == ( + DEFAULT_API_KEY_URL + ) diff --git a/tests/unit/test_image_edit.py b/tests/unit/test_image_edit.py new file mode 100644 index 0000000..2e65122 --- /dev/null +++ b/tests/unit/test_image_edit.py @@ -0,0 +1,101 @@ +""" +Unit tests for image editing (img2img) request shaping. + +The production /v1/images/image2image endpoint accepts ``image`` as either a +single base64 data URI or an array of 1-4 data URIs (multi-image fusion). +These tests use httpx.MockTransport — no real network call ever happens — and +assert that the SDK passes ``image`` through unchanged for both shapes, and +that the 402 → sign → retry dance preserves it on the paid request. +""" + +from __future__ import annotations + +import json + +import httpx + +from blockrun_llm import ImageClient + +from ..helpers import TEST_PRIVATE_KEY, build_payment_required_response + +DATA_URI = "data:image/png;base64,iVBORw0KGgoAAAANSUhEUgAAAAEAAAABCAQAAAC1HAwCAAAAC0lEQVR42mNk+M8AAAMBAQDJ/pLvAAAAAElFTkSuQmCC" + + +def _image_edit_transport(calls: list[httpx.Request]) -> httpx.MockTransport: + """First POST → 402 with payment requirements; retry with signature → 200.""" + + def handler(request: httpx.Request) -> httpx.Response: + calls.append(request) + if "PAYMENT-SIGNATURE" not in request.headers: + return httpx.Response( + 402, + headers={ + "content-type": "application/json", + "payment-required": build_payment_required_response(), + }, + json={"error": "Payment Required", "price": {"amount": "0.04"}}, + ) + return httpx.Response( + 200, + headers={"content-type": "application/json"}, + json={ + "created": 1700000000, + "data": [{"url": "https://blockrun.ai/img/out.png"}], + }, + ) + + return httpx.MockTransport(handler) + + +def _make_client(calls: list[httpx.Request]) -> ImageClient: + client = ImageClient(private_key=TEST_PRIVATE_KEY) + client._client = httpx.Client(transport=_image_edit_transport(calls)) + return client + + +def test_edit_single_image_passes_string_through(): + calls: list[httpx.Request] = [] + client = _make_client(calls) + + result = client.edit("Make the sky purple", image=DATA_URI) + + # 402 dance = exactly two requests; signature only on the retry. + assert len(calls) == 2 + assert "PAYMENT-SIGNATURE" not in calls[0].headers + assert "PAYMENT-SIGNATURE" in calls[1].headers + + body = json.loads(calls[1].content) + assert body["image"] == DATA_URI + assert isinstance(body["image"], str) + assert result.data[0].url == "https://blockrun.ai/img/out.png" + + +def test_edit_defaults_to_gpt_image_2(): + calls: list[httpx.Request] = [] + client = _make_client(calls) + + client.edit("Make the sky purple", image=DATA_URI) + + body = json.loads(calls[1].content) + # Default edit model matches the production schema default. + assert body["model"] == "openai/gpt-image-2" + + +def test_edit_multi_image_passes_list_through(): + calls: list[httpx.Request] = [] + client = _make_client(calls) + + images = [DATA_URI, DATA_URI] + result = client.edit( + "Place the logo on the t-shirt", + image=images, + model="google/nano-banana", + ) + + body = json.loads(calls[1].content) + # The array must survive serialization as a JSON array, not a coerced string. + assert body["image"] == images + assert isinstance(body["image"], list) + assert len(body["image"]) == 2 + assert body["model"] == "google/nano-banana" + assert result.data[0].url == "https://blockrun.ai/img/out.png" diff --git a/tests/unit/test_image_parameter_validation.py b/tests/unit/test_image_parameter_validation.py new file mode 100644 index 0000000..9cb7d5b --- /dev/null +++ b/tests/unit/test_image_parameter_validation.py @@ -0,0 +1,93 @@ +""" +Unit tests for ImageClient parameter validation. + +Ensures that unsupported parameters are caught early with helpful error messages +instead of confusing TypeErrors from the Python runtime. +""" + +from __future__ import annotations + +import pytest + +from blockrun_llm import ImageClient + +from ..helpers import TEST_PRIVATE_KEY + + +def _make_client() -> ImageClient: + """Create a client with test private key.""" + return ImageClient(private_key=TEST_PRIVATE_KEY) + + +def test_generate_rejects_quality_parameter(): + """Unsupported quality parameter should raise TypeError with helpful message.""" + client = _make_client() + + with pytest.raises(TypeError) as excinfo: + client.generate("A cat", quality="hd") + + error = str(excinfo.value) + assert "quality" in error + assert "Valid parameters are" in error + assert "prompt, model, size, n" in error + + +def test_generate_rejects_multiple_invalid_parameters(): + """Multiple invalid parameters should all be listed in error message.""" + client = _make_client() + + with pytest.raises(TypeError) as excinfo: + client.generate("A cat", quality="hd", style="realistic", foo="bar") + + error = str(excinfo.value) + assert "foo" in error + assert "quality" in error + assert "style" in error + + +def test_generate_accepts_all_valid_parameters(): + """Valid parameters should not raise.""" + client = _make_client() + + # This should not raise a validation error (may raise network error, but that's OK). + # We just want to verify the parameter validation passes. + try: + client.generate("A cat", model="google/nano-banana", size="1024x1024", n=2) + except TypeError as e: + # Should NOT be a parameter validation error + if "Valid parameters are" in str(e): + pytest.fail(f"Valid parameters rejected: {e}") + except Exception: + # Network/payment errors are fine for this test + pass + + +def test_edit_rejects_quality_parameter(): + """Unsupported quality parameter in edit() should raise TypeError with helpful message.""" + client = _make_client() + data_uri = "data:image/png;base64,iVBORw0KGgoAAAANSUhEUgAAAAEAAAABCAQAAAC1HAwCAAAAC0lEQVR42mNk+M8AAAMBAQDJ/pLvAAAAAElFTkSuQmCC" + + with pytest.raises(TypeError) as excinfo: + client.edit("Make it red", image=data_uri, quality="hd") + + error = str(excinfo.value) + assert "quality" in error + assert "Valid parameters are" in error + assert "prompt, image, model, mask, size, n" in error + + +def test_edit_accepts_all_valid_parameters(): + """Valid parameters should not raise parameter validation error.""" + client = _make_client() + data_uri = "data:image/png;base64,iVBORw0KGgoAAAANSUhEUgAAAAEAAAABCAQAAAC1HAwCAAAAC0lEQVR42mNk+M8AAAMBAQDJ/pLvAAAAAElFTkSuQmCC" + + try: + client.edit( + "Make it red", image=data_uri, model="google/nano-banana", size="1024x1024", n=1 + ) + except TypeError as e: + if "Valid parameters are" in str(e): + pytest.fail(f"Valid parameters rejected: {e}") + except Exception: + # Network/payment errors are fine + pass diff --git a/tests/unit/test_image_poll.py b/tests/unit/test_image_poll.py new file mode 100644 index 0000000..acc897d --- /dev/null +++ b/tests/unit/test_image_poll.py @@ -0,0 +1,269 @@ +"""Tests for the image-generation 202 + poll_url slow path. + +Regression guard for the silent failure on slow models like +``openai/gpt-image-2`` and ``openai/dall-e-3``: pre-fix, the SDK treated +202 as success and tried to parse the job-stub JSON as an +``ImageResponse``, raising a confusing Pydantic ValidationError. Now the +client transparently polls until the upstream finishes. + +These tests use ``httpx.MockTransport`` so no real network is ever +called. They also patch ``IMAGE_POLL_INTERVAL_SECONDS`` to 0 so the loop +spins instantly. +""" + +from __future__ import annotations + +import httpx +import pytest + +from blockrun_llm import ImageClient +from blockrun_llm.types import APIError, PaymentError + +from ..helpers import TEST_PRIVATE_KEY, build_payment_required_response + + +def _make_client(transport: httpx.MockTransport) -> ImageClient: + client = ImageClient(private_key=TEST_PRIVATE_KEY) + client._client = httpx.Client(transport=transport) + return client + + +def _payment_required_402(request: httpx.Request) -> httpx.Response: + return httpx.Response( + 402, + headers={ + "content-type": "application/json", + "payment-required": build_payment_required_response(), + }, + json={"error": "Payment Required", "price": {"amount": "0.06"}}, + ) + + +def test_image_generate_polls_to_completion_on_202(monkeypatch: pytest.MonkeyPatch) -> None: + """gpt-image-2 routinely exceeds the 30s inline window → 202 + poll_url + → SDK should poll the same URL with the same PAYMENT-SIGNATURE until + status=completed, then return the image.""" + monkeypatch.setattr(ImageClient, "IMAGE_POLL_INTERVAL_SECONDS", 0.0) + + calls: list[httpx.Request] = [] + poll_state = {"count": 0} + + def handler(request: httpx.Request) -> httpx.Response: + calls.append(request) + path = request.url.path + + if request.method == "POST" and path.endswith("/v1/images/generations"): + if "PAYMENT-SIGNATURE" not in request.headers: + return _payment_required_402(request) + # Signed POST → slow path 202 with poll_url + return httpx.Response( + 202, + headers={"content-type": "application/json"}, + json={ + "id": "img_abc123", + "object": "image.generation.job", + "status": "queued", + "model": "openai/gpt-image-2", + "size": "1024x1024", + "n": 1, + "poll_url": "/api/v1/images/generations/img_abc123", + "created": 1700000000, + }, + ) + + if request.method == "GET" and "/v1/images/generations/img_abc123" in path: + poll_state["count"] += 1 + # First poll: still in progress; second poll: completed. + if poll_state["count"] == 1: + return httpx.Response( + 202, + headers={"content-type": "application/json"}, + json={"id": "img_abc123", "status": "in_progress"}, + ) + return httpx.Response( + 200, + headers={"content-type": "application/json"}, + json={ + "id": "img_abc123", + "object": "image.generation.job", + "status": "completed", + "model": "openai/gpt-image-2", + "created": 1700000000, + "data": [{"url": "https://blockrun.ai/img/abc.png"}], + }, + ) + + return httpx.Response(404) + + client = _make_client(httpx.MockTransport(handler)) + result = client.generate("古风汉服少女", model="openai/gpt-image-2", size="1024x1024") + + # 1 probe POST + 1 signed POST + 2 polls = 4 calls + assert len(calls) == 4 + assert calls[0].method == "POST" + assert "PAYMENT-SIGNATURE" not in calls[0].headers + assert calls[1].method == "POST" + assert "PAYMENT-SIGNATURE" in calls[1].headers + # Both polls replay the same signature. + assert calls[2].method == "GET" + assert calls[3].method == "GET" + assert calls[2].headers.get("PAYMENT-SIGNATURE") + assert calls[2].headers["PAYMENT-SIGNATURE"] == calls[1].headers["PAYMENT-SIGNATURE"] + assert calls[3].headers["PAYMENT-SIGNATURE"] == calls[1].headers["PAYMENT-SIGNATURE"] + + assert result.data[0].url == "https://blockrun.ai/img/abc.png" + + +def test_image_poll_surfaces_settlement_failure(monkeypatch: pytest.MonkeyPatch) -> None: + """If the final poll's settlement fails on the facilitator + (e.g. ``transaction_simulation_failed``), the SDK must surface the + gateway's real reason instead of swallowing it.""" + monkeypatch.setattr(ImageClient, "IMAGE_POLL_INTERVAL_SECONDS", 0.0) + + def handler(request: httpx.Request) -> httpx.Response: + path = request.url.path + if request.method == "POST" and path.endswith("/v1/images/generations"): + if "PAYMENT-SIGNATURE" not in request.headers: + return _payment_required_402(request) + return httpx.Response( + 202, + headers={"content-type": "application/json"}, + json={ + "id": "img_xyz", + "status": "queued", + "poll_url": "/api/v1/images/generations/img_xyz", + "created": 1700000000, + "model": "openai/gpt-image-2", + "size": "1024x1024", + "n": 1, + }, + ) + # Settlement failed at the facilitator. + return httpx.Response( + 402, + headers={"content-type": "application/json"}, + json={ + "error": "Payment settlement failed", + "details": "transaction_simulation_failed", + }, + ) + + client = _make_client(httpx.MockTransport(handler)) + + with pytest.raises(PaymentError) as excinfo: + client.generate("a red apple", model="openai/gpt-image-2") + + exc = excinfo.value + assert exc.status_code == 402 + assert exc.response is not None + assert exc.response.get("details") == "transaction_simulation_failed" + assert "transaction_simulation_failed" in str(exc) + + +def test_image_poll_times_out_without_settlement(monkeypatch: pytest.MonkeyPatch) -> None: + """Poll budget exhausted → APIError 504 + no settlement (so no + charge). This is the customer-friendly contract: pay only when the + image is delivered.""" + monkeypatch.setattr(ImageClient, "IMAGE_POLL_INTERVAL_SECONDS", 0.0) + monkeypatch.setattr(ImageClient, "IMAGE_POLL_BUDGET_SECONDS", 0.05) + + def handler(request: httpx.Request) -> httpx.Response: + path = request.url.path + if request.method == "POST" and path.endswith("/v1/images/generations"): + if "PAYMENT-SIGNATURE" not in request.headers: + return _payment_required_402(request) + return httpx.Response( + 202, + headers={"content-type": "application/json"}, + json={ + "id": "img_stuck", + "status": "queued", + "poll_url": "/api/v1/images/generations/img_stuck", + "created": 1700000000, + "model": "openai/gpt-image-2", + "size": "1024x1024", + "n": 1, + }, + ) + # Always still in_progress — never completes. + return httpx.Response( + 202, + headers={"content-type": "application/json"}, + json={"id": "img_stuck", "status": "in_progress"}, + ) + + client = _make_client(httpx.MockTransport(handler)) + + with pytest.raises(APIError) as excinfo: + client.generate("waiting forever", model="openai/gpt-image-2") + + exc = excinfo.value + assert exc.status_code == 504 + assert "did not complete" in str(exc) + assert "no payment was taken" in str(exc).lower() + + +def test_image_generate_fast_path_unchanged(monkeypatch: pytest.MonkeyPatch) -> None: + """Regression: fast models that return 200 inline must still work + identically — the poll path is only entered on 202.""" + monkeypatch.setattr(ImageClient, "IMAGE_POLL_INTERVAL_SECONDS", 0.0) + + calls: list[httpx.Request] = [] + + def handler(request: httpx.Request) -> httpx.Response: + calls.append(request) + if "PAYMENT-SIGNATURE" not in request.headers: + return _payment_required_402(request) + return httpx.Response( + 200, + headers={"content-type": "application/json"}, + json={ + "created": 1700000000, + "data": [{"url": "https://blockrun.ai/img/fast.png"}], + }, + ) + + client = _make_client(httpx.MockTransport(handler)) + result = client.generate("fast model", model="google/nano-banana") + + assert len(calls) == 2 # No polling — went straight through. + assert result.data[0].url == "https://blockrun.ai/img/fast.png" + + +def test_image_poll_surfaces_upstream_failure(monkeypatch: pytest.MonkeyPatch) -> None: + """If the upstream generation fails (content policy, model error, + etc.), the gateway flips ``status: failed`` on the poll. The SDK + should raise APIError with the upstream reason — no settlement.""" + monkeypatch.setattr(ImageClient, "IMAGE_POLL_INTERVAL_SECONDS", 0.0) + + def handler(request: httpx.Request) -> httpx.Response: + if request.method == "POST": + if "PAYMENT-SIGNATURE" not in request.headers: + return _payment_required_402(request) + return httpx.Response( + 202, + headers={"content-type": "application/json"}, + json={ + "id": "img_bad", + "status": "queued", + "poll_url": "/api/v1/images/generations/img_bad", + "created": 1700000000, + "model": "openai/gpt-image-2", + "size": "1024x1024", + "n": 1, + }, + ) + return httpx.Response( + 200, + headers={"content-type": "application/json"}, + json={ + "id": "img_bad", + "status": "failed", + "error": "content policy violation", + }, + ) + + client = _make_client(httpx.MockTransport(handler)) + with pytest.raises(APIError) as excinfo: + client.generate("blocked prompt", model="openai/gpt-image-2") + assert "content policy violation" in str(excinfo.value) diff --git a/tests/unit/test_invalid_message_fail_fast.py b/tests/unit/test_invalid_message_fail_fast.py new file mode 100644 index 0000000..06bb148 --- /dev/null +++ b/tests/unit/test_invalid_message_fail_fast.py @@ -0,0 +1,143 @@ +"""The client half of the verify retry-storm fix (gateway side: blockrun-sol +``x402-solana.ts`` classifyInvalidMessage). + +A payer whose USDC token account was never created fails simulation with +``InvalidAccountData``. The gateway's coarse ``invalidReason`` collapses that to +``transaction_simulation_failed``, which ``_UNRECOVERABLE_PAYMENT_PATTERNS`` +deliberately omits (it IS recoverable under concurrent load) — so the SDK burned +all 5 payment attempts on a wallet that could never pay. The gateway now returns +the facilitator's ``invalidMessage`` alongside the enum; these tests pin the SDK +reading it and failing fast. +""" + +from __future__ import annotations + +from typing import Any + +from blockrun_llm.solana_client import ( + _is_permanent_payment_error, + _is_unrecoverable_payment_error, +) +from blockrun_llm.validation import build_payment_rejected_error + + +class _FakeResponse: + def __init__(self, body: Any) -> None: + self._body = body + + def json(self) -> Any: + return self._body + + +class TestInvalidMessageReachesTheClassifier: + """build_payment_rejected_error must fold invalidMessage into the message — + the classifiers only ever see ``str(exc)``.""" + + def test_invalid_message_is_surfaced_in_the_error_string(self) -> None: + exc = build_payment_rejected_error( + _FakeResponse( + { + "error": "Payment verification failed", + "code": "PAYMENT_INVALID", + "debug": "transaction_simulation_failed", + "invalidMessage": "InvalidAccountData", + } + ) + ) + assert "InvalidAccountData" in str(exc) + assert exc.response is not None + assert exc.response["invalidMessage"] == "InvalidAccountData" + + def test_absent_invalid_message_leaves_the_message_unchanged(self) -> None: + exc = build_payment_rejected_error( + _FakeResponse({"error": "Payment settlement failed", "details": "insufficient_funds"}) + ) + assert "insufficient_funds" in str(exc) + + def test_oversized_invalid_message_is_dropped(self) -> None: + exc = build_payment_rejected_error( + _FakeResponse({"error": "Payment verification failed", "invalidMessage": "x" * 500}) + ) + assert exc.response is not None + assert "invalidMessage" not in exc.response + + def test_machine_readable_code_and_reason_are_preserved(self) -> None: + exc = build_payment_rejected_error( + _FakeResponse( + { + "error": "Payment verification failed", + "code": "PAYMENT_INVALID", + "reason": "expired_signature", + } + ) + ) + assert exc.response is not None + assert exc.response["code"] == "PAYMENT_INVALID" + assert exc.response["reason"] == "expired_signature" + + +class TestUnrecoverableClassification: + def test_invalid_account_data_is_unrecoverable(self) -> None: + """An unfunded wallet: no fresh nonce/blockhash can make this pass.""" + assert ( + _is_unrecoverable_payment_error( + "Payment rejected by gateway: transaction_simulation_failed (InvalidAccountData)" + ) + is True + ) + + def test_spelling_variants_all_classify(self) -> None: + for msg in ( + "invalid account data", + "invalid_account_data", + "InvalidAccountData", + "AccountNotFound", + "Error processing Instruction 0: invalid account data", + ): + assert _is_unrecoverable_payment_error( + f"Payment rejected by gateway: transaction_simulation_failed ({msg})" + ), msg + + def test_bare_simulation_failure_stays_recoverable(self) -> None: + """Without an invalidMessage we know nothing more than before — keep the + whole-request retry that exists to ride out concurrent-load failures.""" + assert ( + _is_unrecoverable_payment_error( + "Payment rejected by gateway: transaction_simulation_failed" + ) + is False + ) + + def test_blockhash_stays_recoverable_on_the_client(self) -> None: + """Deliberate asymmetry with the gateway: it stops retrying the SAME dead + header, but re-signing with a FRESH blockhash is exactly what fixes this, + and re-signing is what the SDK's whole-request retry does.""" + for msg in ("BlockhashNotFound", "BlockHeightExceeded"): + assert ( + _is_unrecoverable_payment_error( + f"Payment rejected by gateway: transaction_simulation_failed ({msg})" + ) + is False + ), msg + + def test_transient_errors_still_retry(self) -> None: + assert _is_unrecoverable_payment_error("503 Service Unavailable") is False + assert _is_unrecoverable_payment_error("") is False + + +class TestPermanentClassifierUnaffected: + """_is_permanent_payment_error governs the *fallback-model* decision and + already treats simulation/blockhash as permanent. The new patterns must not + perturb it.""" + + def test_still_permanent(self) -> None: + assert _is_permanent_payment_error("transaction_simulation_failed") is True + assert ( + _is_permanent_payment_error( + "Payment rejected by gateway: transaction_simulation_failed (InvalidAccountData)" + ) + is True + ) + + def test_still_transient(self) -> None: + assert _is_permanent_payment_error("503 Service Unavailable") is False diff --git a/tests/unit/test_music_poll.py b/tests/unit/test_music_poll.py new file mode 100644 index 0000000..1544410 --- /dev/null +++ b/tests/unit/test_music_poll.py @@ -0,0 +1,203 @@ +"""Tests for the music-generation 202 + poll_url path. + +Music is never fast: MiniMax takes one to three minutes per track and the +gateway answers 202 + poll_url once its inline window is over — which, since +2026-09-08, is at once. This client treated every non-200 as an error, so on +both rails a music request could not succeed at all; the enterprise ledger +showed 11 of 11 creates in 30 days answered 202 and this SDK raised +"API error: 202" for each. The image client already polls; music mirrors it. + +``httpx.MockTransport`` keeps the network out. The poll interval is patched +to 0 so the loop spins instantly. +""" + +from __future__ import annotations + +import httpx +import pytest + +from blockrun_llm import MusicClient +from blockrun_llm.types import APIError + +from ..helpers import TEST_PRIVATE_KEY, build_payment_required_response + +KEY = "brk_live_testkey" + + +def _wallet_client(transport: httpx.MockTransport) -> MusicClient: + client = MusicClient(private_key=TEST_PRIVATE_KEY) + client._client = httpx.Client(transport=transport) + return client + + +def _apikey_client(transport: httpx.MockTransport, monkeypatch: pytest.MonkeyPatch) -> MusicClient: + monkeypatch.setenv("BLOCKRUN_API_KEY", KEY) + monkeypatch.delenv("BLOCKRUN_WALLET_KEY", raising=False) + client = MusicClient() + client._client = httpx.Client(transport=transport, headers=client._client.headers) + return client + + +def _payment_required_402() -> httpx.Response: + return httpx.Response( + 402, + headers={ + "content-type": "application/json", + "payment-required": build_payment_required_response(), + }, + json={"error": "Payment Required", "price": {"amount": "0.1575"}}, + ) + + +def _queued(job_id: str) -> httpx.Response: + return httpx.Response( + 202, + headers={"content-type": "application/json"}, + json={ + "id": job_id, + "object": "audio.generation.job", + "status": "queued", + "model": "minimax/music-2.5+", + "poll_url": f"/api/v1/audio/generations/{job_id}", + "created": 1700000000, + }, + ) + + +def _completed(job_id: str) -> httpx.Response: + return httpx.Response( + 200, + headers={"content-type": "application/json", "x-payment-receipt": "0xabc"}, + json={ + "id": job_id, + "object": "audio.generation.job", + "status": "completed", + "model": "minimax/music-2.5+", + "created": 1700000000, + "data": [{"url": "https://blockrun.ai/media/track.mp3", "duration_seconds": 182}], + "payment": {"status": "settled"}, + }, + ) + + +def test_music_wallet_rail_polls_to_completion(monkeypatch: pytest.MonkeyPatch) -> None: + monkeypatch.setattr(MusicClient, "MUSIC_POLL_INTERVAL_SECONDS", 0.0) + calls: list[httpx.Request] = [] + polls = {"n": 0} + + def handler(request: httpx.Request) -> httpx.Response: + calls.append(request) + if request.method == "POST" and request.url.path.endswith("/v1/audio/generations"): + if "PAYMENT-SIGNATURE" not in request.headers: + return _payment_required_402() + return _queued("mus_1") + if request.method == "GET" and "/v1/audio/generations/mus_1" in request.url.path: + polls["n"] += 1 + if polls["n"] == 1: + return httpx.Response( + 202, + headers={"content-type": "application/json"}, + json={"id": "mus_1", "status": "in_progress"}, + ) + return _completed("mus_1") + return httpx.Response(404) + + result = _wallet_client(httpx.MockTransport(handler)).generate("chill lo-fi beats") + + assert [c.method for c in calls] == ["POST", "POST", "GET", "GET"] + # Every poll replays the signature the create was paid with; the job is + # bound to that wallet and settles on the completed poll. + assert calls[2].headers["PAYMENT-SIGNATURE"] == calls[1].headers["PAYMENT-SIGNATURE"] + assert calls[3].headers["PAYMENT-SIGNATURE"] == calls[1].headers["PAYMENT-SIGNATURE"] + assert result.data[0].url == "https://blockrun.ai/media/track.mp3" + assert result.data[0].duration_seconds == 182 + assert result.txHash == "0xabc" + + +def test_music_api_key_rail_polls_on_first_202(monkeypatch: pytest.MonkeyPatch) -> None: + # The account rail has already paid, so the 202 comes on the FIRST post + # and the polls carry the key, not a signature. + monkeypatch.setattr(MusicClient, "MUSIC_POLL_INTERVAL_SECONDS", 0.0) + calls: list[httpx.Request] = [] + + def handler(request: httpx.Request) -> httpx.Response: + calls.append(request) + if request.method == "POST": + return _queued("mus_2") + if request.method == "GET" and "/v1/audio/generations/mus_2" in request.url.path: + return _completed("mus_2") + return httpx.Response(404) + + result = _apikey_client(httpx.MockTransport(handler), monkeypatch).generate("epic orchestral") + + assert [c.method for c in calls] == ["POST", "GET"] + assert "PAYMENT-SIGNATURE" not in calls[1].headers + assert calls[1].headers.get("authorization") == f"Bearer {KEY}" + # The gateway's poll_url is /api/v1/...; api.blockrun.ai serves it at /v1/... + assert calls[1].url.path == "/v1/audio/generations/mus_2" + assert result.data[0].url == "https://blockrun.ai/media/track.mp3" + + +def test_music_poll_surfaces_upstream_failure(monkeypatch: pytest.MonkeyPatch) -> None: + monkeypatch.setattr(MusicClient, "MUSIC_POLL_INTERVAL_SECONDS", 0.0) + + def handler(request: httpx.Request) -> httpx.Response: + if request.method == "POST": + if "PAYMENT-SIGNATURE" not in request.headers: + return _payment_required_402() + return _queued("mus_3") + return httpx.Response( + 200, + headers={"content-type": "application/json"}, + json={ + "id": "mus_3", + "status": "failed", + "error": "The operation was aborted due to timeout", + "payment_status": "not_charged", + }, + ) + + with pytest.raises(APIError) as excinfo: + _wallet_client(httpx.MockTransport(handler)).generate("waiting") + assert "aborted due to timeout" in str(excinfo.value) + + +def test_music_poll_times_out_without_settlement(monkeypatch: pytest.MonkeyPatch) -> None: + monkeypatch.setattr(MusicClient, "MUSIC_POLL_INTERVAL_SECONDS", 0.0) + monkeypatch.setattr(MusicClient, "MUSIC_POLL_BUDGET_SECONDS", 0.05) + + def handler(request: httpx.Request) -> httpx.Response: + if request.method == "POST": + if "PAYMENT-SIGNATURE" not in request.headers: + return _payment_required_402() + return _queued("mus_4") + return httpx.Response( + 202, + headers={"content-type": "application/json"}, + json={"id": "mus_4", "status": "in_progress"}, + ) + + with pytest.raises(APIError) as excinfo: + _wallet_client(httpx.MockTransport(handler)).generate("forever") + assert excinfo.value.status_code == 504 + assert "no payment was taken" in str(excinfo.value).lower() + + +def test_music_fast_path_unchanged() -> None: + # A track that finishes inline still comes back as the legacy 200 shape. + def handler(request: httpx.Request) -> httpx.Response: + if "PAYMENT-SIGNATURE" not in request.headers: + return _payment_required_402() + return httpx.Response( + 200, + headers={"content-type": "application/json", "x-payment-receipt": "0xfast"}, + json={ + "created": 1700000000, + "model": "minimax/music-2.5+", + "data": [{"url": "https://blockrun.ai/media/fast.mp3"}], + }, + ) + + result = _wallet_client(httpx.MockTransport(handler)).generate("quick jingle") + assert result.data[0].url == "https://blockrun.ai/media/fast.mp3" + assert result.txHash == "0xfast" diff --git a/tests/unit/test_paid_request_error_prefix.py b/tests/unit/test_paid_request_error_prefix.py new file mode 100644 index 0000000..f023e42 --- /dev/null +++ b/tests/unit/test_paid_request_error_prefix.py @@ -0,0 +1,137 @@ +"""The paid-request error message must not claim money moved — or that it didn't. + +"API error after payment" read as *your funds are gone* on every failure, which +is usually false. This is a regression test for a real misdiagnosis: an +image-edit 500 was reported as lost USDC by two readers before anyone checked +the gateway's settle ordering. The wording alone caused it. + +The second half of the file guards the opposite error, which is worse. Both +gateways send the settlement under the x402 v2 name ``PAYMENT-RESPONSE`` and +neither ever sends ``X-PAYMENT-RESPONSE`` — so reading only the legacy name +decodes nothing in production and reports every settled failure as unsettled. +And on Solana's paid chat path, settle runs in parallel with the upstream call +and the error re-raises before it lands, so the requests that DID charge (the +ones the gateway logs as ``CHARGED BUT REQUEST FAILED — refund manually``) are +exactly the ones arriving with no header. Absence cannot be sold as "you weren't +charged". +""" + +import base64 +import json + +import httpx +import pytest + +from blockrun_llm.tx_log import paid_request_error_prefix, read_settlement_header + +# What our gateways actually send (x402 v2). The legacy name is still accepted +# for other facilitators, so both are exercised everywhere it matters. +SPEC_NAME = "PAYMENT-RESPONSE" +LEGACY_NAME = "X-PAYMENT-RESPONSE" +BOTH_NAMES = [SPEC_NAME, LEGACY_NAME] + + +def _settlement_header(tx_hash="0xabc123", **extra): + payload = {"transaction": tx_hash, "network": "base", "success": True, **extra} + return base64.b64encode(json.dumps(payload).encode()).decode() + + +class TestHeaderName: + """The bug that made the whole mechanism dead code in production.""" + + def test_spec_name_is_read(self): + """Regression: the SDK read only the legacy name, which no gateway sends. + + blockrun and blockrun-sol emit `PAYMENT-RESPONSE` (36 and 25 call sites + respectively) and `X-PAYMENT-RESPONSE` zero times. The sidecar hit this + in blockrun-litellm 0.6.0, live-verified against a real paid call. + """ + msg = paid_request_error_prefix(httpx.Headers({SPEC_NAME: _settlement_header("0xfeed")})) + assert "SETTLED" in msg and "0xfeed" in msg + + def test_legacy_name_still_accepted(self): + msg = paid_request_error_prefix(httpx.Headers({LEGACY_NAME: _settlement_header("0xbeef")})) + assert "SETTLED" in msg and "0xbeef" in msg + + def test_spec_name_wins_when_both_present(self): + headers = httpx.Headers( + {SPEC_NAME: _settlement_header("0xspec"), LEGACY_NAME: _settlement_header("0xlegacy")} + ) + assert "0xspec" in paid_request_error_prefix(headers) + + def test_reader_never_raises_on_a_junk_mapping(self): + class Hostile: + def get(self, _name): + raise RuntimeError("headers exploded") + + assert read_settlement_header(Hostile()) is None + + +class TestUnsettled: + """No settlement header means UNKNOWN, and must never be sold as "free".""" + + def test_no_header_does_not_claim_payment_was_taken(self): + msg = paid_request_error_prefix(httpx.Headers({})) + assert "SETTLED" not in msg + # The specific phrase that caused the false alarm must be gone. + assert "after payment" not in msg + + def test_no_header_does_not_claim_payment_was_NOT_taken(self): + """The inverse error, and the more expensive one. + + Solana settles in parallel and re-raises before settle lands, so a + charged-but-failed request carries no header. Promising "payment likely + not taken" there is a false all-clear on exactly the request that needs + a manual refund. + """ + msg = paid_request_error_prefix(httpx.Headers({})) + assert "likely not taken" not in msg + assert "check your wallet history" in msg, "must point somewhere authoritative" + + @pytest.mark.parametrize("name", BOTH_NAMES) + @pytest.mark.parametrize( + "bad", ["", "!!!not-base64!!!", "e30=", base64.b64encode(b"[]").decode()] + ) + def test_unparseable_header_degrades_to_unsettled(self, name, bad): + """An error path must never raise while reporting an error.""" + msg = paid_request_error_prefix(httpx.Headers({name: bad})) + assert "no settlement reported" in msg + + @pytest.mark.parametrize("name", BOTH_NAMES) + def test_header_without_tx_hash_is_not_a_settlement(self, name): + payload = base64.b64encode(json.dumps({"network": "base"}).encode()).decode() + msg = paid_request_error_prefix(httpx.Headers({name: payload})) + assert "no settlement reported" in msg + + def test_success_true_without_tx_hash_is_still_unsettled(self): + """`success` is not a settle signal: the gateways hard-code it to true + even when settle didn't land, so older clients don't surface an error. + A tx hash is the only field that means money moved — which is what the + gateways gate their own revenue accounting on.""" + payload = base64.b64encode(json.dumps({"success": True, "network": "base"}).encode()) + msg = paid_request_error_prefix(httpx.Headers({SPEC_NAME: payload.decode()})) + assert "SETTLED" not in msg + + +class TestSettled: + """A settlement header means funds really did move — say so, and name the tx.""" + + def test_reports_settlement_and_tx_hash(self): + msg = paid_request_error_prefix( + httpx.Headers({SPEC_NAME: _settlement_header("0xdeadbeef")}) + ) + assert "SETTLED" in msg + assert "0xdeadbeef" in msg, "the tx hash is what makes this actionable" + + @pytest.mark.parametrize("name", BOTH_NAMES) + def test_solana_signature_field_also_counts(self, name): + payload = base64.b64encode(json.dumps({"signature": "5xSolSig"}).encode()).decode() + msg = paid_request_error_prefix(httpx.Headers({name: payload})) + assert "SETTLED" in msg and "5xSolSig" in msg + + +def test_the_two_cases_are_distinguishable(): + """The whole point: they used to be the same string.""" + unsettled = paid_request_error_prefix(httpx.Headers({})) + settled = paid_request_error_prefix(httpx.Headers({SPEC_NAME: _settlement_header()})) + assert unsettled != settled diff --git a/tests/unit/test_passthrough_defi_dex_modal.py b/tests/unit/test_passthrough_defi_dex_modal.py new file mode 100644 index 0000000..a7ab409 --- /dev/null +++ b/tests/unit/test_passthrough_defi_dex_modal.py @@ -0,0 +1,157 @@ +"""Unit tests for the DefiLlama / 0x DEX / Modal passthrough methods.""" + +import os + +import pytest + +from blockrun_llm import LLMClient + + +@pytest.fixture +def client(): + # Deterministic dummy key — never signs against a live endpoint in unit + # tests; we only exercise local request/path construction. + os.environ.setdefault("BLOCKRUN_WALLET_KEY", "0x" + "11" * 32) + return LLMClient() + + +@pytest.fixture +def captured(client, monkeypatch): + captured = {} + + def fake_get(endpoint, params=None): + captured["method"] = "GET" + captured["endpoint"] = endpoint + captured["params"] = params + return {"ok": True} + + def fake_post(endpoint, body): + captured["method"] = "POST" + captured["endpoint"] = endpoint + captured["body"] = body + return {"ok": True} + + monkeypatch.setattr(client, "_get_with_payment_raw", fake_get) + monkeypatch.setattr(client, "_request_with_payment_raw", fake_post) + return captured + + +# ── DefiLlama ──────────────────────────────────────────────────────────── + + +def test_defi_generic_path_and_params(client, captured): + client.defi("yields", chain="Base") + assert captured["method"] == "GET" + assert captured["endpoint"] == "/v1/defillama/yields" + assert captured["params"] == {"chain": "Base"} + + +def test_defi_conveniences(client, captured): + client.defi_protocols() + assert captured["endpoint"] == "/v1/defillama/protocols" + client.defi_protocol("aave") + assert captured["endpoint"] == "/v1/defillama/protocol/aave" + client.defi_chains() + assert captured["endpoint"] == "/v1/defillama/chains" + + +def test_defi_prices_joins_coin_list(client, captured): + client.defi_prices(["coingecko:bitcoin", "base:0xabc"]) + assert captured["endpoint"] == "/v1/defillama/prices/coingecko:bitcoin,base:0xabc" + client.defi_prices("coingecko:ethereum") + assert captured["endpoint"] == "/v1/defillama/prices/coingecko:ethereum" + + +# ── 0x DEX ─────────────────────────────────────────────────────────────── + + +def test_dex_get_with_params(client, captured): + client.dex_quote(chainId=8453, sellToken="0xa", buyToken="0xb", sellAmount="1000") + assert captured["method"] == "GET" + assert captured["endpoint"] == "/v1/zerox/quote" + assert captured["params"]["chainId"] == 8453 + + +def test_dex_gasless_submit_is_post(client, captured): + client.dex_gasless_submit({"trade": {"signature": "0xsig"}}) + assert captured["method"] == "POST" + assert captured["endpoint"] == "/v1/zerox/gasless/submit" + assert captured["body"] == {"trade": {"signature": "0xsig"}} + + +def test_dex_gasless_status_embeds_hash(client, captured): + client.dex_gasless_status("0xtradehash") + assert captured["endpoint"] == "/v1/zerox/gasless/status/0xtradehash" + + +def test_dex_chain_discovery(client, captured): + client.dex_chains() + assert captured["endpoint"] == "/v1/zerox/swap/chains" + client.dex_gasless_chains() + assert captured["endpoint"] == "/v1/zerox/gasless/chains" + + +# ── Modal ──────────────────────────────────────────────────────────────── + + +def test_modal_create_exec_lifecycle(client, captured): + client.modal_sandbox_create(image="python:3.11", gpu="T4") + assert captured["method"] == "POST" + assert captured["endpoint"] == "/v1/modal/sandbox/create" + assert captured["body"] == {"image": "python:3.11", "gpu": "T4"} + + client.modal_sandbox_exec("sb_123", ["python", "-c", "print(1)"]) + assert captured["endpoint"] == "/v1/modal/sandbox/exec" + assert captured["body"]["sandbox_id"] == "sb_123" + assert captured["body"]["command"] == ["python", "-c", "print(1)"] + + client.modal_sandbox_status("sb_123") + assert captured["endpoint"] == "/v1/modal/sandbox/status" + + client.modal_sandbox_terminate("sb_123") + assert captured["endpoint"] == "/v1/modal/sandbox/terminate" + assert captured["body"] == {"sandbox_id": "sb_123"} + + +# ── Coinbase Onramp ────────────────────────────────────────────────────── + +ADDR = "0x" + "ab" * 20 # well-formed 0x + 40 hex + + +def test_onramp_path_and_body(client, monkeypatch): + captured = {} + + def fake_post(endpoint, body): + captured["endpoint"] = endpoint + captured["body"] = body + return {"url": "https://pay.coinbase.com/buy/xyz"} + + monkeypatch.setattr(client, "_request_with_payment_raw", fake_post) + result = client.onramp(ADDR) + assert captured["endpoint"] == "/v1/onramp/token" + assert captured["body"] == {"address": ADDR, "network": "base", "asset": "USDC"} + assert result["url"].startswith("https://pay.coinbase.com/") + + +@pytest.mark.parametrize("bad", ["", "0xshort", "not-an-address", "0x" + "zz" * 20]) +def test_onramp_rejects_malformed_address(client, monkeypatch, bad): + # Never reaches the network — validation must fire first. + monkeypatch.setattr( + client, + "_request_with_payment_raw", + lambda *a, **k: pytest.fail("should not POST on bad address"), + ) + with pytest.raises(ValueError): + client.onramp(bad) + + +def test_onramp_rejects_non_coinbase_url(client, monkeypatch): + from blockrun_llm import APIError + + monkeypatch.setattr( + client, + "_request_with_payment_raw", + lambda endpoint, body: {"url": "https://evil.example.com/buy"}, + ) + with pytest.raises(APIError, match="no onramp url"): + client.onramp(ADDR) diff --git a/tests/unit/test_payment_error_helper.py b/tests/unit/test_payment_error_helper.py new file mode 100644 index 0000000..945f5ea --- /dev/null +++ b/tests/unit/test_payment_error_helper.py @@ -0,0 +1,107 @@ +"""Tests for the 402-retry payment-rejected helper. + +These cover the regression where a Solana settlement failure +(``transaction_simulation_failed``, ``insufficient_funds``, ...) was +swallowed by a generic ``"Payment rejected. Check your Solana USDC +balance."`` message, leaving customers no way to diagnose. +""" + +from __future__ import annotations + +from typing import Any + +from blockrun_llm.types import PaymentError +from blockrun_llm.validation import build_payment_rejected_error + + +class _FakeResponse: + """Minimal stand-in for ``httpx.Response`` that ``.json()``.""" + + def __init__(self, body: Any) -> None: + self._body = body + + def json(self) -> Any: + if isinstance(self._body, Exception): + raise self._body + return self._body + + +class TestPaymentErrorEnrichment: + def test_payment_error_carries_status_and_response(self) -> None: + """The new kwargs are public API — callers and proxies use them.""" + exc = PaymentError( + "Payment rejected by gateway: transaction_simulation_failed", + status_code=402, + response={ + "message": "Payment settlement failed", + "details": "transaction_simulation_failed", + }, + ) + assert exc.status_code == 402 + assert exc.response is not None + assert exc.response["details"] == "transaction_simulation_failed" + assert "transaction_simulation_failed" in str(exc) + + def test_payment_error_backwards_compatible_no_kwargs(self) -> None: + """Pre-0.32.0 callers raise ``PaymentError("...")`` — still works.""" + exc = PaymentError("Payment rejected") + assert exc.status_code is None + assert exc.response is None + assert str(exc) == "Payment rejected" + + +class TestBuildPaymentRejectedError: + def test_preserves_gateway_details(self) -> None: + """The whole reason this helper exists: ``details`` must survive + from the gateway's body to ``exc.response`` and into ``str(exc)``.""" + gateway_body: dict[str, Any] = { + "error": "Payment settlement failed", + "details": "transaction_simulation_failed", + } + exc = build_payment_rejected_error(_FakeResponse(gateway_body)) + + assert isinstance(exc, PaymentError) + assert exc.status_code == 402 + assert exc.response is not None + assert exc.response["details"] == "transaction_simulation_failed" + # The message should mention the real reason, not a generic line. + assert "transaction_simulation_failed" in str(exc) + assert "Check your" not in str(exc) # generic fallback should be gone + + def test_truncates_overly_long_details(self) -> None: + """Defensive: if a future server bug stuffs free-form text into + ``details`` we don't want to leak unbounded payloads.""" + huge = "x" * 1024 + exc = build_payment_rejected_error( + _FakeResponse({"error": "Payment settlement failed", "details": huge}) + ) + # details > 256 chars is rejected — falls back to sanitized message + assert exc.response is not None + assert "details" not in exc.response + assert "Payment settlement failed" in str(exc) + + def test_handles_non_string_details(self) -> None: + """If the gateway sends a non-string ``details`` (list, dict, None), + we drop it rather than crashing — message uses the fallback.""" + exc = build_payment_rejected_error( + _FakeResponse({"error": "Payment settlement failed", "details": ["a", "b"]}) + ) + assert exc.response is not None + assert "details" not in exc.response + assert "Payment settlement failed" in str(exc) + + def test_handles_unparseable_body(self) -> None: + """Gateway returned HTML / empty / malformed JSON — we still + raise a usable PaymentError (status_code=402, generic message).""" + exc = build_payment_rejected_error(_FakeResponse(ValueError("not json"))) + assert isinstance(exc, PaymentError) + assert exc.status_code == 402 + # No raise — falls through to generic "Payment rejected by gateway" + assert str(exc).startswith("Payment rejected") + + def test_handles_non_dict_body(self) -> None: + """Gateway returned a JSON array or string — treat as empty dict.""" + exc = build_payment_rejected_error(_FakeResponse(["nope"])) + assert isinstance(exc, PaymentError) + assert exc.status_code == 402 + assert exc.response == {"code": None, "message": "API request failed"} diff --git a/tests/unit/test_portrait.py b/tests/unit/test_portrait.py new file mode 100644 index 0000000..3d2b8bc --- /dev/null +++ b/tests/unit/test_portrait.py @@ -0,0 +1,47 @@ +"""Unit tests for PortraitClient input validation.""" + +import os + +import pytest + +from blockrun_llm import PortraitClient + + +@pytest.fixture +def client(): + # Deterministic dummy key — never actually signs against a live endpoint + # in unit tests; we only exercise local validation paths. + os.environ.setdefault("BLOCKRUN_WALLET_KEY", "0x" + "11" * 32) + return PortraitClient() + + +def test_enroll_rejects_empty_name(client): + with pytest.raises(ValueError, match="name is required"): + client.enroll(name="", image_url="https://example.com/x.jpg") + + +def test_enroll_rejects_whitespace_name(client): + with pytest.raises(ValueError, match="name is required"): + client.enroll(name=" ", image_url="https://example.com/x.jpg") + + +def test_enroll_rejects_long_name(client): + long_name = "a" * 65 + with pytest.raises(ValueError, match="64 chars or fewer"): + client.enroll(name=long_name, image_url="https://example.com/x.jpg") + + +def test_enroll_rejects_non_http_url(client): + with pytest.raises(ValueError, match="image_url must be an http"): + client.enroll(name="ok", image_url="ftp://example.com/x.jpg") + + +def test_enroll_rejects_empty_url(client): + with pytest.raises(ValueError, match="image_url must be an http"): + client.enroll(name="ok", image_url="") + + +def test_get_wallet_address(client): + addr = client.get_wallet_address() + assert addr.startswith("0x") + assert len(addr) == 42 diff --git a/tests/unit/test_realface.py b/tests/unit/test_realface.py new file mode 100644 index 0000000..5719ff1 --- /dev/null +++ b/tests/unit/test_realface.py @@ -0,0 +1,98 @@ +"""Unit tests for RealFaceClient input validation.""" + +import os + +import pytest + +from blockrun_llm import RealFaceClient + + +@pytest.fixture +def client(): + # Deterministic dummy key — never actually signs against a live endpoint + # in unit tests; we only exercise local validation paths. + os.environ.setdefault("BLOCKRUN_WALLET_KEY", "0x" + "11" * 32) + return RealFaceClient() + + +# --- init() validation ------------------------------------------------------ + + +def test_init_rejects_empty_name(client): + with pytest.raises(ValueError, match="name is required"): + client.init(name="") + + +def test_init_rejects_whitespace_name(client): + with pytest.raises(ValueError, match="name is required"): + client.init(name=" ") + + +def test_init_rejects_long_name(client): + with pytest.raises(ValueError, match="64 chars or fewer"): + client.init(name="a" * 65) + + +def test_init_rejects_bad_group_id(client): + with pytest.raises(ValueError, match="legacy_rf_"): + client.init(name="ok", group_id="rf_123") + + +# --- status() / wait_for_active() validation -------------------------------- + + +def test_status_rejects_bad_group_id(client): + with pytest.raises(ValueError, match="legacy_rf_"): + client.status(group_id="not-a-group") + + +def test_status_rejects_empty_group_id(client): + with pytest.raises(ValueError, match="legacy_rf_"): + client.status(group_id="") + + +def test_wait_for_active_rejects_bad_group_id(client): + with pytest.raises(ValueError, match="legacy_rf_"): + client.wait_for_active(group_id="nope") + + +def test_wait_for_active_rejects_nonpositive_interval(client): + with pytest.raises(ValueError, match="poll_interval_seconds must be positive"): + client.wait_for_active(group_id="legacy_rf_1", poll_interval_seconds=0) + + +# --- enroll() validation ---------------------------------------------------- + + +def test_enroll_rejects_empty_name(client): + with pytest.raises(ValueError, match="name is required"): + client.enroll(name="", image_url="https://example.com/x.jpg", group_id="legacy_rf_1") + + +def test_enroll_rejects_long_name(client): + with pytest.raises(ValueError, match="64 chars or fewer"): + client.enroll(name="a" * 65, image_url="https://example.com/x.jpg", group_id="legacy_rf_1") + + +def test_enroll_rejects_non_http_url(client): + with pytest.raises(ValueError, match="image_url must be an http"): + client.enroll(name="ok", image_url="ftp://example.com/x.jpg", group_id="legacy_rf_1") + + +def test_enroll_rejects_empty_url(client): + with pytest.raises(ValueError, match="image_url must be an http"): + client.enroll(name="ok", image_url="", group_id="legacy_rf_1") + + +def test_enroll_rejects_bad_group_id(client): + with pytest.raises(ValueError, match="legacy_rf_"): + client.enroll(name="ok", image_url="https://example.com/x.jpg", group_id="bad") + + +# --- utilities -------------------------------------------------------------- + + +def test_get_wallet_address(client): + addr = client.get_wallet_address() + assert addr.startswith("0x") + assert len(addr) == 42 diff --git a/tests/unit/test_reasoning_usage.py b/tests/unit/test_reasoning_usage.py new file mode 100644 index 0000000..6a40153 --- /dev/null +++ b/tests/unit/test_reasoning_usage.py @@ -0,0 +1,101 @@ +import pytest + +from blockrun_llm.types import ChatUsage + + +def test_chat_usage_exposes_reasoning_breakdown_without_changing_totals() -> None: + usage = ChatUsage( + prompt_tokens=10, + completion_tokens=20, + total_tokens=30, + completion_tokens_details={"reasoning_tokens": 12}, + ) + + assert usage.reasoning_tokens == 12 + assert usage.completion_tokens == 20 + assert usage.model_dump(exclude_none=True)["completion_tokens_details"] == { + "reasoning_tokens": 12 + } + + +def test_chat_usage_reasoning_is_optional() -> None: + usage = ChatUsage(prompt_tokens=1, completion_tokens=2, total_tokens=3) + assert usage.reasoning_tokens is None + + +def test_chat_usage_reads_a_flat_reasoning_extra() -> None: + """`extra = "allow"` is what lets an upstream shape change reach callers. + A property of the same name wins over the extra, so the flat payload has + to be read explicitly or the number silently becomes None.""" + usage = ChatUsage(prompt_tokens=10, completion_tokens=20, total_tokens=30, reasoning_tokens=12) + + assert usage.reasoning_tokens == 12 + assert usage.model_dump()["reasoning_tokens"] == 12 + + +def test_nested_detail_wins_over_a_flat_extra() -> None: + usage = ChatUsage( + prompt_tokens=10, + completion_tokens=20, + total_tokens=30, + reasoning_tokens=99, + completion_tokens_details={"reasoning_tokens": 12}, + ) + + assert usage.reasoning_tokens == 12 + + +@pytest.mark.parametrize( + "detail", + [ + {}, + {"reasoning_tokens": None}, + {"reasoning_tokens": "12"}, + {"reasoning_tokens": 12.5}, + {"reasoning_tokens": -1}, + {"reasoning_tokens": True}, + {"audio_tokens": 0, "accepted_prediction_tokens": 0}, + ], + ids=["empty", "null", "string", "float", "negative", "bool", "other-keys-only"], +) +def test_unusable_values_read_as_absent_not_as_a_count(detail) -> None: + """A malformed count must not become one. `True` is an int subclass, so + without the bool check it would arrive as a token total of 1.""" + usage = ChatUsage( + prompt_tokens=1, completion_tokens=2, total_tokens=3, completion_tokens_details=detail + ) + + assert usage.reasoning_tokens is None + + +def test_zero_is_a_real_answer_not_a_missing_one() -> None: + """The gateway sends reasoning_tokens: 0 on non-reasoning turns — that is a + measurement, not an absence, and must not collapse to None.""" + usage = ChatUsage( + prompt_tokens=19, + completion_tokens=5, + total_tokens=24, + prompt_tokens_details={"audio_tokens": 0, "cached_tokens": 0}, + completion_tokens_details={ + "accepted_prediction_tokens": 0, + "audio_tokens": 0, + "reasoning_tokens": 0, + "rejected_prediction_tokens": 0, + }, + ) + + assert usage.reasoning_tokens == 0 + + +def test_reasoning_is_not_added_on_top_of_the_completion_total() -> None: + """Documented invariant: reasoning tokens are already inside + completion_tokens. A caller summing both would over-report spend.""" + usage = ChatUsage( + prompt_tokens=10, + completion_tokens=20, + total_tokens=30, + completion_tokens_details={"reasoning_tokens": 12}, + ) + + assert usage.reasoning_tokens <= usage.completion_tokens + assert usage.prompt_tokens + usage.completion_tokens == usage.total_tokens diff --git a/tests/unit/test_response_cost.py b/tests/unit/test_response_cost.py new file mode 100644 index 0000000..f89587a --- /dev/null +++ b/tests/unit/test_response_cost.py @@ -0,0 +1,57 @@ +"""ChatResponse carries the real per-call x402 charge. + +These lock in that the client attaches ``cost_usd`` to every chat completion so +downstream consumers (e.g. blockrun-litellm) can report the actual wallet +deduction instead of a token×list-price estimate. The free/200-first path must +report exactly 0.0 (not a stale prior charge). +""" + +from unittest.mock import MagicMock, patch + +from blockrun_llm import LLMClient + +TEST_KEY = "0x" + "1" * 64 + +_CHAT_BODY = { + "id": "chatcmpl-abc", + "object": "chat.completion", + "created": 1_700_000_000, + "model": "openai/gpt-5.5", + "choices": [ + { + "index": 0, + "message": {"role": "assistant", "content": "pong"}, + "finish_reason": "stop", + } + ], + "usage": {"prompt_tokens": 10, "completion_tokens": 1, "total_tokens": 11}, +} + + +@patch("blockrun_llm.client.httpx.Client") +def test_free_call_attaches_zero_cost(mock_client_class): + """A 200 on the first attempt = no payment required => cost_usd == 0.0.""" + mock_client = MagicMock() + mock_client_class.return_value = mock_client + resp200 = MagicMock(status_code=200) + resp200.json.return_value = _CHAT_BODY + mock_client.post.return_value = resp200 + + client = LLMClient(private_key=TEST_KEY) + result = client.chat_completion( + model="openai/gpt-5.5", messages=[{"role": "user", "content": "hi"}] + ) + + assert result.cost_usd == 0.0 + assert result.settlement is None + # cost_usd survives model_dump so the OpenAI-shaped payload carries it. + assert result.model_dump(exclude_none=True)["cost_usd"] == 0.0 + + +def test_chatresponse_cost_fields_default_none(): + from blockrun_llm.types import ChatResponse + + r = ChatResponse(id="x", object="chat.completion", created=0, model="m", choices=[]) + assert r.cost_usd is None and r.settlement is None + # absent by default (exclude_none) — no phantom $0 on objects we didn't charge + assert "cost_usd" not in r.model_dump(exclude_none=True) diff --git a/tests/unit/test_retired_endpoints.py b/tests/unit/test_retired_endpoints.py new file mode 100644 index 0000000..44bde7f --- /dev/null +++ b/tests/unit/test_retired_endpoints.py @@ -0,0 +1,59 @@ +"""Helpers for endpoints Predexon retired must fail fast, not silently 410. + +Probed upstream 2026-08-04 (3 runs each): /v1/pm/markets, /v1/pm/markets/listings +and /v1/pm/outcomes/{id} all return + 410 "This endpoint has been sunset as of 2026-07-20. Market matching is + discontinued." +The helpers are kept rather than deleted so upgrading does not break imports; +this pins that they raise instead of quietly costing a round trip. +""" + +import pytest + +from blockrun_llm import LLMClient, RetiredEndpointError +from blockrun_llm.client import AsyncLLMClient + + +def _bare(cls): + """Instance without running __init__ — no wallet or network needed.""" + return cls.__new__(cls) + + +@pytest.mark.parametrize( + "method,args", + [ + ("pm_markets", ()), + ("pm_listings", ()), + ("pm_outcome", ("PXM-12345",)), + ], +) +def test_sync_helpers_raise(method, args): + with pytest.raises(RetiredEndpointError, match="2026-07-20"): + getattr(_bare(LLMClient), method)(*args) + + +@pytest.mark.parametrize( + "method,args", + [ + ("pm_markets", ()), + ("pm_listings", ()), + ("pm_outcome", ("PXM-12345",)), + ], +) +@pytest.mark.asyncio +async def test_async_helpers_raise(method, args): + # An async def raises on await, not on call — but still before any network + # I/O, which is the point: no paid round trip to learn it is gone. + with pytest.raises(RetiredEndpointError, match="2026-07-20"): + await getattr(_bare(AsyncLLMClient), method)(*args) + + +def test_message_points_at_the_replacement(): + with pytest.raises(RetiredEndpointError) as exc: + _bare(LLMClient).pm_markets() + assert "markets/search" in str(exc.value) + + +def test_surviving_helper_is_untouched(): + # markets/search survived the sunset (422 on a missing q, i.e. alive). + assert not (getattr(LLMClient, "pm_wallet_identity", None) is None) diff --git a/tests/unit/test_retry_after.py b/tests/unit/test_retry_after.py new file mode 100644 index 0000000..c99417a --- /dev/null +++ b/tests/unit/test_retry_after.py @@ -0,0 +1,226 @@ +"""``Retry-After`` has to survive the SDK boundary, on every client. + +The gateway sets the header deliberately — it is what turns a refused request +into a caller that waits instead of one that spins against the limit. It +matters most on the account rail, where limits are per key and a 429 is the +normal way a busy customer is asked to slow down. + +The table below is the point of this file: one 429 fixture replayed through +every public service method, asserting the header arrives on the raised error. +A raise site that forgets it fails here rather than in a customer's retry loop. +""" + +from __future__ import annotations + +import json +from unittest.mock import patch + +import httpx +import pytest + +from blockrun_llm import ( + ImageClient, + LLMClient, + MusicClient, + PhoneClient, + PortraitClient, + PriceClient, + RealFaceClient, + RpcClient, + SearchClient, + SpeechClient, + SurfClient, + VideoClient, + VoiceClient, +) +from blockrun_llm.types import APIError, retry_after_of + +KEY = "brk_live_retry_after_fixture" +PHONE = "+12025550123" +WALLET = "0x" + "01" * 20 +RETRY_AFTER = "17" + +# (class, method, args, kwargs) — every public entry point that can surface a 429. +CASES = [ + (LLMClient, "chat", ("openai/gpt-5.2", "hi"), {}), + (LLMClient, "list_models", (), {}), + (ImageClient, "generate", ("test",), {}), + (VideoClient, "generate", ("test",), {"duration_seconds": 5}), + (MusicClient, "generate", ("test music",), {}), + (SpeechClient, "generate", ("hello",), {}), + (SpeechClient, "sound_effect", ("rain",), {}), + (SpeechClient, "list_voices", (), {}), + (VoiceClient, "call", (PHONE, "Read a test message"), {}), + (VoiceClient, "get_status", ("fixture-call",), {}), + (PhoneClient, "lookup", (PHONE,), {}), + (PhoneClient, "lookup_fraud", (PHONE,), {}), + (PhoneClient, "buy_number", (), {"area_code": "202"}), + (PhoneClient, "renew_number", (PHONE,), {}), + (PhoneClient, "list_numbers", (), {}), + (PhoneClient, "release_number", (PHONE,), {}), + (PortraitClient, "enroll", ("fixture", "https://example.com/test.png"), {}), + (PortraitClient, "list_portraits", (WALLET,), {}), + (RealFaceClient, "init", ("fixture",), {}), + (RealFaceClient, "status", ("legacy_rf_123",), {}), + (RealFaceClient, "enroll", ("fixture", "https://example.com/t.png", "legacy_rf_123"), {}), + (RealFaceClient, "list_realfaces", (WALLET,), {}), + (SearchClient, "search", ("test",), {}), + (SurfClient, "get", ("market/ranking",), {}), + (SurfClient, "post", ("onchain/sql", {"query": "SELECT 1"}), {}), + (PriceClient, "price", ("crypto", "BTC-USD"), {}), + (PriceClient, "price", ("stocks", "AAPL"), {"market": "us"}), + (PriceClient, "history", ("crypto", "BTC-USD"), {"from_ts": 1, "to_ts": 2}), + (RpcClient, "call", ("solana", "getSlot"), {}), + (RpcClient, "batch", ("base", [{"method": "eth_blockNumber"}]), {}), +] + + +@pytest.mark.parametrize("case", CASES, ids=[f"{c[0].__name__}.{c[1]}" for c in CASES]) +def test_a_429_carries_retry_after_to_the_caller(case, monkeypatch): + cls, method, args, kwargs = case + monkeypatch.setenv("BLOCKRUN_API_KEY", KEY) + monkeypatch.setenv("BLOCKRUN_WALLET_KEY", "must-not-read-wallet") + + def handler(request): + return httpx.Response( + 429, + json={"error": {"code": "rate_limited", "message": "slow down"}}, + headers={"retry-after": RETRY_AFTER}, + ) + + with patch("blockrun_llm.wallet.load_wallet", side_effect=AssertionError("wallet read")): + client = cls() + client._client.close() + client._client = httpx.Client( + headers=client._client.headers, transport=httpx.MockTransport(handler) + ) + try: + with pytest.raises(APIError) as failure: + getattr(client, method)(*args, **kwargs) + assert failure.value.status_code == 429 + assert failure.value.retry_after == RETRY_AFTER + assert failure.value.retry_after_seconds == 17.0 + finally: + client.close() + + +class TestRetryAfterParsing: + def test_the_raw_header_is_kept_verbatim(self): + err = APIError("rate limited", 429, None, retry_after="17") + assert err.retry_after == "17" + assert err.retry_after_seconds == 17.0 + + def test_absent_header_is_none_not_zero(self): + """Zero would read as "retry immediately", which is the opposite of + what an unknown wait means.""" + err = APIError("boom", 500) + assert err.retry_after is None + assert err.retry_after_seconds is None + + @pytest.mark.parametrize( + "raw", + ["Wed, 21 Oct 2026 07:28:00 GMT", "soon", "", " ", "-5"], + ids=["http-date", "garbage", "empty", "whitespace", "negative"], + ) + def test_unparseable_delays_do_not_become_a_number(self, raw): + """The HTTP-date form is legal and this SDK does not translate it. A + caller sleeping on a fabricated number is worse than one that knows it + has to decide for itself.""" + err = APIError("rate limited", 429, None, retry_after=raw) + assert err.retry_after_seconds is None + + def test_a_date_header_is_still_handed_back_raw(self): + raw = "Wed, 21 Oct 2026 07:28:00 GMT" + err = APIError("rate limited", 429, None, retry_after=raw) + assert err.retry_after == raw + + def test_fractional_seconds_survive(self): + assert APIError("x", 429, None, retry_after="0.5").retry_after_seconds == 0.5 + + +class TestRetryAfterOf: + def test_reads_the_header_case_insensitively(self): + resp = httpx.Response(429, headers={"Retry-After": "30"}) + assert retry_after_of(resp) == "30" + + def test_missing_header_is_none(self): + assert retry_after_of(httpx.Response(429)) is None + + def test_an_object_without_headers_does_not_explode(self): + """This runs inside an error path. Raising here would replace the real + failure with an AttributeError about the failure.""" + + class Bare: + status_code = 429 + + assert retry_after_of(Bare()) is None + + def test_blank_header_reads_as_absent(self): + assert retry_after_of(httpx.Response(429, headers={"retry-after": " "})) is None + + +def test_from_response_keeps_status_body_and_header(): + resp = httpx.Response(429, json={"error": "limited"}, headers={"retry-after": RETRY_AFTER}) + err = APIError.from_response(resp, "Request failed", json.loads(resp.text)) + assert (err.status_code, err.retry_after, err.response) == ( + 429, + RETRY_AFTER, + {"error": "limited"}, + ) + + +def test_the_signature_stays_backwards_compatible(): + """Every pre-existing call site passes three positional arguments and must + keep working untouched.""" + err = APIError("boom", 502, {"error": "upstream"}) + assert (err.status_code, err.response, err.retry_after) == (502, {"error": "upstream"}, None) + + +class TestUpstreamMessageReachesTheCaller: + """The gateway writes the one line worth reading; the SDK used to drop it. + + Raise sites build their message from the status code and stash the + sanitized body on `.response`, so a free-tier 429 printed as + `API error: 429` while the body said what to actually do about it. Read + against a live gateway, that difference is a caller who retries correctly + versus one who concludes their paid key is being throttled. + """ + + LIVE = ( + "Free tier rate limit reached (30 requests/minute per IP). " + "Retry after 10s, or use a paid model" + ) + + def test_the_gateway_explanation_lands_in_str(self): + err = APIError("API error: 429", 429, {"message": self.LIVE, "code": "rate_limit"}) + assert self.LIVE in str(err) + + def test_the_body_is_still_there_untouched(self): + body = {"message": self.LIVE, "code": "rate_limit"} + assert APIError("API error: 429", 429, body).response == body + + @pytest.mark.parametrize( + "placeholder", + ["API request failed", "Request failed", "Stream request failed", " ", ""], + ) + def test_sanitizer_placeholders_are_not_appended(self, placeholder): + """These are what the sanitizer emits when the body carried nothing. + Appending one restates the status code and buries the real prefix.""" + err = APIError("Image request: HTTP 500", 500, {"message": placeholder}) + assert str(err) == "Image request: HTTP 500" + + def test_an_already_included_message_is_not_repeated(self): + err = APIError("Upstream said: boom", 500, {"message": "boom"}) + assert str(err) == "Upstream said: boom" + + def test_a_bodyless_error_is_unchanged(self): + assert str(APIError("stream probe exhausted retries", 0, None)) == ( + "stream probe exhausted retries" + ) + + def test_a_non_string_message_is_ignored(self): + assert str(APIError("API error: 500", 500, {"message": {"nested": 1}})) == "API error: 500" + + def test_status_and_retry_after_survive_the_rewrite(self): + err = APIError("API error: 429", 429, {"message": self.LIVE}, retry_after="10") + assert (err.status_code, err.retry_after, err.retry_after_seconds) == (429, "10", 10.0) diff --git a/tests/unit/test_router_adapter.py b/tests/unit/test_router_adapter.py new file mode 100644 index 0000000..b7cb1da --- /dev/null +++ b/tests/unit/test_router_adapter.py @@ -0,0 +1,260 @@ +""" +Tests for the BlockRun host glue around Router Core. + +These cover what ``router_adapter`` adds on top of the product-neutral core: +catalog id resolution, the x402 payment floor, capacity filtering against the +whole conversation, and the SDK-only ``free`` profile. +""" + +from __future__ import annotations + +import pytest + +from blockrun_llm.router import route +from blockrun_llm.router_adapter import ( + BASE_MINIMUM_PAYMENT_USD, + FREE_TIERS, + routing_profile_for_model, + routing_text, +) +from blockrun_llm.router_core import DEFAULT_ROUTING_CONFIG +from blockrun_llm.types import RoutingDecision + +# The free chat models that answer as themselves, not via a gateway redirect. +# Verified with a two-pass model-echo probe on 2026-08-31; keep in step with +# router_adapter.FREE_TIERS. +FREE_MODELS = [ + "nvidia/nemotron-3.5-lightning", + "nvidia/nemotron-3-nano-30b", + "nvidia/llama-3.2-11b-vision", + "cohere/north-mini-code", + "poolside/laguna-xs-2.1", +] + + +def _price(input_price: float, output_price: float, flat_price: float = 0) -> dict[str, float]: + return { + "input_price": input_price, + "output_price": output_price, + "flat_price": flat_price, + } + + +CATALOG = { + "google/gemini-2.5-flash": _price(0.15, 0.6), + "google/gemini-2.5-flash-lite": _price(0.1, 0.4), + "google/gemini-3.5-flash": _price(0.5, 3), + "google/gemini-3-flash-preview": _price(0.5, 3), + "google/gemini-3.1-flash-lite": _price(0.25, 1.5), + "google/gemini-3.1-pro": _price(1.25, 10), + "openai/gpt-5.4-nano": _price(0.2, 1.25), + "openai/gpt-5-mini": _price(0.25, 2), + "openai/gpt-5.3-codex": _price(1.75, 14), + "anthropic/claude-opus-4.7": _price(5, 25), + "anthropic/claude-sonnet-5": _price(3, 15), + "anthropic/claude-fable-5": _price(10, 50), + "deepseek/deepseek-chat": _price(0.2, 0.4), + "deepseek/deepseek-v4-pro": _price(0.435, 0.87), + "moonshot/kimi-k2.7": _price(0.95, 4), + "xai/grok-4-1-fast-reasoning": _price(0.2, 0.5), + "xai/grok-4-fast-non-reasoning": _price(0.2, 0.5), + **{model: _price(0, 0) for model in FREE_MODELS}, +} + + +class TestCatalogResolution: + def test_heads_eco_with_the_gateway_native_free_tier(self): + # Since d7bc10c the chains carry gateway-native nvidia/* ids directly, + # so the adapter's free/*->nvidia/* mapping branch is dormant with the + # current pin. It stays because pins move independently; the dropped- + # unpriced-ids test below keeps the drop path honest. (Mirrors the + # TypeScript SDK's retargeting of the same guard.) + catalog = {**CATALOG, "nvidia/nemotron-3.5-lightning": _price(0, 0)} + + decision = route("hi", None, 512, catalog, "eco") + + assert "nvidia/nemotron-3.5-lightning" in [decision["model"], *decision["fallbacks"]] + assert not any( + model.startswith("free/") for model in [decision["model"], *decision["fallbacks"]] + ) + + def test_drops_free_ids_the_catalog_cannot_price(self): + # No nvidia/gpt-oss-* rows here: those ids are hidden from /v1/models, + # and an unmapped free/* id would draw a hard, non-transient 400. + decision = route("hi", None, 512, CATALOG, "eco") + + assert not any( + model.startswith("free/") for model in [decision["model"], *decision["fallbacks"]] + ) + assert decision["model"] in CATALOG + + def test_candidates_lead_with_the_selected_model_and_fallbacks_follow(self): + decision = route("What is 2+2?", None, 512, CATALOG) + + assert decision["candidates"][0] == decision["model"] + assert decision["fallbacks"] == decision["candidates"][1:] + assert decision["model"] not in decision["fallbacks"] + + +class TestCostMetadata: + def test_applies_the_base_chain_payment_floor_to_paid_models(self): + decision = route("What is 2+2?", None, 16, CATALOG) + + assert decision["cost_estimate"] == pytest.approx(BASE_MINIMUM_PAYMENT_USD) + + def test_never_floors_a_free_model_up_to_the_paid_minimum(self): + decision = route("What is 2+2?", None, 512, CATALOG, "free") + + assert decision["cost_estimate"] == 0 + assert decision["savings"] == pytest.approx(1.0) + + def test_premium_profile_reports_no_savings(self): + decision = route("Design a distributed ledger", None, 1024, CATALOG, "premium") + + assert decision["savings"] == 0 + + +class TestCapacityFiltering: + def test_drops_candidates_that_cannot_hold_the_full_conversation(self): + # 8k output is above several small-output models' ceiling. + decision = route("Explain this architecture", None, 20_000, CATALOG) + + assert "xai/grok-4-fast-non-reasoning" not in decision["candidates"] + + def test_keeps_models_absent_from_the_capability_snapshot(self): + catalog = {**CATALOG, "acme/experimental-1": _price(0.1, 0.1)} + config = { + **DEFAULT_ROUTING_CONFIG, + "strategy": "rules", + "tiers": { + tier: {"primary": "acme/experimental-1", "fallback": []} + for tier in DEFAULT_ROUTING_CONFIG["tiers"] + }, + } + from blockrun_llm.router_adapter import route_with_catalog + + decision = route_with_catalog("hi", None, 512, catalog, config=config) + + assert decision["model"] == "acme/experimental-1" + + +class TestFreeProfile: + @pytest.mark.parametrize( + "prompt", + [ + "What is 2+2?", + "Prove the theorem step by step using mathematical induction", + "Refactor this TypeScript function and explain the tradeoffs", + "A" * 5_000, + ], + ) + def test_never_selects_a_billable_model(self, prompt): + decision = route(prompt, None, 512, CATALOG, "free") + + for model in [decision["model"], *decision["fallbacks"]]: + assert CATALOG[model]["input_price"] == 0 + assert CATALOG[model]["output_price"] == 0 + + def test_every_free_tier_keeps_real_fallback_depth(self): + # Membership alone did not catch the 2026-08 rot: the table stayed + # internally consistent while the gateway retired four of its five ids, + # leaving every tier on one model with no fallback. Depth is the signal + # that survives that, so assert it per tier and across the table. + for name, tier in FREE_TIERS.items(): + candidates = [tier["primary"], *tier["fallback"]] + assert len(set(candidates)) >= 3, f"{name} has no fallback depth: {candidates}" + + used = {m for tier in FREE_TIERS.values() for m in [tier["primary"], *tier["fallback"]]} + assert used == set(FREE_MODELS), sorted(set(FREE_MODELS) ^ used) + + def test_every_free_tier_entry_is_live_in_the_catalog(self): + # The previous hand-maintained table rotted silently when NVIDIA EOL'd + # its early free lineup; this asserts the replacement points at models + # the catalog still prices. + for tier in FREE_TIERS.values(): + for model in [tier["primary"], *tier["fallback"]]: + assert model in FREE_MODELS, model + + def test_uses_the_rules_strategy_so_paid_evidence_models_cannot_leak_in(self): + decision = route( + "Fix the TypeScript payment retry bug, run tests, and update the patch.", + None, + 4096, + CATALOG, + "free", + ) + + assert decision["method"] == "rules" + assert "openai/gpt-5.3-codex" not in decision["candidates"] + + +class TestSdkDecisionShape: + def test_the_decision_parses_into_the_public_pydantic_model(self): + decision = route( + "Which answer is correct?\nA. One\nB. Two\nC. Three\nD. Four", None, 512, CATALOG + ) + + parsed = RoutingDecision(**decision) + + assert parsed.model == decision["model"] + assert parsed.method == "portfolio" + assert parsed.router_version == "v3-portfolio" + assert parsed.task_type == "reasoning_mcq" + assert parsed.candidates[0] == parsed.model + assert parsed.candidate_scores + assert parsed.profile == "auto" + + def test_the_free_profile_decision_also_parses(self): + parsed = RoutingDecision(**route("hi", None, 512, CATALOG, "free")) + + assert parsed.method == "rules" + assert parsed.task_type is None + + +class TestRoutingText: + def test_reads_the_whole_transcript_for_capacity_and_the_last_user_turn(self): + view = routing_text( + [ + {"role": "system", "content": "You are terse."}, + {"role": "user", "content": "hello"}, + {"role": "assistant", "content": "hi"}, + {"role": "user", "content": "and now?"}, + ] + ) + + assert view["prompt"] == "and now?" + assert view["system_prompt"] == "You are terse." + assert view["conversation_chars"] == len("You are terse.") + len("hello") + 2 + len( + "and now?" + ) + assert view["has_vision"] is False + + def test_detects_image_parts(self): + view = routing_text( + [ + { + "role": "user", + "content": [ + {"type": "text", "text": "what is this?"}, + {"type": "image_url", "image_url": {"url": "data:image/png;base64,AAA"}}, + ], + } + ] + ) + + assert view["has_vision"] is True + assert view["conversation_chars"] == len("what is this?") + + +class TestVirtualModelIds: + @pytest.mark.parametrize( + ("model", "expected"), + [ + ("blockrun/auto", "auto"), + ("BlockRun/Eco", "eco"), + ("blockrun/premium", "premium"), + ("google/gemini-3.5-flash", None), + ], + ) + def test_maps_virtual_ids_to_profiles(self, model, expected): + assert routing_profile_for_model(model) == expected diff --git a/tests/unit/test_router_core.py b/tests/unit/test_router_core.py new file mode 100644 index 0000000..8145d7d --- /dev/null +++ b/tests/unit/test_router_core.py @@ -0,0 +1,1379 @@ +""" +Parity tests for the Router Core port. + +Every case here is a 1:1 port of an upstream ``@blockrun/router-core`` vitest +case (``portfolio.test.ts``, ``selector.test.ts``, ``strategy.test.ts``, +``tool-intent.test.ts``, ``unavailable-models.test.ts`` at commit +``5ee7c23``). They are the regression guard +that the Python port keeps choosing the same models as the TypeScript SDK — +when upstream is re-synced, re-port these alongside the source. +""" + +from __future__ import annotations + +import math +from datetime import datetime, timezone + +import pytest + +from blockrun_llm.router_core import ( + DEFAULT_ROUTING_CONFIG, + RulesStrategy, + apply_unavailable_models, + calculate_model_cost, + filter_by_exclude_list, + filter_by_tool_calling, + filter_candidates_by_capacity, + get_strategy, + infer_tool_requirement, + register_strategy, + route, +) +from blockrun_llm.router_core.selector import select_model + + +def _price(input_price: float, output_price: float) -> dict[str, float]: + return {"input_price": input_price, "output_price": output_price} + + +PORTFOLIO_PRICING = { + "anthropic/claude-sonnet-4.6": _price(3, 15), + "anthropic/claude-sonnet-5": _price(3, 15), + "anthropic/claude-opus-5": _price(5, 25), + "anthropic/claude-opus-4.8": _price(5, 25), + "openai/gpt-5.3-codex": _price(1.75, 14), + "openai/gpt-5-mini": _price(0.25, 2), + "openai/gpt-4.1": _price(2, 8), + "google/gemini-3.5-flash": _price(0.5, 3), + "google/gemini-3-flash-preview": _price(0.5, 3), + "google/gemini-3.1-pro": _price(2, 12), + "moonshot/kimi-k3": _price(3, 15), + "deepseek/deepseek-v4-pro": _price(0.435, 0.87), + "xai/grok-4.5": _price(2, 10), + "qwen/qwen3.7-max": _price(1.475, 4.425), + "zai/glm-5.2": _price(1.4, 4.4), + "moonshot/kimi-k2.7": _price(0.95, 4), + "moonshot/kimi-k2.6": _price(0.95, 4), + "moonshot/kimi-k2.5": _price(0.6, 3), + "xai/grok-4-1-fast-non-reasoning": _price(0.2, 0.5), + "openai/gpt-4o-mini": _price(0.15, 0.6), + "deepseek/deepseek-chat": _price(0.2, 0.4), + "free/seed-oss-36b": _price(0, 0), +} + +STRATEGY_PRICING = { + "moonshot/kimi-k2.5": _price(0.5, 2.4), + "moonshot/kimi-k2.6": _price(0.95, 4.0), + "anthropic/claude-opus-4.6": _price(5, 25), + "anthropic/claude-opus-4.7": _price(5, 25), + "anthropic/claude-opus-4.8": _price(5, 25), + "google/gemini-2.5-flash": _price(0.15, 0.6), + "google/gemini-2.5-flash-lite": _price(0.1, 0.4), + "deepseek/deepseek-chat": _price(0.14, 0.28), + "anthropic/claude-sonnet-4.6": _price(3, 15), + "google/gemini-3.1-pro": _price(1.25, 10), + "google/gemini-3.5-flash": _price(0.5, 3), + "google/gemini-3-flash-preview": _price(0.5, 3), + "xai/grok-4.5": _price(2, 6), + "anthropic/claude-sonnet-5": _price(3, 15), + "deepseek/deepseek-v4-pro": _price(0.435, 0.87), + "moonshot/kimi-k3": _price(3, 15), + "xai/grok-4-1-fast-reasoning": _price(0.2, 0.5), + "nvidia/gpt-oss-120b": _price(0, 0), + "nvidia/gpt-oss-20b": _price(0, 0), + "nvidia/deepseek-v3.2": _price(0, 0), + "nvidia/deepseek-v4-pro": _price(0, 0), + "nvidia/deepseek-v4-flash": _price(0, 0), + "nvidia/qwen3-coder-480b": _price(0, 0), + "nvidia/glm-4.7": _price(0, 0), + "nvidia/llama-4-maverick": _price(0, 0), + "nvidia/qwen3-next-80b-a3b-thinking": _price(0, 0), + "nvidia/mistral-small-4-119b": _price(0, 0), + "nvidia/nemotron-3-nano-omni-30b-a3b-reasoning": _price(0, 0), + "nvidia/qwen3-next-80b-a3b-instruct": _price(0, 0), + "nvidia/seed-oss-36b": _price(0, 0), + "nvidia/mistral-nemotron": _price(0, 0), + "nvidia/step-3.7-flash": _price(0, 0), + "nvidia/nemotron-nano-9b-v2": _price(0, 0), + "nvidia/nemotron-nano-12b-v2-vl": _price(0, 0), +} + +BASE_OPTIONS = {"config": DEFAULT_ROUTING_CONFIG, "model_pricing": STRATEGY_PRICING} + +TERMINAL_TOOLS = ["TerminalExec", "TerminalInspect", "TerminalSendKeys"] + +AIRLINE_TOOLS = [ + "get_user_details", + "get_reservation_details", + "search_direct_flight", + "update_reservation_flights", + "cancel_reservation", + "book_reservation", + "update_reservation_baggages", +] + +RETURN_TOOLS = [ + "get_order_details", + "return_delivered_order_items", + "transfer_to_human_agents", +] + +KIMI_MODELS = ("moonshot/kimi-k2.7", "moonshot/kimi-k2.6", "moonshot/kimi-k2.5") + + +def _portfolio(prompt: str, max_output_tokens: int, **options): + return route( + prompt, + None, + max_output_tokens, + {"config": DEFAULT_ROUTING_CONFIG, "model_pricing": PORTFOLIO_PRICING, **options}, + ) + + +def _terminal(prompt: str, max_output_tokens: int = 4096): + return _portfolio( + prompt, + max_output_tokens, + routing_profile="auto", + has_tools=True, + requires_tools=True, + tool_count=3, + tool_names=TERMINAL_TOOLS, + ) + + +def _scored_models(decision) -> list[str]: + return [row["model"] for row in decision.get("candidate_scores", [])] + + +# ─── portfolio.test.ts ─── + + +class TestPortfolioStrategy: + def test_keeps_only_tool_capable_models_for_a_coding_agent_request(self): + decision = _portfolio( + "Fix the TypeScript payment retry bug, run tests, and update the patch.", + 4096, + has_tools=True, + ) + + assert decision["method"] == "portfolio" + assert decision["task_type"] == "code_agent" + assert decision["model"] == "openai/gpt-5-mini" + assert decision["model"] in decision["candidates"] + assert "openai/gpt-5.3-codex" in decision["candidates"] + assert decision["model"] not in KIMI_MODELS + assert "google/gemini-3.1-pro" not in decision["candidates"] + + def test_classifies_a_non_code_function_call_as_a_tool_agent(self): + decision = _portfolio("Use the lookup_order tool for order B-42.", 256, has_tools=True) + + assert decision["task_type"] in ("tool_agent", "tool_agent_parallel") + assert decision["model"] == "anthropic/claude-sonnet-5" + assert decision["model"] in decision["candidates"] + assert "google/gemini-3.5-flash" in decision["candidates"] + assert decision["model"] not in KIMI_MODELS + + @pytest.mark.parametrize( + "prompt", + [ + "请问北京的当前天气状况如何?还有,上海的天气情况是怎样的?", + ( + "For breakfast I had a 12 ounce iced coffee and a banana.\n\n" + "For lunch I had a quesadilla.\n\n" + "Breakfast four ounces of asparagus and two eggs." + ), + "¿Cuáles son las condiciones del clima en Cancún, Playa del Carmen y Tulum?", + "Could you tell me the current temperature in Boston, MA and San Francisco, please?", + "What's the snow like in the two cities of Paris and Bordeaux?", + "What's cost of 2 and 4 gb ram machine on aws ec2 with one CPU?", + "能帮我查一下中国广州市和北京市现在的天气状况吗?请使用公制单位。", + ( + "Could you provide the latest news for Paris, France, and also for " + "Letterkenny, Ireland?" + ), + "I'd like to change my food order to a salad, and for the drink, update it to coffee.", + ], + ) + def test_routes_repeated_single_tool_requests_to_the_parallel_specialist(self, prompt): + decision = _portfolio( + prompt, + 600, + routing_profile="auto", + has_tools=True, + requires_tools=True, + tool_count=1, + ) + + assert decision["task_type"] == "tool_agent_parallel" + assert decision["model"] == "anthropic/claude-opus-4.8" + + def test_keeps_an_ordinary_single_lookup_on_the_standard_tool_agent_path(self): + decision = _portfolio( + "Use lookup_order for order B-42.", + 256, + routing_profile="auto", + has_tools=True, + requires_tools=True, + tool_count=1, + ) + + assert decision["task_type"] == "tool_agent" + assert decision["model"] == "anthropic/claude-sonnet-5" + assert "google/gemini-3.5-flash" in decision["candidates"] + + def test_keeps_deep_multi_clue_web_research_on_sonnet_5(self): + decision = _portfolio( + "Research the following clues across multiple public sources and identify the country.", + 2048, + routing_profile="auto", + has_tools=True, + requires_tools=True, + tool_count=2, + tool_names=["web_search", "web_fetch"], + ) + + assert decision["task_type"] in ("tool_agent", "tool_agent_parallel") + assert decision["model"] == "anthropic/claude-sonnet-5" + assert "deepWebResearch=true" in decision["reasoning"] + assert decision["candidates"][:3] == [ + "anthropic/claude-sonnet-5", + "openai/gpt-5-mini", + "google/gemini-3.5-flash", + ] + assert "candidates=" in decision["reasoning"] + + def test_keeps_a_routine_web_lookup_on_sonnet_5(self): + decision = _portfolio( + "Search the official documentation for the current API timeout setting.", + 1024, + routing_profile="auto", + has_tools=True, + requires_tools=True, + tool_count=2, + tool_names=["web_search", "web_fetch"], + ) + + assert decision["model"] == "anthropic/claude-sonnet-5" + assert "deepWebResearch=false" in decision["reasoning"] + + def test_keeps_a_known_cross_reservation_batch_on_the_cost_controlled_model(self): + decision = _portfolio( + "Hi! I’d like to make some changes to my bookings. I need to cancel two of my " + "upcoming reservations and upgrade another one to business class. " + "Can you help me with that?", + 4096, + routing_profile="auto", + has_tools=True, + requires_tools=True, + tool_count=7, + tool_names=AIRLINE_TOOLS, + ) + + assert decision["task_type"] in ("tool_agent", "tool_agent_parallel") + assert "agentRisk=high" in decision["reasoning"] + assert decision["model"] == "openai/gpt-5-mini" + + def test_promotes_conditional_global_airline_work_to_the_complex_band(self): + decision = _portfolio( + "Cancel all your future reservations that contain flights longer than 4 hours. " + "For flights under 3 hours, upgrade to business wherever possible.", + 4096, + routing_profile="auto", + has_tools=True, + requires_tools=True, + tool_count=7, + tool_names=AIRLINE_TOOLS, + ) + + assert "agentRisk=complex_high" in decision["reasoning"] + assert decision["model"] == "anthropic/claude-sonnet-5" + + @pytest.mark.parametrize( + "prompt", + [ + ( + "Create a file called hello.txt in the current directory. " + "Write Hello, world! to it and end with a newline." + ), + "Convert the file /app/data.csv into a Parquet file named /app/data.parquet.", + ( + "Create and run a server on port 3000 with a single GET endpoint /fib " + "that returns JSON." + ), + ( + "A script called 'process_data.sh' in the current directory won't run. " + "Figure out what's wrong and fix it so the script can run successfully." + ), + ], + ) + def test_uses_the_low_cost_code_agent_for_deterministic_local_terminal_work(self, prompt): + decision = _terminal(prompt) + + assert decision["task_type"] == "code_agent" + assert decision["model"] == "openai/gpt-5-mini" + assert "terminalCode=true" in decision["reasoning"] + + def test_promotes_a_multi_script_dependency_repair_to_the_strong_band(self): + decision = _terminal( + "There's a data processing pipeline in the current directory consisting of " + "multiple scripts that need to run in sequence. The main script 'run_pipeline.sh' " + "is failing to execute properly. Identify and fix all issues with the script files " + "and dependencies to make the pipeline run successfully." + ) + + assert decision["task_type"] == "tool_agent" + assert "agentRisk=complex_high" in decision["reasoning"] + assert decision["model"] == "anthropic/claude-sonnet-5" + + def test_promotes_a_cross_runtime_polyglot_artifact_to_the_strong_band(self): + decision = _terminal( + "Write one /app/main.c.rs polyglot file that must compile and run with both " + "rustc main.c.rs and gcc main.c.rs -o cmain." + ) + + assert decision["task_type"] == "code_agent" + assert "agentRisk=complex_high" in decision["reasoning"] + assert decision["model"] == "anthropic/claude-sonnet-5" + + def test_promotes_a_framework_checkpoint_port_to_the_strong_band(self): + decision = _terminal( + "Implement a command line tool programmed in C that runs inference using a " + "pre-trained PyTorch state_dict called simple_mnist.pth. The final output must be " + "a native cli_tool binary plus weights.json." + ) + + assert decision["task_type"] == "code_agent" + assert "agentRisk=complex_high" in decision["reasoning"] + assert decision["model"] == "anthropic/claude-sonnet-5" + + @pytest.mark.parametrize( + "prompt", + [ + ( + "Configure a git server over SSH and deploy two branches through Nginx HTTPS " + "with password authentication." + ), + ( + "Securely decommission the service: encrypt the archive with GPG, shred the " + "sensitive files, then delete them." + ), + ( + "Evaluate an embedding model with the MTEB benchmark and write the official " + "result file." + ), + "Inspect the chess board image and write the best move to a file.", + ( + "Create a JSON processor from three CSV inputs. Requirements: 1. Follow " + "schema.json. 2. Join departments and employees. 3. Calculate statistics." + ), + ], + ) + def test_keeps_complex_or_risky_terminal_operations_on_the_generic_agent_path(self, prompt): + decision = _terminal(prompt) + + assert decision["task_type"] != "code_agent" + assert "terminalCode=false" in decision["reasoning"] + + def test_keeps_codex_below_the_primary_band_for_security_sensitive_file_ops(self): + decision = _terminal( + "Please help me encrypt all the files I have in the data/ folder using rencrypt. " + "Use the most secure encryption and write the outputs to encrypted_data/ with the " + "same basenames." + ) + + assert decision["task_type"] == "tool_agent" + assert "terminalSafety=true" in decision["reasoning"] + assert decision["model"] == "anthropic/claude-sonnet-5" + assert "openai/gpt-5.3-codex" in decision["candidates"] + assert "openai/gpt-5.3-codex" not in _scored_models(decision) + + def test_admits_a_cost_controlled_strong_model_for_sensitive_multi_file_work(self): + decision = _terminal( + "Sanitize this git repository by replacing all AWS, GitHub, and Hugging Face API " + "keys with consistent placeholders across every affected file. Also, do not make " + "any other unnecessary changes to files without sensitive information." + ) + + assert decision["task_type"] == "tool_agent_parallel" + assert decision["model"] == "anthropic/claude-sonnet-5" + assert "anthropic/claude-sonnet-5" in decision["candidates"] + + @pytest.mark.parametrize( + "prompt", + [ + ( + "Reverse engineer the mystery binary, then write and compile image.c so it " + "produces the requested path-traced image." + ), + ( + "Create a local JSON server for Solana devnet with status, block, account, " + "transaction, and paginated program-account endpoints." + ), + ( + "Create a Solana devnet API whose transaction endpoint returns token transfers " + "with account, mint, and amount fields." + ), + ], + ) + def test_cost_controls_complex_terminal_work_that_is_not_safety_sensitive(self, prompt): + decision = _terminal(prompt) + + assert decision["task_type"] in ("tool_agent", "tool_agent_parallel", "code_agent") + assert decision["model"] == "openai/gpt-5-mini" + assert "terminalSafety=false" in decision["reasoning"] + + @pytest.mark.parametrize( + "prompt", + [ + ( + "Rotate the expired authentication token and update the bearer token used by " + "the production service." + ), + ( + "Replace every leaked API key and password in this repository without changing " + "unrelated files." + ), + ], + ) + def test_keeps_credential_bearing_terminal_work_safety_sensitive(self, prompt): + decision = _terminal(prompt) + + assert "terminalSafety=true" in decision["reasoning"] + assert decision["model"] != "openai/gpt-5-mini" + + def test_uses_the_high_risk_model_for_retail_order_tools(self): + decision = _portfolio( + "Exchange both items after I confirm the price difference.", + 512, + routing_profile="auto", + has_tools=True, + requires_tools=True, + tool_count=4, + tool_names=[ + "get_order_details", + "get_product_details", + "exchange_delivered_order_items", + "modify_pending_order_address", + ], + ) + + assert decision["task_type"] in ("tool_agent", "tool_agent_parallel") + assert decision["model"] == "deepseek/deepseek-v4-pro" + assert "openai/gpt-5-mini" in decision["candidates"] + + def test_uses_the_low_cost_model_for_one_local_retail_operation(self): + decision = _portfolio( + "Change the blue earbuds in order W5061109 to red after I confirm.", + 4096, + routing_profile="auto", + has_tools=True, + requires_tools=True, + tool_count=6, + tool_names=[ + "find_user_id_by_name_zip", + "get_order_details", + "get_product_details", + "modify_pending_order_items", + ], + ) + + assert decision["task_type"] == "tool_agent" + assert decision["model"] == "openai/gpt-5-mini" + + def test_keeps_global_retail_choices_on_the_high_risk_model(self): + decision = _portfolio( + "Exchange my tablet for the cheapest available variant in another order.", + 4096, + routing_profile="auto", + has_tools=True, + requires_tools=True, + tool_count=6, + tool_names=[ + "get_order_details", + "get_product_details", + "exchange_delivered_order_items", + ], + ) + + assert decision["model"] == "deepseek/deepseek-v4-pro" + + def test_uses_the_policy_specialist_for_a_refund_to_another_card(self): + decision = _portfolio( + "Return everything except the pet bed and refund it to my Amex card.", + 4096, + routing_profile="auto", + has_tools=True, + requires_tools=True, + tool_count=6, + tool_names=RETURN_TOOLS, + ) + + assert decision["task_type"] in ("tool_agent", "tool_agent_parallel") + assert decision["model"] == "openai/gpt-4.1" + assert "agentRisk=policy_exception" in decision["reasoning"] + + def test_keeps_a_single_comparative_send_back_on_the_low_cost_model(self): + decision = _portfolio( + "Send back the pricier one and get my money back on my credit card.", + 4096, + routing_profile="auto", + has_tools=True, + requires_tools=True, + tool_count=6, + tool_names=RETURN_TOOLS, + ) + + assert decision["model"] == "openai/gpt-5-mini" + assert "agentRisk=policy_exception_simple" in decision["reasoning"] + + def test_uses_the_policy_specialist_when_a_named_card_refund_covers_two_objects(self): + decision = _portfolio( + "Return these two skateboards and refund them to my credit card.", + 4096, + routing_profile="auto", + has_tools=True, + requires_tools=True, + tool_count=6, + tool_names=RETURN_TOOLS, + ) + + assert decision["model"] == "openai/gpt-4.1" + assert "agentRisk=policy_exception" in decision["reasoning"] + + def test_treats_a_simple_looking_retail_return_as_a_negotiated_high_risk_workflow(self): + decision = _portfolio( + "I want to return an office chair that arrived broken.", + 4096, + routing_profile="auto", + has_tools=True, + requires_tools=True, + tool_count=6, + tool_names=[ + "get_order_details", + "get_product_details", + "return_delivered_order_items", + "exchange_delivered_order_items", + ], + ) + + assert decision["task_type"] == "tool_agent" + assert decision["model"] == "deepseek/deepseek-v4-pro" + assert "agentRisk=high" in decision["reasoning"] + + def test_uses_the_cost_efficient_model_for_airline_tools(self): + decision = _portfolio( + "Change my flight after checking the reservation.", + 512, + routing_profile="auto", + has_tools=True, + requires_tools=True, + tool_count=3, + tool_names=[ + "get_reservation_details", + "search_direct_flight", + "update_reservation_flights", + ], + ) + + assert decision["task_type"] in ("tool_agent", "tool_agent_parallel") + assert decision["model"] == "openai/gpt-5-mini" + assert "anthropic/claude-sonnet-5" in decision["candidates"] + + def test_does_not_mistake_airline_cabin_class_for_a_code_agent_task(self): + decision = _portfolio( + "Move my flight to May 24 and upgrade all passengers to business class.", + 4096, + routing_profile="auto", + has_tools=True, + requires_tools=True, + tool_count=8, + tool_names=[ + "get_reservation_details", + "search_direct_flight", + "update_reservation_flights", + ], + ) + + assert decision["task_type"] != "code_agent" + assert decision["model"] == "openai/gpt-5-mini" + assert "agentRisk=high" in decision["reasoning"] + + def test_reserves_the_airline_specialist_for_global_itinerary_optimization(self): + decision = _portfolio( + "Show my gift card and certificate balances, then change my reservation to the " + "cheapest business round trip without changing the dates.", + 4096, + routing_profile="auto", + has_tools=True, + requires_tools=True, + tool_count=8, + tool_names=[ + "get_user_details", + "get_reservation_details", + "search_onestop_flight", + "cancel_reservation", + "book_reservation", + ], + ) + + assert decision["task_type"] != "code_agent" + assert decision["model"] == "anthropic/claude-sonnet-5" + assert "agentRisk=complex_high" in decision["reasoning"] + + def test_does_not_mistake_a_lookup_plus_explanation_for_parallel_tool_use(self): + decision = _portfolio( + "Get the weather for London and explain whether I need an umbrella.", + 256, + routing_profile="auto", + has_tools=True, + requires_tools=True, + tool_count=1, + tool_names=["get_current_weather"], + ) + + assert decision["task_type"] == "tool_agent" + + def test_uses_two_distinctive_visible_tool_names_as_a_multi_operation_signal(self): + decision = _portfolio( + "Add task draft release notes, then delete task obsolete draft.", + 256, + routing_profile="auto", + has_tools=True, + requires_tools=True, + tool_count=2, + tool_names=["add_task", "delete_task"], + ) + + assert decision["task_type"] == "tool_agent_parallel" + + def test_does_not_spend_upgrade_a_large_numbered_multi_tool_plan(self): + decision = _portfolio( + "Do all the following:\n1. Clone the repository.\n2. Analyze it.\n" + "3. Create Docker and Kubernetes files.\n4. Commit and push.", + 600, + routing_profile="auto", + has_tools=True, + requires_tools=True, + tool_count=7, + tool_names=[ + "clone_repo", + "analyze_repo", + "create_docker_file", + "create_kubernetes_yaml", + "commit_changes", + "push_changes", + "read_file", + ], + ) + + assert decision["task_type"] != "tool_agent_parallel" + assert decision["model"] != "anthropic/claude-opus-4.8" + + def test_detects_an_explicit_multi_object_request_with_a_distractor_tool(self): + decision = _portfolio( + "What's the weather like in the two cities of Boston and San Francisco?", + 600, + routing_profile="auto", + has_tools=True, + requires_tools=True, + tool_count=2, + ) + + assert decision["task_type"] == "tool_agent_parallel" + assert decision["model"] == "anthropic/claude-opus-4.8" + + def test_does_not_classify_ordinary_qa_as_a_tool_task(self): + decision = _portfolio( + "Which answer is correct?\nA. One\nB. Two\nC. Three\nD. Four", + 256, + has_tools=True, + requires_tools=False, + ) + + assert decision["task_type"] == "reasoning_mcq" + assert decision["profile"] == "auto" + assert decision["model"] == "google/gemini-3-flash-preview" + + def test_adds_current_long_context_models_instead_of_a_legacy_tier_chain(self): + decision = _portfolio("A" * 340_000, 1_024) + + assert decision["task_type"] == "long_context" + assert "deepseek/deepseek-v4-pro" not in _scored_models(decision) + assert "deepseek/deepseek-v4-pro" in decision["candidates"] + assert decision["model"] == "google/gemini-3.1-pro" + assert decision["model"] in decision["candidates"] + + def test_keeps_mandarin_extraction_in_the_source_language_affinity_band(self): + decision = _portfolio( + "只输出 JSON:从订单 A-17,数量 3,状态已发货中提取 orderId、quantity、status 三个字段。", + 256, + ) + + assert decision["task_type"] == "extraction" + assert decision["model"] == "moonshot/kimi-k3" + assert decision["candidates"][0] == "moonshot/kimi-k3" + + def test_does_not_promote_a_generic_recovery_fallback_without_task_affinity(self): + decision = _portfolio("Patch this API secret validation error.", 256) + + # DeepSeek Chat is a valid availability fallback in the SIMPLE tier, but + # is not an explicitly profiled code-edit specialist. It must not win the + # Auto ranking simply because it is inexpensive. + assert "deepseek/deepseek-chat" not in _scored_models(decision) + assert "deepseek/deepseek-chat" in decision["candidates"] + + def test_does_not_let_a_flash_lite_sibling_inherit_flash_task_affinity(self): + exact_name_config = { + **DEFAULT_ROUTING_CONFIG, + "tiers": { + tier: { + "primary": "google/gemini-2.5-flash", + "fallback": ["google/gemini-2.5-flash-lite"], + } + for tier in DEFAULT_ROUTING_CONFIG["tiers"] + }, + } + decision = route( + "Explain the deployment status.", + None, + 256, + { + "config": exact_name_config, + "model_pricing": { + "google/gemini-2.5-flash": _price(1, 1), + "google/gemini-2.5-flash-lite": _price(0.1, 0.1), + }, + }, + ) + + assert decision["candidates"][0] == "google/gemini-2.5-flash" + assert "google/gemini-2.5-flash-lite" in decision["candidates"] + assert "google/gemini-2.5-flash-lite" not in _scored_models(decision) + + def test_filters_models_that_cannot_satisfy_the_requested_output_length(self): + decision = _portfolio("Explain this architecture", 20_000) + + assert "xai/grok-4-fast-non-reasoning" not in decision["candidates"] + + def test_only_lets_fresh_performance_observations_influence_candidate_order(self): + two_candidate_config = { + **DEFAULT_ROUTING_CONFIG, + "tiers": { + tier: { + "primary": "xai/grok-4-1-fast-non-reasoning", + "fallback": ["openai/gpt-4o-mini"], + } + for tier in DEFAULT_ROUTING_CONFIG["tiers"] + }, + } + decision = route( + "Extract the fields as JSON", + None, + 512, + { + "config": two_candidate_config, + "model_pricing": { + "xai/grok-4-1-fast-non-reasoning": _price(1, 1), + "openai/gpt-4o-mini": _price(1, 1), + }, + "now": datetime(2026, 7, 21, tzinfo=timezone.utc), + "model_performance": { + "openai/gpt-4o-mini": { + "measured_at": "2026-07-21T00:00:00Z", + "latency_ms": 600, + "output_tokens_per_second": 250, + "intelligence_index": 50, + } + }, + }, + ) + + assert decision["task_type"] == "extraction" + assert decision["candidates"][0] == "openai/gpt-4o-mini" + + def test_treats_a_small_performance_probe_as_a_tie_breaker(self): + two_candidate_config = { + **DEFAULT_ROUTING_CONFIG, + "tiers": { + tier: { + "primary": "xai/grok-4-1-fast-non-reasoning", + "fallback": ["openai/gpt-4o-mini"], + } + for tier in DEFAULT_ROUTING_CONFIG["tiers"] + }, + } + decision = route( + "Explain the deployment status.", + None, + 512, + { + "config": two_candidate_config, + "model_pricing": { + "xai/grok-4-1-fast-non-reasoning": _price(1, 1), + "openai/gpt-4o-mini": _price(1, 1), + }, + "now": datetime(2026, 7, 21, tzinfo=timezone.utc), + "model_performance": { + "openai/gpt-4o-mini": { + "measured_at": "2026-07-21T00:00:00Z", + "latency_ms": 600, + "output_tokens_per_second": 250, + "intelligence_index": 50, + "samples": 1, + } + }, + }, + ) + + assert decision["candidates"][0] == "xai/grok-4-1-fast-non-reasoning" + + def test_ignores_a_malformed_performance_timestamp(self): + decision = _portfolio( + "Extract the fields as JSON", + 512, + now=datetime(2026, 7, 21, tzinfo=timezone.utc), + model_performance={ + "openai/gpt-4o-mini": { + "measured_at": "not-a-timestamp", + "latency_ms": 1, + "output_tokens_per_second": 10_000, + "intelligence_index": 50, + } + }, + ) + + assert all(math.isfinite(row["score"]) for row in decision.get("candidate_scores", [])) + + def test_falls_back_to_the_rules_decision_when_a_tier_has_no_usable_candidate(self): + empty_tiers = { + tier: {"primary": "", "fallback": []} for tier in DEFAULT_ROUTING_CONFIG["tiers"] + } + decision = route( + "hello", + None, + 128, + { + "config": {**DEFAULT_ROUTING_CONFIG, "tiers": empty_tiers}, + "model_pricing": PORTFOLIO_PRICING, + }, + ) + + assert decision["method"] == "rules" + assert decision["model"] == "" + + def test_lets_a_host_capability_snapshot_override_the_built_in_catalog(self): + decision = _portfolio( + "Use the lookup_order tool for order B-42.", + 256, + has_tools=True, + requires_tools=True, + model_capabilities={ + "anthropic/claude-sonnet-5": { + "context_window": 1_000_000, + "max_output_tokens": 128_000, + "supports_tools": False, + "supports_vision": True, + } + }, + ) + + assert "anthropic/claude-sonnet-5" not in decision["candidates"] + + +# ─── selector.test.ts ─── + +SELECTOR_TIER_CONFIGS = { + tier: {"primary": "moonshot/kimi-k2.5", "fallback": []} + for tier in ("SIMPLE", "MEDIUM", "COMPLEX", "REASONING") +} +SELECTOR_PRICING = { + "moonshot/kimi-k2.5": _price(0.5, 2.4), + "anthropic/claude-opus-4.7": _price(5, 25), + "anthropic/claude-opus-4.8": _price(5, 25), +} + + +def _supports_tool_calling(model: str) -> bool: + return model not in ("minimax/minimax-m2.5", "nvidia/gpt-oss-120b") + + +class TestSelector: + def test_select_model_uses_opus_4_7_as_the_savings_baseline(self): + decision = select_model( + "SIMPLE", + 0.95, + "rules", + "test", + SELECTOR_TIER_CONFIGS, + SELECTOR_PRICING, + 1000, + 1000, + ) + + assert decision["baseline_cost"] > 0 + assert decision["savings"] > 0 + + def test_calculate_model_cost_uses_opus_4_7_as_the_baseline(self): + costs = calculate_model_cost("moonshot/kimi-k2.5", SELECTOR_PRICING, 1000, 1000) + + assert costs["baseline_cost"] > 0 + assert costs["savings"] > 0 + + def test_filter_by_tool_calling_removes_models_without_tool_support(self): + models = ["moonshot/kimi-k2.5", "minimax/minimax-m2.5", "deepseek/deepseek-chat"] + + assert filter_by_tool_calling(models, True, _supports_tool_calling) == [ + "moonshot/kimi-k2.5", + "deepseek/deepseek-chat", + ] + + def test_filter_by_tool_calling_keeps_every_model_when_the_request_has_no_tools(self): + models = ["moonshot/kimi-k2.5", "minimax/minimax-m2.5", "nvidia/gpt-oss-120b"] + + assert filter_by_tool_calling(models, False, _supports_tool_calling) == models + + def test_filter_by_tool_calling_never_returns_an_empty_chain(self): + unsupported = ["minimax/minimax-m2.5", "nvidia/gpt-oss-120b"] + + assert filter_by_tool_calling(unsupported, True, _supports_tool_calling) == unsupported + + def test_filter_by_exclude_list(self): + chain = ["moonshot/kimi-k2.5", "deepseek/deepseek-chat", "anthropic/claude-sonnet-4.6"] + + assert filter_by_exclude_list(chain, {"deepseek/deepseek-chat"}) == [ + "moonshot/kimi-k2.5", + "anthropic/claude-sonnet-4.6", + ] + assert filter_by_exclude_list(chain, set(chain)) == chain + assert filter_by_exclude_list(chain, set()) == chain + + def test_filter_candidates_by_capacity(self): + capabilities = { + "small": {"context_window": 8_000, "max_output": 2_000}, + "large": {"context_window": 128_000, "max_output": 32_000}, + } + + assert filter_candidates_by_capacity( + ["small", "large"], 10_000, 4_000, capabilities.get + ) == ["large"] + assert filter_candidates_by_capacity(["small"], 100_000, 40_000, capabilities.get) == [] + + +# ─── strategy.test.ts ─── + + +class TestRulesStrategy: + def test_returns_tier_configs_in_the_decision(self): + decision = RulesStrategy().route("hello", None, 100, BASE_OPTIONS) + + assert decision["tier_configs"] is not None + for tier in ("SIMPLE", "MEDIUM", "COMPLEX", "REASONING"): + assert tier in decision["tier_configs"] + + def test_returns_profile_in_the_decision(self): + decision = RulesStrategy().route("hello", None, 100, BASE_OPTIONS) + + assert decision["profile"] in ("auto", "eco", "premium", "agentic") + + def test_honors_the_protocol_structured_output_requirement(self): + decision = RulesStrategy().route( + "hello", None, 100, {**BASE_OPTIONS, "requires_structured_output": True} + ) + + assert decision["tier"] == "MEDIUM" + assert "structured output" in decision["reasoning"] + + def test_sets_eco_profile_when_routing_profile_is_eco(self): + decision = RulesStrategy().route( + "hello", None, 100, {**BASE_OPTIONS, "routing_profile": "eco"} + ) + + assert decision["profile"] == "eco" + assert decision["tier_configs"] == DEFAULT_ROUTING_CONFIG["eco_tiers"] + + def test_sets_premium_profile_when_routing_profile_is_premium(self): + decision = RulesStrategy().route( + "hello", None, 100, {**BASE_OPTIONS, "routing_profile": "premium"} + ) + + assert decision["profile"] == "premium" + assert decision["tier_configs"] == DEFAULT_ROUTING_CONFIG["premium_tiers"] + + def test_eco_tiers_none_falls_back_to_regular_tiers_without_dropping_into_auto(self): + decision = RulesStrategy().route( + "hello", + None, + 100, + { + **BASE_OPTIONS, + "config": {**DEFAULT_ROUTING_CONFIG, "eco_tiers": None}, + "routing_profile": "eco", + "has_tools": True, + "now": datetime(2025, 1, 1, tzinfo=timezone.utc), + }, + ) + + assert decision["profile"] == "eco" + assert decision["tier_configs"] == DEFAULT_ROUTING_CONFIG["tiers"] + + def test_premium_tiers_none_falls_back_to_regular_tiers(self): + decision = RulesStrategy().route( + "hello", + None, + 100, + { + **BASE_OPTIONS, + "config": {**DEFAULT_ROUTING_CONFIG, "premium_tiers": None}, + "routing_profile": "premium", + "has_tools": True, + "now": datetime(2025, 1, 1, tzinfo=timezone.utc), + }, + ) + + assert decision["profile"] == "premium" + assert decision["tier_configs"] == DEFAULT_ROUTING_CONFIG["tiers"] + + def test_sets_agentic_profile_when_tools_are_present(self): + decision = RulesStrategy().route("hello", None, 100, {**BASE_OPTIONS, "has_tools": True}) + + assert decision["profile"] == "agentic" + assert decision["tier_configs"] == DEFAULT_ROUTING_CONFIG["agentic_tiers"] + + def test_sets_auto_profile_for_default_requests(self): + decision = RulesStrategy().route( + "what is the capital of France", + None, + 100, + {**BASE_OPTIONS, "now": datetime(2025, 1, 1, tzinfo=timezone.utc)}, + ) + + assert decision["profile"] == "auto" + assert decision["tier_configs"] == DEFAULT_ROUTING_CONFIG["tiers"] + + def test_agentic_mode_false_disables_agentic_tiers_even_with_tools(self): + config = { + **DEFAULT_ROUTING_CONFIG, + "overrides": {**DEFAULT_ROUTING_CONFIG["overrides"], "agentic_mode": False}, + } + decision = RulesStrategy().route( + "hello", + None, + 100, + { + **BASE_OPTIONS, + "config": config, + "has_tools": True, + "now": datetime(2025, 1, 1, tzinfo=timezone.utc), + }, + ) + + assert decision["profile"] == "auto" + assert decision["tier_configs"] == DEFAULT_ROUTING_CONFIG["tiers"] + + def test_agentic_mode_true_forces_agentic_tiers_even_without_tools(self): + config = { + **DEFAULT_ROUTING_CONFIG, + "overrides": {**DEFAULT_ROUTING_CONFIG["overrides"], "agentic_mode": True}, + } + decision = RulesStrategy().route( + "hello", + None, + 100, + { + **BASE_OPTIONS, + "config": config, + "has_tools": False, + "now": datetime(2025, 1, 1, tzinfo=timezone.utc), + }, + ) + + assert decision["profile"] == "agentic" + assert decision["tier_configs"] == DEFAULT_ROUTING_CONFIG["agentic_tiers"] + + +class TestStrategyRegistry: + def test_retrieves_the_default_rules_strategy(self): + strategy = get_strategy("rules") + + assert isinstance(strategy, RulesStrategy) + assert strategy.name == "rules" + + def test_raises_for_an_unknown_strategy(self): + with pytest.raises(ValueError, match="Unknown routing strategy: nonexistent"): + get_strategy("nonexistent") + + def test_registers_and_retrieves_a_custom_strategy(self): + class CustomStrategy: + name = "custom-test" + + def route(self, prompt, system_prompt, max_output_tokens, options): + return { + "model": "test/model", + "tier": "SIMPLE", + "confidence": 1, + "method": "rules", + "reasoning": "custom strategy", + "cost_estimate": 0, + "baseline_cost": 0, + "savings": 0, + "tier_configs": options["config"]["tiers"], + "profile": "auto", + } + + register_strategy(CustomStrategy()) + retrieved = get_strategy("custom-test") + + assert retrieved.name == "custom-test" + decision = retrieved.route("test", None, 100, BASE_OPTIONS) + assert decision["model"] == "test/model" + assert decision["reasoning"] == "custom strategy" + + +class TestPortfolioDefault: + def test_route_uses_the_v3_portfolio_while_retaining_rule_tiers(self): + simple = route("hello", None, 100, BASE_OPTIONS) + + assert simple["tier"] == "SIMPLE" + assert simple["method"] == "portfolio" + assert simple["model"] + assert simple["candidates"][0] == simple["model"] + assert simple["router_version"] == "v3-portfolio" + + reasoning = route( + "prove the theorem step by step using mathematical induction", None, 4096, BASE_OPTIONS + ) + + assert reasoning["tier"] == "REASONING" + assert reasoning["method"] == "portfolio" + assert simple["tier_configs"] is not None + assert simple["profile"] is not None + assert reasoning["tier_configs"] is not None + assert reasoning["profile"] is not None + + def test_supports_a_config_only_rollback_to_the_v2_rules_strategy(self): + decision = route( + "hello", + None, + 100, + {**BASE_OPTIONS, "config": {**DEFAULT_ROUTING_CONFIG, "strategy": "rules"}}, + ) + + assert decision["method"] == "rules" + + def test_recognizes_multiple_choice_reasoning(self): + decision = route( + "Which statement is correct?\n\nA. First\nB. Second\nC. Third\nD. Fourth\n\n" + "Return the final answer choice.", + None, + 512, + BASE_OPTIONS, + ) + + assert decision["task_type"] == "reasoning_mcq" + assert decision["tier"] == "REASONING" + assert decision["model"] == "google/gemini-3-flash-preview" + assert "xai/grok-4.5" in decision["candidates"] + assert decision["tier_configs"]["REASONING"]["primary"] == decision["model"] + + def test_recognizes_compact_multilingual_arithmetic(self): + decision = route( + "Una caja tiene 12 libros. Hay 4 cajas. ¿Cuántos libros hay en total?", + None, + 512, + BASE_OPTIONS, + ) + + assert decision["task_type"] == "reasoning_math" + assert decision["tier"] == "REASONING" + assert decision["model"] == "google/gemini-3.5-flash" + + def test_recognizes_math_word_problems_without_question_marks(self): + decision = route( + "เรือแล่นได้เร็ว 10 ไมล์ต่อชั่วโมง ตั้งแต่ 13.00 น. ถึง 16.00 น. " "และกลับด้วยความเร็ว 6 ไมล์ต่อชั่วโมง", + None, + 512, + BASE_OPTIONS, + ) + + assert decision["task_type"] == "reasoning_math" + + +# ─── tool-intent.test.ts ─── + + +class TestInferToolRequirement: + def test_does_not_confuse_available_tools_with_a_tool_requirement(self): + assert not infer_tool_requirement( + "Which option best explains the observation?\nA. One\nB. Two\nC. Three\nD. Four" + ) + assert not infer_tool_requirement("What is 17 times 9?") + + def test_recognizes_explicit_tool_repository_web_and_stateful_actions(self): + assert infer_tool_requirement("Use the lookup_order tool for order B-42.") + assert infer_tool_requirement("Patch the repository and run the tests.") + assert infer_tool_requirement( + "Calculate the average and save it in a file called result.txt." + ) + assert infer_tool_requirement("Search the web for today's weather in Shanghai.") + assert infer_tool_requirement("Cancel my flight booking and refund the ticket.") + assert infer_tool_requirement("修改仓库里的文件,然后运行测试。") + + def test_honors_the_openai_tool_choice_contract(self): + assert infer_tool_requirement("Retrieve the account details.", None, "required") + assert infer_tool_requirement( + "Retrieve the account details.", + None, + {"type": "function", "function": {"name": "get_account"}}, + ) + assert not infer_tool_requirement("What is 17 times 9?", None, "auto") + assert not infer_tool_requirement( + "Cancel my flight booking and refund the ticket.", None, "none" + ) + + def test_does_not_treat_host_tool_descriptions_as_a_per_turn_requirement(self): + system_prompt = ( + "You can use web_search to look up documentation, run tests, " + "and update account records." + ) + + assert not infer_tool_requirement("What is 17 times 9?", system_prompt) + + +class TestDimensionWeightKeys: + """Every scored dimension must find its weight. + + The config transpile that produced ``config.py`` snake_cased key names, and + ``imperativeVerbs`` is both a keyword-list field *and* a dimension name — so + the weight landed under ``imperative_verbs`` while the classifier emitted + ``imperativeVerbs``. ``weights.get(name, 0)`` then silently scored that + dimension at zero, diverging from the TypeScript SDK on any prompt whose + imperative verbs would have crossed a tier boundary. + """ + + def test_every_emitted_dimension_has_a_weight(self): + from blockrun_llm.router_core.rules import classify_by_rules + + scoring = DEFAULT_ROUTING_CONFIG["scoring"] + result = classify_by_rules("Build and deploy the service", None, 10, scoring) + emitted = {dimension["name"] for dimension in result["dimensions"]} + weighted = set(scoring["dimension_weights"]) + + assert emitted - weighted == set(), "scored dimensions with no weight" + assert weighted - emitted == set(), "weights that match no scored dimension" + + def test_the_weights_match_the_upstream_values(self): + # Ported verbatim from router-core config.ts at 5ee7c23. + assert DEFAULT_ROUTING_CONFIG["scoring"]["dimension_weights"] == { + "tokenCount": 0.08, + "codePresence": 0.15, + "reasoningMarkers": 0.18, + "technicalTerms": 0.1, + "creativeMarkers": 0.05, + "simpleIndicators": 0.02, + "multiStepPatterns": 0.12, + "questionComplexity": 0.05, + "imperativeVerbs": 0.03, + "constraintCount": 0.04, + "outputFormat": 0.03, + "referenceComplexity": 0.02, + "negationComplexity": 0.01, + "domainSpecificity": 0.02, + "agenticTask": 0.04, + } + + +class TestUnavailableModels: + """1:1 port of ``unavailable-models.test.ts`` (5ee7c23).""" + + TIERS = { + "SIMPLE": {"primary": "a/one", "fallback": ["a/two", "a/three"]}, + "MEDIUM": {"primary": "b/one", "fallback": ["b/two"]}, + "COMPLEX": {"primary": "c/one", "fallback": []}, + "REASONING": {"primary": "d/one", "fallback": ["d/two"]}, + } + + @staticmethod + def _options(**overrides): + from blockrun_llm.router_core import DEFAULT_MODEL_CAPABILITIES + + pricing = { + model: {"input_price": 1.0, "output_price": 3.0} for model in DEFAULT_MODEL_CAPABILITIES + } + base = { + "config": DEFAULT_ROUTING_CONFIG, + "model_pricing": pricing, + "now": datetime(2026, 8, 20, tzinfo=timezone.utc), + } + base.update(overrides) + return base + + def test_is_the_identity_for_an_absent_or_empty_list(self): + assert apply_unavailable_models(self.TIERS, None) is self.TIERS + assert apply_unavailable_models(self.TIERS, []) is self.TIERS + + def test_promotes_the_first_surviving_fallback_when_the_primary_is_dead(self): + result = apply_unavailable_models(self.TIERS, ["a/one"]) + assert result["SIMPLE"] == {"primary": "a/two", "fallback": ["a/three"]} + # Untouched tiers keep their original config objects. + assert result["MEDIUM"] is self.TIERS["MEDIUM"] + + def test_removes_dead_rungs_from_the_middle_of_a_chain(self): + result = apply_unavailable_models(self.TIERS, ["a/two"]) + assert result["SIMPLE"] == {"primary": "a/one", "fallback": ["a/three"]} + + def test_keeps_the_original_config_when_a_tiers_whole_chain_is_dead(self): + result = apply_unavailable_models(self.TIERS, ["c/one"]) + assert result["COMPLEX"] is self.TIERS["COMPLEX"] + + def test_does_not_mutate_its_input(self): + apply_unavailable_models(self.TIERS, ["a/one", "b/one"]) + assert self.TIERS["SIMPLE"]["primary"] == "a/one" + assert self.TIERS["MEDIUM"]["primary"] == "b/one" + + def test_never_selects_or_lists_a_model_the_host_declared_dead(self): + baseline = route("What is the capital of France?", None, 256, self._options()) + dead = baseline["model"] + decision = route( + "What is the capital of France?", + None, + 256, + self._options(unavailable_models=[dead]), + ) + assert decision["model"] != dead + assert dead not in (decision.get("candidates") or []) + + def test_keeps_dead_evidence_candidates_out_of_the_portfolio_chain(self): + math_prompt = "Solve for x: 3x^2 - 12x + 9 = 0. Show your work." + baseline = route(math_prompt, None, 1024, self._options()) + evidence = baseline.get("candidates") or [] + assert len(evidence) > 1 + dead = evidence[0] + decision = route(math_prompt, None, 1024, self._options(unavailable_models=[dead])) + assert decision["model"] != dead + assert dead not in (decision.get("candidates") or []) + + def test_applies_to_the_rules_strategy_as_well(self): + config = {**DEFAULT_ROUTING_CONFIG, "strategy": "rules"} + baseline = route("What is the capital of France?", None, 256, self._options(config=config)) + dead = baseline["model"] + decision = route( + "What is the capital of France?", + None, + 256, + self._options(config=config, unavailable_models=[dead]), + ) + assert decision["model"] != dead + + def test_survives_killing_an_entire_tier_chain(self): + chain = [ + DEFAULT_ROUTING_CONFIG["tiers"]["SIMPLE"]["primary"], + *DEFAULT_ROUTING_CONFIG["tiers"]["SIMPLE"]["fallback"], + ] + decision = route( + "What is the capital of France?", + None, + 256, + self._options(unavailable_models=chain), + ) + assert len(decision["model"]) > 0 diff --git a/tests/unit/test_router_core_snapshot.py b/tests/unit/test_router_core_snapshot.py new file mode 100644 index 0000000..922e293 --- /dev/null +++ b/tests/unit/test_router_core_snapshot.py @@ -0,0 +1,164 @@ +""" +Cross-language decision-snapshot parity. + +``router_core_decisions.snapshot.json`` is a verbatim copy of upstream +``decisions.snapshot.json`` at commit ``5ee7c23`` — 88 complete decisions the +TypeScript engine produced for a frozen corpus (22 prompts x 4 profiles with +rotating tool/vision/structured-output shapes, frozen pricing, frozen clock). +This test recomputes every decision with the Python port and compares field +by field, floats included: same IEEE-754 operations must yield the same +doubles, and reasoning strings must match to the character because hosts +assert on their wording. + +When upstream re-syncs, copy the regenerated fixture over and re-run — a +mismatch means the port has drifted, not that the fixture is stale. +""" + +from __future__ import annotations + +import json +from datetime import datetime, timezone +from pathlib import Path + +from blockrun_llm.router_core import ( + DEFAULT_MODEL_CAPABILITIES, + DEFAULT_ROUTING_CONFIG, + route, +) + +FIXTURE = Path(__file__).with_name("router_core_decisions.snapshot.json") + +# Upstream decisions.snapshot.test.ts, transliterated. The pricing hash is the +# JS one (charCodeAt * 31, unsigned 32-bit) so both engines price identically. + + +def _name_hash(model: str) -> int: + value = 0 + for char in model: + value = (value * 31 + ord(char)) & 0xFFFFFFFF + return value + + +PRICING = { + model: { + "input_price": 0.1 + (_name_hash(model) % 7) * 0.7, + "output_price": 0.4 + (_name_hash(model) % 5) * 2.1, + } + for model in DEFAULT_MODEL_CAPABILITIES +} +PRICING["anthropic/claude-opus-4.7"] = {"input_price": 5.0, "output_price": 25.0} + +NOW = datetime(2026, 8, 20, tzinfo=timezone.utc) + +PROMPTS = [ + "What is the capital of France?", + "hi", + "Explain the difference between TCP and UDP in one paragraph.", + "Write a Python function that checks if a string is a valid IPv4 address. Include edge cases.", + "Prove that the sum of two odd integers is even, step by step.", + "Refactor this React component to use hooks:\n```jsx\nclass Foo extends React.Component { render() { return
} }\n```", + "Cancel order B-42 and book the 9am flight to SFO.", + "What's the weather in Tokyo, Paris, and New York?", + "帮我总结这篇文章的要点,不超过三句话。", + "设计一个分布式限流器,要求支持滑动窗口和多机房容灾,并给出伪代码。", + "debug: TypeError: Cannot read properties of undefined (reading 'map') at UserList.render", + "Summarize the following contract clause and list any obligations: " + "lorem ipsum " * 400, + "Which of the following is NOT a prime? (a) 17 (b) 21 (c) 23 (d) 29. Answer with the letter only.", + "Solve for x: 3x^2 - 12x + 9 = 0. Show your work.", + "Extract all email addresses and phone numbers from this text as JSON: contact bob@x.com or 555-1234", + "rm -rf the old build directory, then rerun the release pipeline and paste the log tail", + "Investigate why the checkout page p95 regressed after Tuesday's deploy. Check the CDN config, the API gateway logs, and the database slow query log.", + "Write a haiku about autumn rain.", + "Translate 'the quick brown fox jumps over the lazy dog' into German, French, and Japanese.", + "Plan a 7-day itinerary for Kyoto in November with a daily budget of $150, must include one onsen day and avoid Mondays for museums.", + "Design the architecture for a multi-tenant SaaS billing system: requirements, data model, service boundaries, failure modes, migration plan from the legacy monolith, and a rollout strategy with feature flags.", + "Here is our full incident log, produce a postmortem timeline: " + + "07:14 api-gw 502 spike; 07:16 pod restart loop; " * 1200, +] + +SHAPES = [ + {}, + { + "has_tools": True, + "requires_tools": True, + "tool_count": 4, + "tool_names": ["cancel_order", "book_flight", "search_flights", "get_user"], + }, + {"has_vision": True}, + {"requires_structured_output": True}, + { + "has_tools": True, + "requires_tools": False, + "tool_count": 12, + "tool_names": ["read_file", "write_file", "run_shell", "search_code"], + }, +] + +PROFILES = [None, "eco", "auto", "premium"] + +#: snake_case decision key -> camelCase fixture key, for every pinned field. +KEY_MAP = { + "model": "model", + "tier": "tier", + "confidence": "confidence", + "method": "method", + "reasoning": "reasoning", + "cost_estimate": "costEstimate", + "baseline_cost": "baselineCost", + "savings": "savings", + "agentic_score": "agenticScore", + "profile": "profile", + "candidates": "candidates", + "task_type": "taskType", + "router_version": "routerVersion", +} + + +def _candidate_scores(entries): + return [ + { + "model": entry["model"], + "score": entry["score"], + "quality": entry["quality"], + "cost": entry["cost"], + "speed": entry["speed"], + "reliability": entry["reliability"], + } + for entry in entries + ] + + +def test_the_python_port_reproduces_every_upstream_decision(): + expected_rows = json.loads(FIXTURE.read_text()) + assert len(expected_rows) == len(PROFILES) * len(PROMPTS) + + mismatches: list[str] = [] + index = 0 + for profile in PROFILES: + for i, prompt in enumerate(PROMPTS): + expected = expected_rows[index] + index += 1 + options = { + "config": DEFAULT_ROUTING_CONFIG, + "model_pricing": PRICING, + "routing_profile": profile, + "now": NOW, + **SHAPES[i % len(SHAPES)], + } + decision = route( + prompt, + "You are a helpful assistant." if i % 3 == 0 else None, + 256 + (i % 4) * 1024, + options, + ) + row = f"prompt={i} profile={profile or 'default'}" + for snake, camel in KEY_MAP.items(): + if decision.get(snake) != expected.get(camel): + mismatches.append( + f"{row} {camel}: py={decision.get(snake)!r} ts={expected.get(camel)!r}" + ) + got_scores = _candidate_scores(decision.get("candidate_scores") or []) + if got_scores != (expected.get("candidateScores") or []): + mismatches.append(f"{row} candidateScores differ") + + assert not mismatches, "\n".join(mismatches[:20]) + f"\n({len(mismatches)} total)" diff --git a/tests/unit/test_routing_parity.py b/tests/unit/test_routing_parity.py new file mode 100644 index 0000000..9fbcdf5 --- /dev/null +++ b/tests/unit/test_routing_parity.py @@ -0,0 +1,199 @@ +""" +Routing surface parity across the four clients. + +Base and Solana, sync and async, must expose the same routing: route(), +smart_chat(), smart_chat_completion(), the blockrun/* virtual model ids, and a +ranked fallback chain on the ordinary chat paths. Before 1.12.0 the Solana +clients had none of it and the Base clients had no message-list routing, so a +Solana user got no routing at all and an agent transcript could not be routed. +""" + +from __future__ import annotations + +import inspect +from unittest.mock import patch + +import pytest + +from blockrun_llm import AsyncLLMClient, AsyncSolanaLLMClient, LLMClient, SolanaLLMClient +from blockrun_llm.router_adapter import build_model_pricing, routing_profile_for_model + +CLIENTS = [LLMClient, AsyncLLMClient, SolanaLLMClient, AsyncSolanaLLMClient] + +CATALOG = [ + {"id": "google/gemini-2.5-flash", "pricing": {"input": 0.15, "output": 0.6}}, + {"id": "google/gemini-3.5-flash", "pricing": {"input": 0.5, "output": 3}}, + {"id": "google/gemini-3-flash-preview", "pricing": {"input": 0.5, "output": 3}}, + {"id": "google/gemini-3.1-flash-lite", "pricing": {"input": 0.25, "output": 1.5}}, + {"id": "google/gemini-3.1-pro", "pricing": {"input": 1.25, "output": 10}}, + {"id": "anthropic/claude-opus-4.7", "pricing": {"input": 5, "output": 25}}, + {"id": "anthropic/claude-sonnet-5", "pricing": {"input": 3, "output": 15}}, + {"id": "openai/gpt-5-mini", "pricing": {"input": 0.25, "output": 2}}, + {"id": "openai/gpt-5.3-codex", "pricing": {"input": 1.75, "output": 14}}, + {"id": "deepseek/deepseek-v4-pro", "pricing": {"input": 0.435, "output": 0.87}}, + {"id": "moonshot/kimi-k2.7", "pricing": {"input": 0.95, "output": 4}}, + {"id": "nvidia/nemotron-3.5-lightning", "pricing": {"input": 0, "output": 0}}, + {"id": "nvidia/nemotron-3-nano-30b", "pricing": {"input": 0, "output": 0}}, + {"id": "nvidia/llama-3.2-11b-vision", "pricing": {"input": 0, "output": 0}}, + {"id": "cohere/north-mini-code", "pricing": {"input": 0, "output": 0}}, + {"id": "poolside/laguna-xs-2.1", "pricing": {"input": 0, "output": 0}}, + # Unavailable rows must never win routing. + {"id": "dead/model", "pricing": {"input": 0.01, "output": 0.01}, "available": False}, +] + + +class TestSurfaceParity: + @pytest.mark.parametrize("client", CLIENTS, ids=lambda c: c.__name__) + @pytest.mark.parametrize("method", ["route", "smart_chat", "smart_chat_completion"]) + def test_every_client_exposes_the_routing_surface(self, client, method): + assert hasattr(client, method), f"{client.__name__} is missing {method}()" + + @pytest.mark.parametrize("client", CLIENTS, ids=lambda c: c.__name__) + def test_the_ordinary_chat_paths_accept_a_fallback_chain(self, client): + # Routing hands back a ranked chain; it is useless if chat() cannot walk it. + for method in ("chat", "chat_completion"): + params = inspect.signature(getattr(client, method)).parameters + assert "fallback_models" in params, f"{client.__name__}.{method}" + + @pytest.mark.parametrize("client", CLIENTS, ids=lambda c: c.__name__) + def test_routing_profile_is_selectable_everywhere(self, client): + for method in ("route", "smart_chat", "smart_chat_completion"): + params = inspect.signature(getattr(client, method)).parameters + assert "routing_profile" in params, f"{client.__name__}.{method}" + + +class TestVirtualModelIds: + @pytest.mark.parametrize( + ("model", "expected"), + [ + ("blockrun/auto", "auto"), + ("blockrun/eco", "eco"), + ("blockrun/premium", "premium"), + ("BLOCKRUN/AUTO", "auto"), + ("google/gemini-2.5-flash", None), + ("blockrun/nonsense", None), + ], + ) + def test_only_the_three_profiles_are_virtual(self, model, expected): + assert routing_profile_for_model(model) == expected + + def test_chat_completion_routes_a_virtual_id_instead_of_calling_it(self): + client = LLMClient(private_key="0x" + "11" * 32) + with ( + patch.object(LLMClient, "list_models", return_value=CATALOG), + patch.object(LLMClient, "chat_completion", wraps=client.chat_completion) as spy, + patch.object(LLMClient, "_request_with_payment") as request, + ): + request.return_value = None + try: + client.chat_completion("blockrun/auto", [{"role": "user", "content": "hi"}]) + except Exception: # the transport is stubbed; routing is the subject + pass + # Re-entered through the routed path with a concrete model. + routed = [call.args[0] for call in spy.call_args_list if call.args] + assert "blockrun/auto" in routed + assert any(m != "blockrun/auto" for m in routed), "never resolved to a real model" + + +class TestPricingMap: + def test_skips_rows_the_catalog_marks_unavailable(self): + pricing = build_model_pricing(CATALOG) + + assert "dead/model" not in pricing + assert pricing["google/gemini-2.5-flash"] == { + "input_price": 0.15, + "output_price": 0.6, + "flat_price": 0.0, + } + + def test_accepts_the_legacy_top_level_price_shape(self): + pricing = build_model_pricing([{"id": "a/b", "inputPrice": 1, "outputPrice": 2}]) + + assert pricing["a/b"]["input_price"] == 1 + assert pricing["a/b"]["output_price"] == 2 + + +class TestDecisionsMatchAcrossChains: + """Base and Solana share one engine: same catalog in, same model out. + + Only the cost floor differs — Base signs a $0.002 minimum, Solana $0.001. + """ + + @pytest.mark.parametrize( + "prompt", + [ + "Summarize this changelog entry in one line", + "Prove that the square root of 2 is irrational, step by step", + "Refactor this TypeScript function to use async/await", + ], + ) + def test_same_model_on_both_chains(self, prompt): + base = LLMClient(private_key="0x" + "11" * 32) + solana = SolanaLLMClient.__new__(SolanaLLMClient) + solana._model_pricing_cache = build_model_pricing(CATALOG) + + with patch.object(LLMClient, "list_models", return_value=CATALOG): + base_decision = base.route(prompt) + solana_decision = SolanaLLMClient.route(solana, prompt) + + assert base_decision.model == solana_decision.model + assert base_decision.tier == solana_decision.tier + assert base_decision.task_type == solana_decision.task_type + assert base_decision.candidates == solana_decision.candidates + # Chain-specific payment floor, same routing. + assert base_decision.cost_estimate >= solana_decision.cost_estimate + + def test_free_profile_is_free_on_solana_too(self): + solana = SolanaLLMClient.__new__(SolanaLLMClient) + solana._model_pricing_cache = build_model_pricing(CATALOG) + + decision = SolanaLLMClient.route(solana, "What is 2+2?", routing_profile="free") + + assert decision.cost_estimate == 0 + for model in [decision.model, *decision.fallbacks]: + assert solana._model_pricing_cache[model]["input_price"] == 0 + assert solana._model_pricing_cache[model]["output_price"] == 0 + + +class TestRetriableStatuses: + """A saturated upstream must hand the turn to the next ranked model. + + Observed live: a rate-limited free model answered 429 and the three + remaining free models in the chain were never tried, because 429 was not in + the retriable set. The TypeScript adapter has always treated it as + transient — same upstream saturated, next model is a different upstream. + """ + + @pytest.mark.parametrize("status", [429, 502, 503, 504, 522, 524]) + def test_saturation_and_availability_errors_walk_the_chain(self, status): + from blockrun_llm.client import _should_fallback + from blockrun_llm.solana_client import _should_fallback_solana + from blockrun_llm.types import APIError + + exc = APIError(f"API error: {status}", status_code=status) + + assert _should_fallback(exc), f"Base refuses to fall back on {status}" + assert _should_fallback_solana(exc), f"Solana refuses to fall back on {status}" + + @pytest.mark.parametrize("status", [400, 401, 403, 404, 422]) + def test_client_errors_do_not_walk_the_chain(self, status): + from blockrun_llm.client import _should_fallback + from blockrun_llm.solana_client import _should_fallback_solana + from blockrun_llm.types import APIError + + exc = APIError(f"API error: {status}", status_code=status) + + assert not _should_fallback(exc) + assert not _should_fallback_solana(exc) + + def test_a_settled_payment_is_never_retried(self): + # The next model would sign a second transfer for one call. + from blockrun_llm.client import _mark_settled, _should_fallback + from blockrun_llm.solana_client import _should_fallback_solana + from blockrun_llm.types import APIError + + exc = APIError("API error: 503", status_code=503) + _mark_settled(exc) + + assert not _should_fallback(exc) + assert not _should_fallback_solana(exc) diff --git a/tests/unit/test_rpc.py b/tests/unit/test_rpc.py new file mode 100644 index 0000000..ba56729 --- /dev/null +++ b/tests/unit/test_rpc.py @@ -0,0 +1,142 @@ +"""Unit tests for RpcClient request construction and response parsing.""" + +import os + +import httpx +import pytest + +from blockrun_llm import NETWORK_ALIASES, SUPPORTED_NETWORKS, RpcClient, RpcResponse + + +@pytest.fixture +def client(): + # Deterministic dummy key — never actually signs against a live endpoint + # in unit tests; we only exercise local request/response paths. + os.environ.setdefault("BLOCKRUN_WALLET_KEY", "0x" + "11" * 32) + return RpcClient() + + +def _headers(**extra): + base = {"x-network": "ethereum", "x-cache": "MISS"} + base.update(extra) + return httpx.Headers(base) + + +def test_call_builds_jsonrpc_body(client, monkeypatch): + captured = {} + + def fake_request(network, body): + captured["network"] = network + captured["body"] = body + return {"jsonrpc": "2.0", "id": 1, "result": "0x10"}, _headers() + + monkeypatch.setattr(client, "_request_with_payment", fake_request) + + resp = client.call("ethereum", "eth_blockNumber") + + assert captured["network"] == "ethereum" + assert captured["body"] == {"jsonrpc": "2.0", "id": 1, "method": "eth_blockNumber"} + assert resp.result == "0x10" + assert resp.network == "ethereum" + assert resp.cache_hit is False + + +def test_call_includes_params_and_custom_id(client, monkeypatch): + captured = {} + + def fake_request(network, body): + captured["body"] = body + return {"jsonrpc": "2.0", "id": body["id"], "result": "0x0"}, _headers() + + monkeypatch.setattr(client, "_request_with_payment", fake_request) + + client.call("base", "eth_getBalance", ["0xabc", "latest"], id="bal-1") + + assert captured["body"] == { + "jsonrpc": "2.0", + "id": "bal-1", + "method": "eth_getBalance", + "params": ["0xabc", "latest"], + } + + +def test_call_surfaces_cache_hit_and_tx_hash(client, monkeypatch): + def fake_request(network, body): + return ( + {"jsonrpc": "2.0", "id": 1, "result": "0x1"}, + _headers(**{"x-cache": "HIT", "x-payment-receipt": "0xdeadbeef"}), + ) + + monkeypatch.setattr(client, "_request_with_payment", fake_request) + + resp = client.call("ethereum", "eth_chainId") + assert resp.cache_hit is True + assert resp.tx_hash == "0xdeadbeef" + + +def test_call_parses_jsonrpc_error(client, monkeypatch): + def fake_request(network, body): + return ( + {"jsonrpc": "2.0", "id": 1, "error": {"code": -32601, "message": "no method"}}, + _headers(), + ) + + monkeypatch.setattr(client, "_request_with_payment", fake_request) + + resp = client.call("ethereum", "eth_bogus") + assert resp.result is None + assert resp.error is not None + assert resp.error.code == -32601 + assert resp.error.message == "no method" + + +def test_batch_fills_jsonrpc_and_ids(client, monkeypatch): + captured = {} + + def fake_request(network, body): + captured["body"] = body + return [ + {"jsonrpc": "2.0", "id": 1, "result": "0x10"}, + {"jsonrpc": "2.0", "id": 7, "result": "0x3b9aca00"}, + ], _headers() + + monkeypatch.setattr(client, "_request_with_payment", fake_request) + + out = client.batch( + "polygon", + [{"method": "eth_blockNumber"}, {"method": "eth_gasPrice", "id": 7}], + ) + + assert captured["body"] == [ + {"jsonrpc": "2.0", "id": 1, "method": "eth_blockNumber"}, + {"jsonrpc": "2.0", "id": 7, "method": "eth_gasPrice"}, + ] + assert len(out) == 2 + assert all(isinstance(r, RpcResponse) for r in out) + assert out[1].id == 7 + + +def test_batch_rejects_empty_and_missing_method(client): + with pytest.raises(ValueError, match="at least one"): + client.batch("ethereum", []) + with pytest.raises(ValueError, match="missing 'method'"): + client.batch("ethereum", [{"params": []}]) + + +def test_network_registry_mirrors_backend(): + # 40 curated chains, 29 EVM + 11 non-EVM (backend src/lib/tatum.ts) + assert len(SUPPORTED_NETWORKS) == 40 + assert len(SUPPORTED_NETWORKS) == len(set(SUPPORTED_NETWORKS)) + for must in ("ethereum", "base", "solana", "bitcoin", "ripple", "sui"): + assert must in SUPPORTED_NETWORKS + # Aliases resolve to curated keys + for alias, canonical in NETWORK_ALIASES.items(): + assert canonical in SUPPORTED_NETWORKS, f"{alias} -> {canonical} not curated" + assert NETWORK_ALIASES["xrpl"] == "ripple" + assert NETWORK_ALIASES["sol"] == "solana" + + +def test_get_wallet_address(client): + addr = client.get_wallet_address() + assert addr.startswith("0x") + assert len(addr) == 42 diff --git a/tests/unit/test_settled_payment.py b/tests/unit/test_settled_payment.py new file mode 100644 index 0000000..871d30e --- /dev/null +++ b/tests/unit/test_settled_payment.py @@ -0,0 +1,373 @@ +"""Tests for the settled-payment boundary and the gateway clamp warning. + +Both mechanisms exist to protect money, and both were shipped without coverage. +The rule they encode: signing is settlement, so exactly one PAYMENT-SIGNATURE +leaves the process per user-initiated call, no matter how the paid leg fails. +""" + +import httpx +import pytest + +from blockrun_llm import LLMClient +from blockrun_llm.client import ( + _SETTLED_ATTR, + _mark_settled, + _should_fallback, + _warn_if_clamped, +) +from blockrun_llm.types import APIError, PaymentError + +from ..helpers import ( + TEST_PRIVATE_KEY, + build_chat_response, + build_payment_required_response, +) + + +def _client(handler) -> LLMClient: + client = LLMClient(private_key=TEST_PRIVATE_KEY) + client._client = httpx.Client(transport=httpx.MockTransport(handler)) + return client + + +class TestSettledTagClassification: + def test_untagged_timeout_still_falls_back(self): + """An unpaid timeout is a genuine transient failure; keep retrying.""" + assert _should_fallback(httpx.ReadTimeout("boom")) is True + + def test_untagged_503_still_falls_back(self): + assert _should_fallback(APIError("upstream", 503, None)) is True + + def test_settled_timeout_does_not_fall_back(self): + assert _should_fallback(_mark_settled(httpx.ReadTimeout("boom"))) is False + + def test_settled_network_error_does_not_fall_back(self): + assert _should_fallback(_mark_settled(httpx.ConnectError("boom"))) is False + + def test_settled_5xx_does_not_fall_back(self): + """The dominant post-settlement failure. Tagging only timeouts left + this open, so the six-settlement path survived the first fix.""" + assert _should_fallback(_mark_settled(APIError("upstream", 503, None))) is False + + def test_mark_settled_preserves_identity_and_type(self): + exc = httpx.ReadTimeout("boom") + assert _mark_settled(exc) is exc + with pytest.raises(httpx.TimeoutException): + raise exc + + +class TestTagScope: + """What must NOT be tagged, driven through the real client. The handlers + were once `except Exception`, which labeled rejected payments and SDK bugs + as settled payments. These fail if the handlers widen again.""" + + def test_payment_rejection_is_not_tagged_as_settled(self): + """A paid-leg 402 means the facilitator refused; the funds did not + move. Calling that 'settled' is exactly backwards.""" + + def handler(request: httpx.Request) -> httpx.Response: + return httpx.Response( + 402, + json={"error": "Payment Required"}, + headers={"payment-required": build_payment_required_response()}, + ) + + client = _client(handler) + with pytest.raises(PaymentError) as exc: + client.chat_completion("a/b", [{"role": "user", "content": "hi"}]) + assert getattr(exc.value, _SETTLED_ATTR, False) is False + + def test_programming_error_in_paid_leg_is_not_tagged(self, monkeypatch): + """An AttributeError raised after the paid response is an SDK bug. It + must propagate as itself, not as a settled-payment failure.""" + + def handler(request: httpx.Request) -> httpx.Response: + if "PAYMENT-SIGNATURE" in request.headers: + return httpx.Response(200, json=build_chat_response()) + return httpx.Response( + 402, + json={"error": "Payment Required"}, + headers={"payment-required": build_payment_required_response()}, + ) + + client = _client(handler) + + def boom(_response): + raise AttributeError("SDK bug in settlement capture") + + monkeypatch.setattr(client, "_capture_settlement", boom) + + with pytest.raises(AttributeError) as exc: + client.chat_completion("a/b", [{"role": "user", "content": "hi"}]) + assert getattr(exc.value, _SETTLED_ATTR, False) is False + + def test_tagged_types_are_a_superset_of_fallback_eligible(self): + """The handlers catch `(httpx.HTTPError, APIError)`. That must cover + everything _should_fallback says yes to, or a settled failure escapes + untagged and the chain pays again.""" + assert issubclass(httpx.TimeoutException, httpx.HTTPError) + assert issubclass(httpx.NetworkError, httpx.HTTPError) + assert not issubclass(PaymentError, APIError) + + def test_tagging_preserves_traceback_and_context(self): + """Handlers re-raise bare rather than `from None`, so an opaque wrapper + error keeps the cause that explains it.""" + try: + try: + raise ValueError("underlying base64 failure") + except ValueError: + raise APIError("invalid format", 500, None) + except APIError as outer: + _mark_settled(outer) + assert isinstance(outer.__context__, ValueError) + assert outer.__suppress_context__ is False + + +class TestNoSecondSettlement: + """One user-initiated call must never settle more than once.""" + + def _paid_leg_fails(self, failure): + # Count distinct signatures, not signed requests. The paid leg retries + # 502/503 once with the SAME PAYMENT-SIGNATURE, which is one settlement + # replayed, not a second charge. A new signature is a new settlement. + signed = set() + + def handler(request: httpx.Request) -> httpx.Response: + if "PAYMENT-SIGNATURE" in request.headers: + signed.add(request.headers["PAYMENT-SIGNATURE"]) + return failure() + return httpx.Response( + 402, + json={"error": "Payment Required"}, + headers={"payment-required": build_payment_required_response()}, + ) + + return signed, handler + + def test_timeout_after_payment_does_not_pay_the_next_model(self): + def fail(): + raise httpx.ReadTimeout("upstream hung after settlement") + + signed, handler = self._paid_leg_fails(fail) + client = _client(handler) + + with pytest.raises(httpx.TimeoutException): + client.chat_completion( + "primary/slow", + [{"role": "user", "content": "hi"}], + fallback_models=["fallback/good", "fallback/other"], + ) + assert len(signed) == 1, f"settled {len(signed)} times for one call" + + def test_paid_5xx_does_not_pay_the_next_model(self, monkeypatch): + """Regression: a 402-then-503 chain across 3 models signed six payments + and returned nothing.""" + monkeypatch.setattr("time.sleep", lambda _s: None) + signed, handler = self._paid_leg_fails( + lambda: httpx.Response(503, json={"error": "upstream down"}) + ) + client = _client(handler) + + with pytest.raises(APIError): + client.chat_completion( + "a/b", + [{"role": "user", "content": "hi"}], + fallback_models=["c/d", "e/f"], + ) + assert len(signed) == 1, f"settled {len(signed)} times for one call" + + def test_unpaid_failure_still_walks_the_chain(self): + """The guard must not disable legitimate free retries: if nothing was + signed, falling back costs the caller nothing.""" + seen = [] + + def handler(request: httpx.Request) -> httpx.Response: + import json as _json + + model = _json.loads(request.read())["model"] + seen.append(model) + if model == "primary/bad": + return httpx.Response(503, json={"error": "down"}) + return httpx.Response(200, json=build_chat_response()) + + client = _client(handler) + client.chat_completion( + "primary/bad", + [{"role": "user", "content": "hi"}], + fallback_models=["fallback/good"], + ) + # primary appears twice: the unpaid leg retries 502/503 once before the + # chain advances. What matters is that it advanced at all. + assert "fallback/good" in seen + + +class TestPaidStreamCleanup: + """Extracting the paid phase into its own generator changed who is + responsible for closing it.""" + + def test_abandoned_async_paid_stream_closes_its_inner_generator(self): + """Drives the real AsyncLLMClient. `async for` does not aclose the inner + generator when the outer is closed, so the paid + `async with self._client.stream(...)` would stay suspended and hold the + connection until GC finalization. Fails if the explicit + `finally: await paid.aclose()` is removed. + """ + import asyncio + + from blockrun_llm import AsyncLLMClient + + closed = [] + + def handler(request: httpx.Request) -> httpx.Response: + return httpx.Response( + 402, + json={"error": "Payment Required"}, + headers={"payment-required": build_payment_required_response()}, + ) + + async def run(): + client = AsyncLLMClient(private_key=TEST_PRIVATE_KEY) + client._client = httpx.AsyncClient(transport=httpx.MockTransport(handler)) + + async def fake_paid_phase(*_a, **_kw): + try: + yield {"chunk": 1} + yield {"chunk": 2} + finally: + closed.append(True) + + client._astream_paid_phase = fake_paid_phase + + stream = client.chat_completion_stream("a/b", [{"role": "user", "content": "hi"}]) + assert await stream.__anext__() == {"chunk": 1} + # Abandon mid-stream, exactly like `break` in a caller's loop. + await stream.aclose() + # Assert HERE, not after asyncio.run(). Without the explicit + # aclose, CPython's asyncgen finalizer still closes the inner + # generator eventually during loop teardown — which is precisely + # the "connection held until GC" behavior being fixed. Only a + # synchronous check at the close point tells the two apart. + return list(closed) + + assert asyncio.run(run()) == [True], "inner paid generator was not closed at aclose()" + + def test_sync_delegation_closes_via_yield_from(self): + """The sync path gets this for free, which is why only the async path + needed the explicit close. Recorded so nobody 'fixes' it symmetrically.""" + closed = [] + + def inner(): + try: + yield 1 + yield 2 + finally: + closed.append(True) + + def outer(): + yield from inner() + + gen = outer() + assert next(gen) == 1 + gen.close() + assert closed == [True] + + +class TestWarnIfClamped: + def test_warns_when_quoted_below_requested(self, capsys): + _warn_if_clamped( + {"model": "claude-opus-4.8", "max_tokens": 262144}, + "claude-opus-4.8 chat completion, 128000 max output tokens", + ) + err = capsys.readouterr().err + assert "clamped" in err and "262144" in err and "128000" in err + + def test_parses_comma_grouped_ceiling(self, capsys): + _warn_if_clamped({"max_tokens": 200000}, "gpt-5.5, 128,000 max output tokens") + assert "128000" in capsys.readouterr().err + + def test_silent_when_quoted_meets_the_request(self, capsys): + _warn_if_clamped({"max_tokens": 128000}, "128000 max output tokens") + _warn_if_clamped({"max_tokens": 1000}, "128000 max output tokens") + assert capsys.readouterr().err == "" + + def test_silent_when_no_ceiling_in_description(self, capsys): + _warn_if_clamped({"max_tokens": 999999}, "BlockRun AI API call") + _warn_if_clamped({"max_tokens": 999999}, None) + _warn_if_clamped({"max_tokens": 999999}, "") + assert capsys.readouterr().err == "" + + def test_bool_is_not_a_token_count(self, capsys): + _warn_if_clamped({"max_tokens": True}, "1 max output tokens") + assert capsys.readouterr().err == "" + + def test_non_string_description_does_not_raise(self, capsys): + """The field is server-controlled and JSON allows anything. A warning + must never be the reason a paid request fails.""" + _warn_if_clamped({"max_tokens": 100}, {"nested": "128 max output tokens"}) + _warn_if_clamped({"max_tokens": 100}, 12345) + assert capsys.readouterr().err == "" + + def test_ambiguous_description_stays_silent(self, capsys): + """A per-unit rate is not a ceiling. Two candidates means the format is + not what we think it is, so say nothing.""" + _warn_if_clamped( + {"max_tokens": 4096, "model": "m"}, + "$0.002 per 1000 max output tokens, 128000 max output tokens", + ) + assert capsys.readouterr().err == "" + + def test_long_description_does_not_hang(self, capsys): + """Outer defense: the scan limit caps what reaches the regex at all.""" + import time + + start = time.perf_counter() + _warn_if_clamped({"max_tokens": 100}, "9" * 100_000) + assert time.perf_counter() - start < 1.0 + assert capsys.readouterr().err == "" + + def test_pattern_itself_is_not_backtracking(self): + """Inner defense, pinned separately so removing the scan limit cannot + silently reintroduce the ReDoS. The old `(\\d[\\d,]*)` pattern took + ~1.95s on this input; the bounded one takes ~0.0002s. + """ + import time + + from blockrun_llm.client import _QUOTED_MAX_TOKENS_RE + + start = time.perf_counter() + _QUOTED_MAX_TOKENS_RE.search("9" * 16_000) + assert time.perf_counter() - start < 0.05 + + def test_pattern_still_matches_the_real_shapes(self): + from blockrun_llm.client import _QUOTED_MAX_TOKENS_RE + + for text, expected in ( + ("claude-opus-4.8 - 128000 max output tokens", "128000"), + ("gpt-5.5, 128,000 max output tokens", "128,000"), + ("262144 MAX OUTPUT TOKENS", "262144"), + ): + assert _QUOTED_MAX_TOKENS_RE.findall(text) == [expected], text + + def test_warning_fires_on_the_real_402_leg(self, capsys): + def handler(request: httpx.Request) -> httpx.Response: + if "PAYMENT-SIGNATURE" in request.headers: + return httpx.Response(200, json=build_chat_response()) + return httpx.Response( + 402, + json={"error": "Payment Required"}, + headers={ + "payment-required": build_payment_required_response( + resource={ + "url": "https://blockrun.ai/api/v1/chat/completions", + "description": "claude-opus-4.8 - 128000 max output tokens", + } + ) + }, + ) + + _client(handler).chat_completion( + "claude-opus-4.8", + [{"role": "user", "content": "hi"}], + max_tokens=262144, + ) + assert "max_tokens clamped" in capsys.readouterr().err diff --git a/tests/unit/test_solana_apikey_init.py b/tests/unit/test_solana_apikey_init.py new file mode 100644 index 0000000..fb179a2 --- /dev/null +++ b/tests/unit/test_solana_apikey_init.py @@ -0,0 +1,29 @@ +"""An account credential must never initialize either Solana signer.""" + +import pytest + +from blockrun_llm import AsyncSolanaLLMClient, SolanaLLMClient, solana_client + + +@pytest.mark.parametrize("client_class", [SolanaLLMClient, AsyncSolanaLLMClient]) +@pytest.mark.parametrize("explicit", [True, False]) +def test_account_client_does_not_construct_signer(monkeypatch, client_class, explicit): + key = "brk_live_AccountAcceptanceFixture123456" + monkeypatch.setenv("BLOCKRUN_API_KEY", key) + monkeypatch.setenv("SOLANA_WALLET_KEY", "invalid-leftover-wallet") + + def forbidden(*args, **kwargs): + pytest.fail("API account initialized a wallet signer") + + monkeypatch.setattr(solana_client, "_create_signer", forbidden) + client = client_class(private_key=key if explicit else None) + assert client.payment_mode == "apikey" + assert client._private_key is None + assert client._client.headers["Authorization"] == f"Bearer {key}" + assert client._x402_client is None + if client_class is SolanaLLMClient: + client.close() + else: + import asyncio + + asyncio.run(client.close()) diff --git a/tests/unit/test_solana_client.py b/tests/unit/test_solana_client.py new file mode 100644 index 0000000..102440d --- /dev/null +++ b/tests/unit/test_solana_client.py @@ -0,0 +1,81 @@ +"""Unit tests for SolanaLLMClient.""" + +import os + +import pytest + +from blockrun_llm.solana_client import AsyncSolanaLLMClient, SolanaLLMClient + +TEST_BS58_KEY = ( + "433C7KFcM4y1ZEVdZYSH7wheSNAM384UcbgXEyD5FV7Q2HsQ1BwjEDx4GbBZUqPkZTVhFPyLyuZnzK8wCeAkU7wG" +) + + +class TestSolanaLLMClientInit: + def test_init_with_key(self): + client = SolanaLLMClient(private_key=TEST_BS58_KEY) + assert client is not None + + def test_init_from_env(self): + os.environ["SOLANA_WALLET_KEY"] = TEST_BS58_KEY + client = SolanaLLMClient() + assert client is not None + del os.environ["SOLANA_WALLET_KEY"] + + def test_raises_without_key(self, monkeypatch): + # No env var AND no wallet session on disk → must still raise. Patch the + # session loader so the test is deterministic regardless of whether the + # machine running it happens to have ~/.blockrun/.solana-session. + monkeypatch.delenv("SOLANA_WALLET_KEY", raising=False) + monkeypatch.setattr("blockrun_llm.solana_wallet.load_solana_wallet", lambda: None) + with pytest.raises(ValueError, match="No credential configured"): + SolanaLLMClient() + + def test_init_from_session_file(self, monkeypatch): + # No env var, but a wallet session exists on disk → auto-load it (parity + # with the Base LLMClient, which already falls back to load_wallet()). + monkeypatch.delenv("SOLANA_WALLET_KEY", raising=False) + monkeypatch.setattr("blockrun_llm.solana_wallet.load_solana_wallet", lambda: TEST_BS58_KEY) + client = SolanaLLMClient() + assert client is not None + + @pytest.mark.asyncio + async def test_async_init_from_session_file(self, monkeypatch): + # Same disk fallback on the async client (identical code path). + monkeypatch.delenv("SOLANA_WALLET_KEY", raising=False) + monkeypatch.setattr("blockrun_llm.solana_wallet.load_solana_wallet", lambda: TEST_BS58_KEY) + client = AsyncSolanaLLMClient() + assert client is not None + await client.close() + + def test_raises_on_invalid_key(self, monkeypatch): + # A malformed key (here from the disk fallback) must surface a clean + # ValueError, not a raw base58/solders exception. Parity with Base. + monkeypatch.delenv("SOLANA_WALLET_KEY", raising=False) + monkeypatch.setattr( + "blockrun_llm.solana_wallet.load_solana_wallet", lambda: "not-a-valid-key" + ) + with pytest.raises(ValueError, match="[Ii]nvalid Solana private key"): + SolanaLLMClient() + + def test_default_api_url(self): + client = SolanaLLMClient(private_key=TEST_BS58_KEY) + assert client.is_solana() + + def test_custom_api_url(self): + client = SolanaLLMClient( + private_key=TEST_BS58_KEY, api_url="https://custom.example.com/api" + ) + assert not client.is_solana() + + def test_get_wallet_address(self): + client = SolanaLLMClient(private_key=TEST_BS58_KEY) + addr = client.get_wallet_address() + assert isinstance(addr, str) + assert len(addr) >= 32 + + def test_get_spending_initial(self): + client = SolanaLLMClient(private_key=TEST_BS58_KEY) + spending = client.get_spending() + assert spending["total_usd"] == 0.0 + assert spending["calls"] == 0 diff --git a/tests/unit/test_solana_max_tokens.py b/tests/unit/test_solana_max_tokens.py new file mode 100644 index 0000000..614a87d --- /dev/null +++ b/tests/unit/test_solana_max_tokens.py @@ -0,0 +1,86 @@ +"""max_tokens validation on the Solana chain (issue #31). + +The guard was Base-only: `validate_max_tokens` was called from client.py and +nowhere else, while solana_client.py put the caller's value straight into paid +request bodies. `max_tokens=2_000_000` raised on Base and was signed and sent on +Solana. Since the gateway clamps rather than rejects, there was no server-side +backstop behind the missing client-side one. + +Guarded with importorskip: the 3.9 CI job installs without the solana extra, and +an unguarded Solana test file turns that job red (see #19/#20). +""" + +from __future__ import annotations + +from unittest import mock + +import httpx +import pytest + +pytest.importorskip("x402") +pytest.importorskip("solders") + +from blockrun_llm import SolanaLLMClient +from blockrun_llm.validation import MAX_TOKENS_SANITY_LIMIT + +MESSAGES = [{"role": "user", "content": "hi"}] + + +def _client() -> SolanaLLMClient: + """A client whose transport fails loudly. Validation must reject before any + request is built, so a passing test proves nothing reached the network.""" + + def explode(request: httpx.Request) -> httpx.Response: + raise AssertionError(f"validation let a bad max_tokens reach {request.url}") + + with ( + mock.patch("blockrun_llm.solana_client.register_exact_svm_client"), + mock.patch("blockrun_llm.solana_client._create_signer"), + ): + client = SolanaLLMClient( + private_key="bogus_signer_is_patched", + api_url="https://sol.blockrun.ai/api", + rpc_url="http://test", + ) + client._x402_client = mock.MagicMock() + client._client = httpx.Client(transport=httpx.MockTransport(explode)) + client._address = "11111111111111111111111111111111" + return client + + +class TestSolanaMaxTokensValidation: + """Every Solana chat entry point that puts max_tokens in a paid body.""" + + def test_chat_completion_rejects_implausible(self): + with pytest.raises(ValueError, match="implausibly large"): + _client().chat_completion("a/b", MESSAGES, max_tokens=2_000_000) + + def test_chat_completion_stream_rejects_implausible(self): + with pytest.raises(ValueError, match="implausibly large"): + list(_client().chat_completion_stream("a/b", MESSAGES, max_tokens=2_000_000)) + + def test_rejects_zero_and_negative(self): + for bad in (0, -1): + with pytest.raises(ValueError, match="positive"): + _client().chat_completion("a/b", MESSAGES, max_tokens=bad) + + def test_rejects_bool(self): + with pytest.raises(ValueError, match="bool"): + _client().chat_completion("a/b", MESSAGES, max_tokens=True) + + def test_real_ceilings_are_not_capped(self): + """The bound must never be the binding constraint on either chain. + These reach the transport, which is what the AssertionError proves.""" + for real_ceiling in (128_000, 262_144, MAX_TOKENS_SANITY_LIMIT): + with pytest.raises(AssertionError, match="reach"): + _client().chat_completion("a/b", MESSAGES, max_tokens=real_ceiling) + + def test_both_chains_share_one_bound(self): + """Base and Solana must agree, or the SDK's guard is not an invariant.""" + from blockrun_llm import LLMClient + + base = LLMClient(private_key="0x" + "11" * 32) + with pytest.raises(ValueError, match="implausibly large"): + base.chat_completion("a/b", MESSAGES, max_tokens=2_000_000) + with pytest.raises(ValueError, match="implausibly large"): + _client().chat_completion("a/b", MESSAGES, max_tokens=2_000_000) diff --git a/tests/unit/test_solana_media.py b/tests/unit/test_solana_media.py new file mode 100644 index 0000000..bb5433a --- /dev/null +++ b/tests/unit/test_solana_media.py @@ -0,0 +1,751 @@ +"""Unit tests for the Solana media surface added in #16 (video/music/speech/ +sound-effects/price/list_voices) plus the mid-poll re-sign payment-terms guard. + +Payment flow is mocked at the httpx transport level (402 on the unsigned probe, +success once a PAYMENT-SIGNATURE is present); the x402 codec + signer are +stubbed so no wallet or network is needed — same approach as +test_solana_timeout_routing.py. +""" + +from __future__ import annotations + +from types import SimpleNamespace +from typing import Any +from unittest import mock + +import httpx +import pytest + +# Solana x402 extras (x402[svm]) require Python >= 3.10; skip the whole module +# on 3.9, where they aren't installed and the codec stubs below have nothing to +# patch. Mirrors test_solana_timeout_routing.py. +pytest.importorskip("x402") +pytest.importorskip("solders") + +from blockrun_llm.solana_client import ( + AsyncSolanaLLMClient, + SolanaLLMClient, + _assert_same_payment_terms, +) +from blockrun_llm.types import ( + APIError, + MusicResponse, + PaymentError, + SpeechResponse, +) + +# --------------------------------------------------------------------------- +# _assert_same_payment_terms — the mid-poll re-sign guard +# --------------------------------------------------------------------------- + + +def _payload(amount: str, pay_to: str) -> SimpleNamespace: + return SimpleNamespace(accepted=SimpleNamespace(amount=amount, pay_to=pay_to)) + + +class TestPaymentTermsGuard: + def test_same_terms_pass(self) -> None: + # Identical amount + recipient (the normal stale-blockhash re-sign) is + # allowed through with no exception. + _assert_same_payment_terms(_payload("1000000", "WALLET_A"), "1000000", "WALLET_A") + + def test_amount_change_rejected(self) -> None: + with pytest.raises(PaymentError, match="changed the payment terms"): + _assert_same_payment_terms(_payload("9999999", "WALLET_A"), "1000000", "WALLET_A") + + def test_recipient_change_rejected(self) -> None: + with pytest.raises(PaymentError, match="changed the payment terms"): + _assert_same_payment_terms(_payload("1000000", "ATTACKER"), "1000000", "WALLET_A") + + def test_amount_type_coerced_before_compare(self) -> None: + # int vs str for the same value must not trip the guard. + _assert_same_payment_terms(_payload(1000000, "WALLET_A"), "1000000", "WALLET_A") + + +# --------------------------------------------------------------------------- +# Media dispatch — body construction + response parsing over the mocked flow +# --------------------------------------------------------------------------- + + +@pytest.fixture(autouse=True) +def _stub_x402_codec(monkeypatch: pytest.MonkeyPatch) -> None: + monkeypatch.setattr( + "blockrun_llm.solana_client.decode_payment_required_header", + lambda header: {"stub": True}, + ) + monkeypatch.setattr( + "blockrun_llm.solana_client.encode_payment_signature_header", + lambda payload: "stub-signature", + ) + + +@pytest.fixture(autouse=True) +def _no_disk_cache(monkeypatch: pytest.MonkeyPatch) -> None: + monkeypatch.setattr("blockrun_llm.cache.get_cached", lambda *a, **k: None) + monkeypatch.setattr("blockrun_llm.cache.save_to_cache", lambda *a, **k: None) + + +def _make_client(handler: Any) -> SolanaLLMClient: + with ( + mock.patch("blockrun_llm.solana_client.register_exact_svm_client"), + mock.patch("blockrun_llm.solana_client._create_signer"), + ): + client = SolanaLLMClient( + private_key="bogus_signer_is_patched", + api_url="https://sol.blockrun.ai/api", + rpc_url="http://test", + ) + + class _FakePayload: + class accepted: + amount = "1000000" + pay_to = "GsbwXfJraMomNxBcpR3DBNxnKwZbyq7YCoDdSLDwzxdV" + + client._x402_client = mock.MagicMock() + client._x402_client.create_payment_payload.return_value = _FakePayload() + client._client = httpx.Client(transport=httpx.MockTransport(handler)) + client._address = "11111111111111111111111111111111" + return client + + +def _paid_flow(calls: list[httpx.Request], ok_body: dict[str, Any]): + """402 on the unsigned probe, then ``ok_body`` once signed. Captures the + signed request so tests can assert the forwarded JSON body + path.""" + + def handler(request: httpx.Request) -> httpx.Response: + if "PAYMENT-SIGNATURE" not in request.headers: + return httpx.Response( + 402, + headers={"content-type": "application/json", "payment-required": "stub"}, + json={"error": "Payment Required"}, + ) + calls.append(request) + return httpx.Response(200, json=ok_body, headers={"content-type": "application/json"}) + + return handler + + +_MUSIC_OK = {"created": 1, "model": "minimax/music-2.5+", "data": [{"url": "https://cdn/x.mp3"}]} +_SPEECH_OK = { + "created": 1, + "model": "elevenlabs/flash-v2.5", + "data": [{"url": "https://cdn/x.wav"}], +} + + +class TestMediaDispatch: + def test_music_body_and_response(self) -> None: + import json + + calls: list[httpx.Request] = [] + client = _make_client(_paid_flow(calls, _MUSIC_OK)) + resp = client.music("lo-fi beats") + assert isinstance(resp, MusicResponse) + assert resp.data[0].url == "https://cdn/x.mp3" + assert calls[-1].url.path == "/api/v1/audio/generations" + sent = json.loads(calls[-1].content) + assert sent["model"] == "minimax/music-2.5+" + assert sent["instrumental"] is True + + def test_speech_body_and_response(self) -> None: + import json + + calls: list[httpx.Request] = [] + client = _make_client(_paid_flow(calls, _SPEECH_OK)) + resp = client.speech("hello world", voice="sarah") + assert isinstance(resp, SpeechResponse) + assert resp.data[0].url == "https://cdn/x.wav" + assert calls[-1].url.path == "/api/v1/audio/speech" + sent = json.loads(calls[-1].content) + assert sent["input"] == "hello world" + assert sent["voice"] == "sarah" + + def test_sound_effect_endpoint(self) -> None: + calls: list[httpx.Request] = [] + client = _make_client(_paid_flow(calls, _SPEECH_OK)) + client.sound_effect("thunder clap") + assert calls[-1].url.path == "/api/v1/audio/sound-effects" + + def test_list_voices_returns_list_not_envelope(self) -> None: + # Regression: the gateway returns {"data": [...]}, and list_voices must + # return the list, not the whole dict. + voices = [{"id": "sarah"}, {"id": "adam"}] + + def handler(request: httpx.Request) -> httpx.Response: + return httpx.Response(200, json={"data": voices}) + + client = _make_client(handler) + assert client.list_voices() == voices + + +# --------------------------------------------------------------------------- +# Local validation — must reject before any HTTP / payment +# --------------------------------------------------------------------------- + + +class TestLocalValidation: + def test_music_lyrics_with_instrumental_rejected(self) -> None: + client = _make_client(lambda r: httpx.Response(500)) # never reached + with pytest.raises(ValueError, match="lyrics"): + client.music("pop", instrumental=True, lyrics="la la la") + + def test_video_mutually_exclusive_image_and_face(self) -> None: + client = _make_client(lambda r: httpx.Response(500)) + with pytest.raises(ValueError, match="mutually exclusive"): + client.video("a cat", image_url="https://x/y.png", real_face_asset_id="ta_abc") + + def test_video_bad_face_id_prefix(self) -> None: + client = _make_client(lambda r: httpx.Response(500)) + with pytest.raises(ValueError, match="ta_"): + client.video("a cat", real_face_asset_id="not_a_valid_id") + + def test_portrait_enroll_requires_http_url(self) -> None: + client = _make_client(lambda r: httpx.Response(500)) + with pytest.raises(ValueError, match="image_url"): + client.portrait_enroll("Alice", "ftp://bad/url") + + +# --------------------------------------------------------------------------- +# price() — missing "price" in a paid body must not raise a raw KeyError +# --------------------------------------------------------------------------- + + +class TestPriceRobustness: + def test_missing_price_field_is_clean_error_not_keyerror(self) -> None: + def handler(request: httpx.Request) -> httpx.Response: + if "PAYMENT-SIGNATURE" not in request.headers: + return httpx.Response( + 402, + headers={"content-type": "application/json", "payment-required": "stub"}, + json={"error": "Payment Required"}, + ) + # Paid 200 but the body is missing "price" — must surface as a + # pydantic validation error, not a bare KeyError. + return httpx.Response(200, json={"symbol": "BTCUSD"}) + + client = _make_client(handler) + with pytest.raises(Exception) as exc_info: + client.price("crypto", "BTCUSD") + assert not isinstance(exc_info.value, KeyError) + + +# --------------------------------------------------------------------------- +# Path-segment guard — LLM-controlled values can't escape the URL path +# --------------------------------------------------------------------------- + + +class TestPathSegmentGuard: + def test_symbol_with_slash_rejected(self) -> None: + client = _make_client(lambda r: httpx.Response(500)) + with pytest.raises(ValueError, match="symbol"): + client.price("crypto", "../../secret") + + def test_network_with_traversal_rejected(self) -> None: + client = _make_client(lambda r: httpx.Response(500)) + with pytest.raises(ValueError, match="network"): + client.rpc("../evil", "eth_blockNumber") + + +# --------------------------------------------------------------------------- +# poll_url host pinning — the signed PAYMENT-SIGNATURE must not go off-host +# --------------------------------------------------------------------------- + + +class TestPollUrlHostPin: + def test_relative_poll_url_resolved_to_api_host(self) -> None: + client = _make_client(lambda r: httpx.Response(500)) + assert ( + client._absolute_url("/api/v1/videos/generations/JOB") + == "https://sol.blockrun.ai/api/v1/videos/generations/JOB" + ) + + def test_absolute_same_host_passes(self) -> None: + client = _make_client(lambda r: httpx.Response(500)) + url = "https://sol.blockrun.ai/api/v1/videos/generations/JOB" + assert client._absolute_url(url) == url + + def test_absolute_cross_host_rejected(self) -> None: + client = _make_client(lambda r: httpx.Response(500)) + with pytest.raises(APIError, match="off-host"): + client._absolute_url("https://evil.example.com/api/v1/videos/generations/JOB") + + def test_absolute_http_downgrade_rejected(self) -> None: + client = _make_client(lambda r: httpx.Response(500)) + with pytest.raises(APIError, match="off-host"): + client._absolute_url("http://sol.blockrun.ai/api/v1/videos/generations/JOB") + + +# --------------------------------------------------------------------------- +# Mid-poll re-sign — end-to-end through the poll loop (sync + async parity) +# --------------------------------------------------------------------------- + + +def _make_async_client(handler: Any) -> AsyncSolanaLLMClient: + with ( + mock.patch("blockrun_llm.solana_client.register_exact_svm_client"), + mock.patch("blockrun_llm.solana_client._create_signer"), + ): + client = AsyncSolanaLLMClient( + private_key="bogus_signer_is_patched", + api_url="https://sol.blockrun.ai/api", + rpc_url="http://test", + ) + + class _FakePayload: + class accepted: + amount = "1000000" + pay_to = "GsbwXfJraMomNxBcpR3DBNxnKwZbyq7YCoDdSLDwzxdV" + + client._x402_client = mock.MagicMock() + # Async _sign_payment awaits create_payment_payload — must return a coroutine. + client._x402_client.create_payment_payload = mock.AsyncMock(return_value=_FakePayload()) + client._client = httpx.AsyncClient(transport=httpx.MockTransport(handler)) + client._address = "11111111111111111111111111111111" + return client + + +def _resign_handler(signed_poll_codes: list[int]): + """Drive a video job through the mid-poll re-sign path. + + probe → 402; signed POST → 202 + poll_url; each *signed* GET poll returns + the next code from ``signed_poll_codes`` (402 = settlement failed, 200 = + completed); an *unsigned* GET is the re-challenge and always hands back a + fresh 402 payment-required so the client re-signs. + """ + pr = {"content-type": "application/json", "payment-required": "stub"} + completed = { + "status": "completed", + "created": 1, + "model": "xai/grok-imagine-video", + "data": [{"url": "https://cdn/v.mp4"}], + } + state = {"i": 0} + + def handler(request: httpx.Request) -> httpx.Response: + has_sig = "PAYMENT-SIGNATURE" in request.headers + if request.method == "POST": + if not has_sig: # unsigned probe + return httpx.Response(402, headers=pr, json={"error": "Payment Required"}) + return httpx.Response( # signed submit + 202, + json={ + "id": "JOB", + "poll_url": "/api/v1/videos/generations/JOB", + "status": "queued", + }, + ) + if not has_sig: # unsigned re-challenge → trigger a re-sign + return httpx.Response(402, headers=pr, json={"error": "Payment Required"}) + code = signed_poll_codes[min(state["i"], len(signed_poll_codes) - 1)] + state["i"] += 1 + if code == 200: + return httpx.Response(200, json=completed, headers={"content-type": "application/json"}) + return httpx.Response(402, headers=pr, json={"error": "settlement failed"}) + + return handler + + +_HELPER_KW: dict[str, Any] = { + "poll_budget_seconds": 5.0, + "poll_interval_seconds": 0.001, + "max_resigns": 2, + "label": "Video generation", +} +_VIDEO_BODY = {"model": "xai/grok-imagine-video", "prompt": "a cat"} + + +class TestResignEndToEnd: + def test_sync_resign_same_terms_then_completes(self) -> None: + # poll 402 (stale blockhash) → re-challenge → re-sign (same terms, guard + # passes) → next poll 200 completed. + client = _make_client(_resign_handler([402, 200])) + data = client._request_image_with_payment( + "/v1/videos/generations", dict(_VIDEO_BODY), **_HELPER_KW + ) + assert data["data"][0]["url"] == "https://cdn/v.mp4" + + def test_sync_resign_reprice_propagates(self, monkeypatch: pytest.MonkeyPatch) -> None: + # A guard rejection on the re-signed challenge must propagate, NOT be + # swallowed by the re-sign try/except and masked as a generic 402. + monkeypatch.setattr( + "blockrun_llm.solana_client._assert_same_payment_terms", + mock.Mock(side_effect=PaymentError("repriced")), + ) + client = _make_client(_resign_handler([402, 200])) + with pytest.raises(PaymentError, match="repriced"): + client._request_image_with_payment( + "/v1/videos/generations", dict(_VIDEO_BODY), **_HELPER_KW + ) + + async def test_async_resign_same_terms_then_completes(self) -> None: + client = _make_async_client(_resign_handler([402, 200])) + try: + data = await client._request_image_with_payment( + "/v1/videos/generations", dict(_VIDEO_BODY), **_HELPER_KW + ) + assert data["data"][0]["url"] == "https://cdn/v.mp4" + finally: + await client._client.aclose() + + async def test_async_resign_reprice_propagates(self, monkeypatch: pytest.MonkeyPatch) -> None: + # Async parity with the sync guard: a re-price is rejected and the + # PaymentError propagates out of the poll loop. + monkeypatch.setattr( + "blockrun_llm.solana_client._assert_same_payment_terms", + mock.Mock(side_effect=PaymentError("repriced")), + ) + client = _make_async_client(_resign_handler([402, 200])) + try: + with pytest.raises(PaymentError, match="repriced"): + await client._request_image_with_payment( + "/v1/videos/generations", dict(_VIDEO_BODY), **_HELPER_KW + ) + finally: + await client._client.aclose() + + +# --------------------------------------------------------------------------- +# Proactive per-poll re-sign — keeps the settlement blockhash fresh even when +# NO poll ever 402s (the 1080p Seedance case: upstream status flaps +# completed<->in_progress for minutes and would otherwise settle a stale +# signature). Distinct from the on-402 re-sign guard tested above. +# --------------------------------------------------------------------------- + + +def _fresh_sig_handler(n_in_progress: int, poll_sigs: list[str]): + """Video job that NEVER 402s on a poll: n_in_progress in-progress polls, + then completed. Records the PAYMENT-SIGNATURE seen on every signed poll so a + test can assert the proactive re-sign refreshed it each time.""" + completed = { + "status": "completed", + "created": 1, + "model": "xai/grok-imagine-video", + "data": [{"url": "https://cdn/v.mp4"}], + } + state = {"i": 0} + + def handler(request: httpx.Request) -> httpx.Response: + if request.method == "POST": + if "PAYMENT-SIGNATURE" not in request.headers: + return httpx.Response( + 402, + headers={"content-type": "application/json", "payment-required": "stub"}, + json={"error": "Payment Required"}, + ) + return httpx.Response( + 202, + json={ + "id": "JOB", + "poll_url": "/api/v1/videos/generations/JOB", + "status": "queued", + }, + ) + poll_sigs.append(request.headers.get("PAYMENT-SIGNATURE")) + state["i"] += 1 + if state["i"] <= n_in_progress: + return httpx.Response( + 202, json={"status": "in_progress"}, headers={"content-type": "application/json"} + ) + return httpx.Response(200, json=completed, headers={"content-type": "application/json"}) + + return handler + + +class TestProactiveResign: + def test_sync_refreshes_signature_every_poll(self, monkeypatch: pytest.MonkeyPatch) -> None: + import itertools + + counter = itertools.count() + monkeypatch.setattr( + "blockrun_llm.solana_client.encode_payment_signature_header", + lambda payload: f"sig-{next(counter)}", + ) + # Fire the proactive re-sign on every poll (0s freshness window). + monkeypatch.setattr(SolanaLLMClient, "MEDIA_RESIGN_FRESH_SECONDS", 0.0) + + poll_sigs: list[str] = [] + client = _make_client(_fresh_sig_handler(3, poll_sigs)) + data = client._request_image_with_payment( + "/v1/videos/generations", dict(_VIDEO_BODY), **_HELPER_KW + ) + assert data["data"][0]["url"] == "https://cdn/v.mp4" + # 3 in-progress + 1 completed, and every signed poll carried a DISTINCT + # (freshly re-signed) signature — the completed poll never reused the + # stale submit-time one. + assert len(poll_sigs) == 4 + assert len(set(poll_sigs)) == 4, poll_sigs + + @pytest.mark.asyncio + async def test_async_refreshes_signature_every_poll( + self, monkeypatch: pytest.MonkeyPatch + ) -> None: + import itertools + + counter = itertools.count() + monkeypatch.setattr( + "blockrun_llm.solana_client.encode_payment_signature_header", + lambda payload: f"sig-{next(counter)}", + ) + monkeypatch.setattr(SolanaLLMClient, "MEDIA_RESIGN_FRESH_SECONDS", 0.0) + + poll_sigs: list[str] = [] + client = _make_async_client(_fresh_sig_handler(3, poll_sigs)) + try: + data = await client._request_image_with_payment( + "/v1/videos/generations", dict(_VIDEO_BODY), **_HELPER_KW + ) + assert data["data"][0]["url"] == "https://cdn/v.mp4" + assert len(poll_sigs) == 4 + assert len(set(poll_sigs)) == 4, poll_sigs + finally: + await client._client.aclose() + + # max_resigns == 0 (the image path) must NOT proactively re-sign, even with a + # 0s freshness window: every poll reuses the single submit-time signature so + # the image flow is provably untouched by the video-only fix. + _IMAGE_KW: dict[str, Any] = { + "poll_budget_seconds": 5.0, + "poll_interval_seconds": 0.001, + "max_resigns": 0, + "label": "Image generation", + } + + def test_sync_image_path_never_resigns(self, monkeypatch: pytest.MonkeyPatch) -> None: + import itertools + + counter = itertools.count() + monkeypatch.setattr( + "blockrun_llm.solana_client.encode_payment_signature_header", + lambda payload: f"sig-{next(counter)}", + ) + monkeypatch.setattr(SolanaLLMClient, "MEDIA_RESIGN_FRESH_SECONDS", 0.0) + + poll_sigs: list[str] = [] + client = _make_client(_fresh_sig_handler(3, poll_sigs)) + data = client._request_image_with_payment( + "/v1/images/generations", dict(_VIDEO_BODY), **self._IMAGE_KW + ) + assert data["data"][0]["url"] == "https://cdn/v.mp4" + # 3 in-progress + 1 completed, every poll carrying the SAME submit-time + # signature — the proactive re-sign never fired for max_resigns == 0. + assert len(poll_sigs) == 4 + assert len(set(poll_sigs)) == 1, poll_sigs + + @pytest.mark.asyncio + async def test_async_image_path_never_resigns(self, monkeypatch: pytest.MonkeyPatch) -> None: + import itertools + + counter = itertools.count() + monkeypatch.setattr( + "blockrun_llm.solana_client.encode_payment_signature_header", + lambda payload: f"sig-{next(counter)}", + ) + monkeypatch.setattr(SolanaLLMClient, "MEDIA_RESIGN_FRESH_SECONDS", 0.0) + + poll_sigs: list[str] = [] + client = _make_async_client(_fresh_sig_handler(3, poll_sigs)) + try: + data = await client._request_image_with_payment( + "/v1/images/generations", dict(_VIDEO_BODY), **self._IMAGE_KW + ) + assert data["data"][0]["url"] == "https://cdn/v.mp4" + assert len(poll_sigs) == 4 + assert len(set(poll_sigs)) == 1, poll_sigs + finally: + await client._client.aclose() + + +_IMAGE_OK = {"created": 1, "model": "openai/gpt-image-2", "data": [{"url": "https://cdn/x.png"}]} +_VIDEO_OK = { + "created": 1, + "model": "xai/grok-imagine-video", + "data": [{"url": "https://cdn/x.mp4"}], +} +_DATA_URI = "data:image/png;base64,AA==" + + +class TestSolanaImageQuality: + """`quality` is a Solana-only latency/fidelity knob (openai/gpt-image-* on + the gateway). The Base gateway has no such field and zod would silently + strip it, which is why ImageClient deliberately rejects it — see + test_image_parameter_validation.test_generate_rejects_quality_parameter. + """ + + def test_image_forwards_quality(self) -> None: + import json + + calls: list[httpx.Request] = [] + client = _make_client(_paid_flow(calls, _IMAGE_OK)) + client.image("a cat", model="openai/gpt-image-2", quality="low") + sent = json.loads(calls[-1].content) + assert sent["quality"] == "low" + + def test_image_omits_quality_when_unset(self) -> None: + import json + + calls: list[httpx.Request] = [] + client = _make_client(_paid_flow(calls, _IMAGE_OK)) + client.image("a cat") + assert "quality" not in json.loads(calls[-1].content) + + def test_image_edit_forwards_quality(self) -> None: + import json + + calls: list[httpx.Request] = [] + client = _make_client(_paid_flow(calls, _IMAGE_OK)) + client.image_edit("make it green", _DATA_URI, quality="high") + sent = json.loads(calls[-1].content) + assert sent["quality"] == "high" + assert calls[-1].url.path == "/api/v1/images/image2image" + + @pytest.mark.parametrize("value", ["low", "medium", "high", "auto"]) + def test_image_accepts_every_gateway_quality(self, value: str) -> None: + import json + + calls: list[httpx.Request] = [] + client = _make_client(_paid_flow(calls, _IMAGE_OK)) + client.image("a cat", model="openai/gpt-image-2", quality=value) + assert json.loads(calls[-1].content)["quality"] == value + + def test_image_rejects_unknown_quality_before_paying(self) -> None: + calls: list[httpx.Request] = [] + client = _make_client(_paid_flow(calls, _IMAGE_OK)) + with pytest.raises(ValueError, match="quality must be one of"): + client.image("a cat", model="openai/gpt-image-2", quality="hd") + assert calls == [] # rejected locally — no request, no payment + + def test_image_edit_rejects_unknown_quality_before_paying(self) -> None: + calls: list[httpx.Request] = [] + client = _make_client(_paid_flow(calls, _IMAGE_OK)) + with pytest.raises(ValueError, match="quality must be one of"): + client.image_edit("make it green", _DATA_URI, quality="ultra") + assert calls == [] + + +class TestSolanaVideoInputType: + def test_video_forwards_input_type(self) -> None: + import json + + calls: list[httpx.Request] = [] + client = _make_client(_paid_flow(calls, _VIDEO_OK)) + client.video( + "the flower blooms", + image_url="https://example.com/bud.jpg", + last_frame_url="https://example.com/bloom.jpg", + input_type="first_last_frame", + ) + assert json.loads(calls[-1].content)["input_type"] == "first_last_frame" + + def test_video_omits_input_type_when_unset(self) -> None: + import json + + calls: list[httpx.Request] = [] + client = _make_client(_paid_flow(calls, _VIDEO_OK)) + client.video("a calm lake") + assert "input_type" not in json.loads(calls[-1].content) + + def test_video_rejects_unknown_input_type_before_paying(self) -> None: + calls: list[httpx.Request] = [] + client = _make_client(_paid_flow(calls, _VIDEO_OK)) + with pytest.raises(ValueError, match="input_type must be one of"): + client.video("x", input_type="img") + assert calls == [] + + +class TestSharedVideoBodyBuilder: + """Sync and async video() share _build_video_body so they can't drift.""" + + def test_input_type_reaches_body(self) -> None: + body = SolanaLLMClient._build_video_body( + "x", + model=None, + image_url=None, + last_frame_url=None, + reference_image_urls=None, + real_face_asset_id=None, + duration_seconds=None, + aspect_ratio=None, + resolution=None, + generate_audio=None, + seed=None, + watermark=None, + return_last_frame=None, + input_type="text", + ) + assert body["input_type"] == "text" + + +class TestAsyncMediaParamParity: + """The async client duplicates the sync call shape, so it needs the same + body assertions. A signature check would pass even if the param were + accepted and then never forwarded — the regression worth catching. + """ + + @pytest.mark.asyncio + async def test_async_video_forwards_input_type(self) -> None: + import json + + calls: list[httpx.Request] = [] + client = _make_async_client(_paid_flow(calls, _VIDEO_OK)) + await client.video("a calm lake", input_type="text") + assert json.loads(calls[-1].content)["input_type"] == "text" + + @pytest.mark.asyncio + async def test_async_video_rejects_unknown_input_type_before_paying(self) -> None: + calls: list[httpx.Request] = [] + client = _make_async_client(_paid_flow(calls, _VIDEO_OK)) + with pytest.raises(ValueError, match="input_type must be one of"): + await client.video("x", input_type="img") + assert calls == [] + + @pytest.mark.asyncio + async def test_async_image_forwards_quality(self) -> None: + import json + + calls: list[httpx.Request] = [] + client = _make_async_client(_paid_flow(calls, _IMAGE_OK)) + await client.image("a cat", model="openai/gpt-image-2", quality="low") + assert json.loads(calls[-1].content)["quality"] == "low" + + @pytest.mark.asyncio + async def test_async_image_edit_forwards_quality(self) -> None: + import json + + calls: list[httpx.Request] = [] + client = _make_async_client(_paid_flow(calls, _IMAGE_OK)) + await client.image_edit("make it green", _DATA_URI, quality="high") + assert json.loads(calls[-1].content)["quality"] == "high" + assert calls[-1].url.path == "/api/v1/images/image2image" + + @pytest.mark.asyncio + async def test_async_image_rejects_unknown_quality_before_paying(self) -> None: + calls: list[httpx.Request] = [] + client = _make_async_client(_paid_flow(calls, _IMAGE_OK)) + with pytest.raises(ValueError, match="quality must be one of"): + await client.image("a cat", quality="hd") + assert calls == [] + + +@pytest.mark.asyncio +async def test_async_mixed_video_references(): + import json + + calls: list[httpx.Request] = [] + client = _make_async_client(_paid_flow(calls, _VIDEO_OK)) + await client.video( + "test", + model="bytedance/seedance-2.0", + reference_image_urls=["https://example.com/person.png"], + reference_videos=[{"url": "https://example.com/motion.mp4"}], + reference_audios=[{"url": "https://example.com/music.mp3"}], + bitrate_mode="high", + safety_identifier="test", + return_last_frame=True, + ) + body = json.loads(calls[0].content) + assert body["reference_videos"] == [{"url": "https://example.com/motion.mp4"}] + assert body["reference_audios"] == [{"url": "https://example.com/music.mp3"}] + assert body["reference_image_urls"] == ["https://example.com/person.png"] + assert body["bitrate_mode"] == "high" + assert body["return_last_frame"] is True + await client.close() diff --git a/tests/unit/test_solana_retry_classification.py b/tests/unit/test_solana_retry_classification.py new file mode 100644 index 0000000..f230dca --- /dev/null +++ b/tests/unit/test_solana_retry_classification.py @@ -0,0 +1,124 @@ +"""Tests for the permanent-vs-transient classification on Solana paths. + +Covers issue #6: ``transaction_simulation_failed`` (and a couple of close +cousins) must be classified as PERMANENT so the SDK does not waste +5+ minutes of wall-clock re-running 30-180s upstream generations only +to hit the same wall. Even when the exception type is "transient" +(httpx.Timeout / NetworkError), if the underlying reason text matches a +permanent classification, no fallback. +""" + +from __future__ import annotations + +import httpx +import pytest + +from blockrun_llm.solana_client import ( + _is_permanent_payment_error, + _should_fallback_solana, +) +from blockrun_llm.types import APIError, PaymentError + + +class TestIsPermanentPaymentError: + @pytest.mark.parametrize( + "reason", + [ + "transaction_simulation_failed", + "invalid_exact_svm_payload_transaction_simulation_failed", # CDP long form + "blockhash not found", + "block height exceeded", + "insufficient funds", + "insufficient balance for tx fee", + "invalid signature on payload", + "invalid_payload: amount mismatch", + "payment_expired after 300s", + "authorization is used (replay)", + "TRANSACTION_SIMULATION_FAILED", # case-insensitive + ], + ) + def test_known_permanent_reasons(self, reason: str) -> None: + assert _is_permanent_payment_error(reason) is True + + @pytest.mark.parametrize( + "reason", + [ + "", # empty + "503 Service Unavailable", + "Connection reset by peer", + "Read timeout after 60s", + "facilitator internal error", + "rate limit exceeded", + ], + ) + def test_transient_reasons_pass_through(self, reason: str) -> None: + assert _is_permanent_payment_error(reason) is False + + +class TestShouldFallbackSolana: + """``fallback_models`` decision matches Base semantics + permanent guard.""" + + def test_payment_error_never_falls_back(self) -> None: + # PaymentError always propagates — fallback would re-run on a new model + # but the payment is wallet-side, not provider-side. + exc = PaymentError( + "Payment rejected by gateway: transaction_simulation_failed", + status_code=402, + response={"details": "transaction_simulation_failed"}, + ) + assert _should_fallback_solana(exc) is False + + def test_timeout_falls_back_for_transient_reason(self) -> None: + exc = httpx.ReadTimeout("upstream took too long") + assert _should_fallback_solana(exc) is True + + def test_timeout_does_NOT_fall_back_when_reason_is_permanent(self) -> None: + """Defensive guard from issue #6: even a transient exception type + must not trigger fallback if the wrapped reason is a permanent + payment classification.""" + exc = httpx.ReadTimeout("transaction_simulation_failed during settle") + assert _should_fallback_solana(exc) is False + + def test_network_error_falls_back(self) -> None: + exc = httpx.NetworkError("connection reset") + assert _should_fallback_solana(exc) is True + + def test_network_error_with_permanent_reason_does_not(self) -> None: + exc = httpx.NetworkError("blockhash not found in cache") + assert _should_fallback_solana(exc) is False + + @pytest.mark.parametrize("status", [502, 503, 504, 522, 524]) + def test_5xx_api_error_falls_back(self, status: int) -> None: + exc = APIError("upstream sick", status_code=status, response=None) + assert _should_fallback_solana(exc) is True + + @pytest.mark.parametrize("status", [400, 401, 403, 404, 422]) + def test_4xx_api_error_does_not_fall_back(self, status: int) -> None: + exc = APIError("client error", status_code=status, response=None) + assert _should_fallback_solana(exc) is False + + +class TestPerMethodTimeoutConstants: + """v0.34.0 introduces per-use-case defaults; pin the values.""" + + def test_constants_have_workload_appropriate_values(self) -> None: + from blockrun_llm import solana_client as mod + + # Chat: long enough for streaming opus + 8k tokens + assert mod.DEFAULT_CHAT_TIMEOUT >= 120.0 + # Image: covers gpt-image-2 at 1536px (~180s server-side) + assert mod.DEFAULT_IMAGE_TIMEOUT >= 180.0 + # Search: Grok Live Search with deep web/X tool-use + assert mod.DEFAULT_SEARCH_TIMEOUT >= 180.0 + # Fast lookups: pyth / x_user_info return in ~1-2s + assert mod.DEFAULT_FAST_TIMEOUT <= 60.0 + # Backwards compatibility: flat DEFAULT_TIMEOUT must be ≥ chat + assert mod.DEFAULT_TIMEOUT >= mod.DEFAULT_CHAT_TIMEOUT + + def test_default_timeout_no_longer_60s(self) -> None: + """The historical 60s default truncated long chats and slow images. + v0.34.0 raises it to chat-grade so legacy callers stop dying at 60s.""" + from blockrun_llm import solana_client as mod + + assert mod.DEFAULT_TIMEOUT != 60.0 + assert mod.DEFAULT_TIMEOUT > 60.0 diff --git a/tests/unit/test_solana_safe_resign.py b/tests/unit/test_solana_safe_resign.py new file mode 100644 index 0000000..ee0fcba --- /dev/null +++ b/tests/unit/test_solana_safe_resign.py @@ -0,0 +1,445 @@ +"""Safety contract for Solana paid-leg re-sign retries. + +The line between safe and unsafe is the payment PHASE, not the cause: a +pre-broadcast rejection never settled and is free to re-sign, a settlement +failure may already have paid. These tests pin both directions, and they build +the errors the way production does — through +:func:`blockrun_llm.validation.build_payment_rejected_error` from the literal +402 bodies blockrun-sol emits — so a gateway wording change fails here rather +than silently disabling the retry. +""" + +from __future__ import annotations + +from typing import Any +from unittest.mock import AsyncMock, Mock + +import pytest + +from blockrun_llm.solana_client import ( + AsyncSolanaLLMClient, + SolanaLLMClient, + _is_safe_resign_error, + _normalize_reason, +) +from blockrun_llm.types import PaymentError +from blockrun_llm.validation import build_payment_rejected_error + + +class _FakeResponse: + """Minimal stand-in for the httpx.Response that build_* consumes.""" + + def __init__(self, body: Any) -> None: + self._body = body + + def json(self) -> Any: + return self._body + + +def gateway_error(body: dict[str, object]) -> PaymentError: + """Build a PaymentError exactly as the paid legs do, from a raw 402 body.""" + return build_payment_rejected_error(_FakeResponse(body)) + + +def payment_error(body: dict[str, object]) -> PaymentError: + return PaymentError("payment rejected", status_code=402, response=body) + + +# --- Literal gateway bodies -------------------------------------------------- +# +# Family A — /v1/chat/completions: carries `code` and a `message`. +# Family B — the ~16 other paid routes: `error` + `reason` only, NO code, +# NO message. These are the routes the raw POST/GET wrappers serve. + +CHAT_VERIFY_EXPIRED = { + "error": "Payment verification failed", + "message": "Message @bc1max on Telegram for help.", + "code": "PAYMENT_INVALID", + "reason": "expired_signature", +} +CHAT_VERIFY_UNAVAILABLE = { + "error": "Payment verification failed", + "message": "Message @bc1max on Telegram for help.", + "code": "PAYMENT_INVALID", + "reason": "verification_unavailable", +} +CHAT_VERIFY_CATCHALL = { + "error": "Payment verification failed", + "message": "Message @bc1max on Telegram for help.", + "code": "PAYMENT_INVALID", + "reason": "verification_failed", +} +CHAT_REPLAY = { + "error": "Payment authorization already used", + "message": "This payment signature was already redeemed. Sign a new payment for each request.", + "code": "PAYMENT_REPLAY", +} +CHAT_UNDERPAID = { + "error": "Payment below quoted price", + "message": ( + "The signed payment is less than the quoted price for this request. " + "Re-fetch the 402 quote and sign the amount it specifies." + ), + "code": "PAYMENT_UNDERPAID", +} +CHAT_SETTLE = { + "error": "Payment settlement failed", + "message": "Message @bc1max on Telegram for help.", + "code": "SETTLEMENT_FAILED", + "reason": "expired_signature", +} + +RAW_VERIFY_EXPIRED = {"error": "Payment verification failed", "reason": "expired_signature"} +RAW_VERIFY_UNAVAILABLE = { + "error": "Payment verification failed", + "reason": "verification_unavailable", +} +RAW_VERIFY_CATCHALL = {"error": "Payment verification failed", "reason": "verification_failed"} +RAW_VERIFY_NO_FUNDS = {"error": "Payment verification failed", "reason": "insufficient_funds"} +RAW_SETTLE_EXPIRED = {"error": "Payment settlement failed", "reason": "expired_signature"} +RAW_SETTLE_CATCHALL = {"error": "Payment settlement failed", "reason": "settlement_failed"} + +# The Anthropic-compatible route folds the phase and the reason into one string. +MESSAGES_VERIFY_EXPIRED = {"error": "Payment verification failed: expired_signature"} + + +@pytest.mark.parametrize( + "body", + [ + pytest.param(CHAT_VERIFY_EXPIRED, id="chat-expired-signature"), + pytest.param(CHAT_VERIFY_UNAVAILABLE, id="chat-verifier-outage"), + pytest.param(CHAT_VERIFY_CATCHALL, id="chat-verify-catchall"), + pytest.param(CHAT_REPLAY, id="chat-replay-nonce"), + pytest.param(CHAT_UNDERPAID, id="chat-underpaid"), + pytest.param(RAW_VERIFY_EXPIRED, id="raw-expired-signature"), + pytest.param(RAW_VERIFY_UNAVAILABLE, id="raw-verifier-outage"), + pytest.param(RAW_VERIFY_CATCHALL, id="raw-verify-catchall"), + pytest.param(MESSAGES_VERIFY_EXPIRED, id="messages-folded-reason"), + ], +) +def test_pre_broadcast_rejections_are_safe_to_resign(body: dict[str, object]) -> None: + """Nothing was broadcast, so a fresh signature costs the payer nothing.""" + assert _is_safe_resign_error(gateway_error(body)) is True + + +@pytest.mark.parametrize( + "body", + [ + pytest.param(CHAT_SETTLE, id="chat-settlement-failed"), + pytest.param(RAW_SETTLE_EXPIRED, id="raw-settle-expired-signature"), + pytest.param(RAW_SETTLE_CATCHALL, id="raw-settle-catchall"), + pytest.param(RAW_VERIFY_NO_FUNDS, id="raw-insufficient-funds"), + pytest.param({"message": "transaction_simulation_failed"}, id="bare-simulation-failure"), + pytest.param( + {"error": "Payment verification failed", "invalidMessage": "InvalidAccountData"}, + id="no-usdc-token-account", + ), + pytest.param({"error": "Some unrelated gateway error"}, id="unknown-title"), + ], +) +def test_settlement_and_terminal_failures_are_never_resigned(body: dict[str, object]) -> None: + assert _is_safe_resign_error(gateway_error(body)) is False + + +@pytest.mark.parametrize( + "body", + [ + pytest.param( + {"error": "Payment verification failed", "reason": "settlement_failed"}, + id="reason-contradicts-title", + ), + pytest.param( + {"error": "Payment verification failed", "code": "SETTLEMENT_FAILED"}, + id="code-contradicts-title", + ), + pytest.param({"reason": "settlement_failed"}, id="reason-only"), + ], +) +def test_any_settlement_marker_wins_over_a_verify_title(body: dict[str, object]) -> None: + """The phase gate is checked first and on all three fields, so a body whose + title says verify but whose code/reason says settlement stays terminal. A + settle rejection must never be re-signed on the strength of one field.""" + assert _is_safe_resign_error(gateway_error(body)) is False + + +def test_phase_titles_match_by_prefix_not_substring() -> None: + """`_normalize_reason` strips separators, so a substring test would straddle + word boundaries. A verify failure whose prose merely mentions settlement + must still be recognized as verify phase.""" + body = { + "error": "Payment verification failed", + "message": ( + "Payment verification failed: upstream reported that a prior " + "payment settlement failed and was retried" + ), + "code": "PAYMENT_INVALID", + } + normalized_message = _normalize_reason(str(body["message"])) + # The settlement title really is present mid-string — a substring test here + # would misread a verify-phase rejection as a broadcast and refuse to retry. + assert "paymentsettlementfailed" in normalized_message + assert not normalized_message.startswith("paymentsettlementfailed") + assert _is_safe_resign_error(payment_error(body)) is True + + +@pytest.mark.parametrize( + "exc", + [ + pytest.param(PaymentError("402 response but no payment requirements found"), id="no-body"), + pytest.param(PaymentError("x", status_code=402, response=None), id="none-response"), + pytest.param(PaymentError("x", status_code=402, response="not-a-dict"), id="str-response"), + ], +) +def test_missing_or_malformed_response_is_terminal(exc: PaymentError) -> None: + """Silence is never treated as permission to re-sign.""" + assert _is_safe_resign_error(exc) is False + + +# --- Retry wiring: all eight call sites -------------------------------------- + + +def _sync_client() -> SolanaLLMClient: + client = object.__new__(SolanaLLMClient) + client._PAYMENT_RETRY_BACKOFFS = (0.0, 0.0, 0.0, 0.0) # type: ignore[misc] + return client + + +def _async_client() -> AsyncSolanaLLMClient: + client = object.__new__(AsyncSolanaLLMClient) + client._PAYMENT_RETRY_BACKOFFS = (0.0, 0.0, 0.0, 0.0) # type: ignore[misc] + return client + + +SYNC_SITES = [ + ("_request_once", "_request_with_payment", ("/v1/chat/completions", {}), {}, "ok"), + ( + "_request_with_payment_raw_once", + "_request_with_payment_raw", + ("/v1/search", {}), + {}, + {"ok": True}, + ), + ("_get_with_payment_raw_once", "_get_with_payment_raw", ("/v1/pm/markets",), {}, {"ok": True}), +] + +ASYNC_SITES = [ + ("_request_once", "_request_with_payment", ("/v1/chat/completions", {}), {}, "ok"), + ( + "_request_with_payment_raw_once", + "_request_with_payment_raw", + ("/v1/search", {}), + {}, + {"ok": True}, + ), + ("_get_with_payment_raw_once", "_get_with_payment_raw", ("/v1/rpc/solana",), {}, {"ok": True}), +] + + +@pytest.mark.parametrize("once_name,wrapper_name,args,kwargs,result", SYNC_SITES) +def test_sync_sites_retry_a_pre_broadcast_rejection( + monkeypatch: pytest.MonkeyPatch, + once_name: str, + wrapper_name: str, + args: tuple, + kwargs: dict, + result: object, +) -> None: + client = _sync_client() + once = Mock(side_effect=[gateway_error(RAW_VERIFY_EXPIRED), result]) + monkeypatch.setattr(client, once_name, once) + assert getattr(client, wrapper_name)(*args, **kwargs) == result + assert once.call_count == 2 + + +@pytest.mark.parametrize("once_name,wrapper_name,args,kwargs,result", SYNC_SITES) +def test_sync_sites_never_replay_a_settlement_failure( + monkeypatch: pytest.MonkeyPatch, + once_name: str, + wrapper_name: str, + args: tuple, + kwargs: dict, + result: object, +) -> None: + client = _sync_client() + once = Mock(side_effect=gateway_error(RAW_SETTLE_EXPIRED)) + monkeypatch.setattr(client, once_name, once) + with pytest.raises(PaymentError): + getattr(client, wrapper_name)(*args, **kwargs) + assert once.call_count == 1 + + +@pytest.mark.parametrize("once_name,wrapper_name,args,kwargs,result", SYNC_SITES) +def test_sync_sites_bound_the_retry_and_raise_payment_error( + monkeypatch: pytest.MonkeyPatch, + once_name: str, + wrapper_name: str, + args: tuple, + kwargs: dict, + result: object, +) -> None: + """The loop must stop at _MAX_PAYMENT_RETRIES + 1 attempts and surface the + gateway's PaymentError, not the loop-exhausted guard.""" + client = _sync_client() + once = Mock(side_effect=[gateway_error(RAW_VERIFY_EXPIRED) for _ in range(20)]) + monkeypatch.setattr(client, once_name, once) + with pytest.raises(PaymentError, match="Payment rejected by gateway"): + getattr(client, wrapper_name)(*args, **kwargs) + assert once.call_count == SolanaLLMClient._MAX_PAYMENT_RETRIES + 1 + + +@pytest.mark.asyncio +@pytest.mark.parametrize("once_name,wrapper_name,args,kwargs,result", ASYNC_SITES) +async def test_async_sites_retry_a_pre_broadcast_rejection( + monkeypatch: pytest.MonkeyPatch, + once_name: str, + wrapper_name: str, + args: tuple, + kwargs: dict, + result: object, +) -> None: + client = _async_client() + once = AsyncMock(side_effect=[gateway_error(RAW_VERIFY_EXPIRED), result]) + monkeypatch.setattr(client, once_name, once) + assert await getattr(client, wrapper_name)(*args, **kwargs) == result + assert once.await_count == 2 + + +@pytest.mark.asyncio +@pytest.mark.parametrize("once_name,wrapper_name,args,kwargs,result", ASYNC_SITES) +async def test_async_sites_never_replay_a_settlement_failure( + monkeypatch: pytest.MonkeyPatch, + once_name: str, + wrapper_name: str, + args: tuple, + kwargs: dict, + result: object, +) -> None: + client = _async_client() + once = AsyncMock(side_effect=gateway_error(RAW_SETTLE_EXPIRED)) + monkeypatch.setattr(client, once_name, once) + with pytest.raises(PaymentError): + await getattr(client, wrapper_name)(*args, **kwargs) + assert once.await_count == 1 + + +@pytest.mark.asyncio +@pytest.mark.parametrize("once_name,wrapper_name,args,kwargs,result", ASYNC_SITES) +async def test_async_sites_bound_the_retry( + monkeypatch: pytest.MonkeyPatch, + once_name: str, + wrapper_name: str, + args: tuple, + kwargs: dict, + result: object, +) -> None: + client = _async_client() + once = AsyncMock(side_effect=[gateway_error(RAW_VERIFY_EXPIRED) for _ in range(20)]) + monkeypatch.setattr(client, once_name, once) + with pytest.raises(PaymentError, match="Payment rejected by gateway"): + await getattr(client, wrapper_name)(*args, **kwargs) + assert once.await_count == AsyncSolanaLLMClient._MAX_PAYMENT_RETRIES + 1 + + +# --- Streaming: output is never replayed ------------------------------------- + + +def test_sync_stream_does_not_resign_once_a_chunk_was_yielded() -> None: + """The paid leg already delivered output; re-signing would bill twice for + one answer even though the rejection itself is pre-broadcast.""" + client = _sync_client() + calls = {"n": 0} + + def once(endpoint: str, body: dict, timeout: float | None = None): + calls["n"] += 1 + yield {"chunk": calls["n"]} + raise gateway_error(RAW_VERIFY_EXPIRED) + + client._stream_once = once # type: ignore[assignment] + stream = client._stream_with_payment("/v1/chat/completions", {}) + assert next(stream) == {"chunk": 1} + with pytest.raises(PaymentError): + next(stream) + assert calls["n"] == 1 + + +def test_sync_stream_resigns_when_nothing_was_yielded() -> None: + client = _sync_client() + calls = {"n": 0} + + def once(endpoint: str, body: dict, timeout: float | None = None): + calls["n"] += 1 + if calls["n"] == 1: + raise gateway_error(RAW_VERIFY_EXPIRED) + yield {"chunk": "ok"} + + client._stream_once = once # type: ignore[assignment] + assert list(client._stream_with_payment("/v1/chat/completions", {})) == [{"chunk": "ok"}] + assert calls["n"] == 2 + + +def test_sync_stream_never_replays_a_settlement_failure() -> None: + client = _sync_client() + calls = {"n": 0} + + def once(endpoint: str, body: dict, timeout: float | None = None): + calls["n"] += 1 + raise gateway_error(RAW_SETTLE_EXPIRED) + yield # pragma: no cover - makes `once` a generator + + client._stream_once = once # type: ignore[assignment] + with pytest.raises(PaymentError): + list(client._stream_with_payment("/v1/chat/completions", {})) + assert calls["n"] == 1 + + +@pytest.mark.asyncio +async def test_async_stream_does_not_resign_once_a_chunk_was_yielded() -> None: + client = _async_client() + calls = {"n": 0} + + async def once(endpoint: str, body: dict, timeout: float | None = None): + calls["n"] += 1 + yield {"chunk": calls["n"]} + raise gateway_error(RAW_VERIFY_EXPIRED) + + client._stream_once = once # type: ignore[assignment] + stream = client._stream_with_payment("/v1/chat/completions", {}) + assert await stream.__anext__() == {"chunk": 1} + with pytest.raises(PaymentError): + await stream.__anext__() + assert calls["n"] == 1 + + +@pytest.mark.asyncio +async def test_async_stream_resigns_when_nothing_was_yielded() -> None: + client = _async_client() + calls = {"n": 0} + + async def once(endpoint: str, body: dict, timeout: float | None = None): + calls["n"] += 1 + if calls["n"] == 1: + raise gateway_error(RAW_VERIFY_EXPIRED) + yield {"chunk": "ok"} + + client._stream_once = once # type: ignore[assignment] + seen = [c async for c in client._stream_with_payment("/v1/chat/completions", {})] + assert seen == [{"chunk": "ok"}] + assert calls["n"] == 2 + + +# --- Backoff table ----------------------------------------------------------- + + +def test_real_backoff_table_is_used_in_order(monkeypatch: pytest.MonkeyPatch) -> None: + """Exercises the shipped tuple and the index clamp, which the zeroed + per-test override otherwise hides.""" + import time as _time + + client = object.__new__(SolanaLLMClient) + slept: list[float] = [] + monkeypatch.setattr(_time, "sleep", lambda s: slept.append(s)) + once = Mock(side_effect=[gateway_error(RAW_VERIFY_EXPIRED) for _ in range(20)]) + monkeypatch.setattr(client, "_request_once", once) + with pytest.raises(PaymentError): + client._request_with_payment("/v1/chat/completions", {}) + assert slept == list(SolanaLLMClient._PAYMENT_RETRY_BACKOFFS) diff --git a/tests/unit/test_solana_settled_payment.py b/tests/unit/test_solana_settled_payment.py new file mode 100644 index 0000000..51a8049 --- /dev/null +++ b/tests/unit/test_solana_settled_payment.py @@ -0,0 +1,65 @@ +"""Solana half of the settled-payment guard. + +Signing is settlement on either chain. The Base client learned not to let the +fallback chain buy a retry after a payment went out; this pins the same rule for +Solana, where the transfer is SPL USDC. + +Guarded with importorskip: the 3.9 CI job installs the SDK without the solana +extra, and an unguarded Solana test file turns that job red (see #19/#20). +""" + +from __future__ import annotations + +import httpx +import pytest + +pytest.importorskip("x402") +pytest.importorskip("solders") + +from blockrun_llm.client import _mark_settled +from blockrun_llm.solana_client import _should_fallback_solana +from blockrun_llm.types import APIError, PaymentError + + +class TestSolanaSettledTag: + """Mirrors TestSettledTagClassification in test_settled_payment.py.""" + + def test_untagged_timeout_still_falls_back(self): + assert _should_fallback_solana(httpx.ReadTimeout("boom")) is True + + def test_untagged_503_still_falls_back(self): + assert _should_fallback_solana(APIError("upstream", 503, None)) is True + + def test_settled_timeout_does_not_fall_back(self): + assert _should_fallback_solana(_mark_settled(httpx.ReadTimeout("boom"))) is False + + def test_settled_network_error_does_not_fall_back(self): + assert _should_fallback_solana(_mark_settled(httpx.ConnectError("boom"))) is False + + def test_settled_5xx_does_not_fall_back(self): + """The dominant post-settlement failure, and the one the Base fix + originally missed.""" + assert _should_fallback_solana(_mark_settled(APIError("upstream", 503, None))) is False + + def test_payment_error_still_refused(self): + """Pre-existing behavior must survive the new first check.""" + assert _should_fallback_solana(PaymentError("insufficient balance")) is False + + def test_permanent_payment_reason_still_refused(self): + """The issue #6 guard: a transient type carrying a permanent reason.""" + assert _should_fallback_solana(httpx.ReadTimeout("transaction_simulation_failed")) is False + + def test_both_chains_agree_on_the_tag(self): + """The tag has to mean the same thing in both fallback chains, or one + of them keeps paying twice.""" + from blockrun_llm.client import _should_fallback + + for exc in ( + httpx.ReadTimeout("boom"), + httpx.ConnectError("boom"), + APIError("upstream", 503, None), + ): + assert _should_fallback(exc) is True + assert _should_fallback_solana(exc) is True + assert _should_fallback(_mark_settled(exc)) is False + assert _should_fallback_solana(_mark_settled(exc)) is False diff --git a/tests/unit/test_solana_timeout_routing.py b/tests/unit/test_solana_timeout_routing.py new file mode 100644 index 0000000..eba6d78 --- /dev/null +++ b/tests/unit/test_solana_timeout_routing.py @@ -0,0 +1,255 @@ +"""Tests for per-use-case + per-call HTTP timeout routing on the Solana client. + +Covers the second half of issue #7. v0.34.0 defines ``DEFAULT_CHAT_TIMEOUT`` +(120s), ``DEFAULT_IMAGE_TIMEOUT`` (200s) and ``DEFAULT_SEARCH_TIMEOUT`` (300s), +but the constants are only useful if each method actually *applies* the right +one to its httpx request — and if a caller's per-call ``timeout=`` overrides it. + +Pre-fix every method flowed through the single ``httpx.Client(timeout=...)`` +default, so ``image()`` and ``search()`` silently used the chat budget. These +tests assert the real per-request timeout that reaches the transport via +``request.extensions["timeout"]`` so a future regression that drops the routing +fails loudly. + +Mocked at the httpx transport level (no network); the x402 signer and the +decode/encode helpers are stubbed exactly like ``test_streaming_solana``. +""" + +from __future__ import annotations + +from unittest import mock + +import httpx +import pytest + +pytest.importorskip("x402") +pytest.importorskip("solders") + +from blockrun_llm import SolanaLLMClient +from blockrun_llm import solana_client as sol + +# --------------------------------------------------------------------------- +# Fixtures / helpers +# --------------------------------------------------------------------------- + + +def _read_timeout(request: httpx.Request) -> float: + """The resolved read timeout that reached the transport for this request.""" + return request.extensions["timeout"]["read"] + + +@pytest.fixture(autouse=True) +def _no_disk_cache(monkeypatch: pytest.MonkeyPatch) -> None: + """Force a cache miss + no-op writes so every call hits the transport.""" + monkeypatch.setattr("blockrun_llm.cache.get_cached", lambda *a, **k: None) + monkeypatch.setattr("blockrun_llm.cache.save_to_cache", lambda *a, **k: None) + + +@pytest.fixture(autouse=True) +def _stub_x402_codec(monkeypatch: pytest.MonkeyPatch) -> None: + monkeypatch.setattr( + "blockrun_llm.solana_client.decode_payment_required_header", + lambda header: {"stub": True}, + ) + monkeypatch.setattr( + "blockrun_llm.solana_client.encode_payment_signature_header", + lambda payload: "stub-signature", + ) + + +def _make_client(transport: httpx.MockTransport, **kwargs: float) -> SolanaLLMClient: + with ( + mock.patch("blockrun_llm.solana_client.register_exact_svm_client"), + mock.patch("blockrun_llm.solana_client._create_signer"), + ): + client = SolanaLLMClient( + private_key="bogus_signer_is_patched", + api_url="https://sol.blockrun.ai/api", + rpc_url="http://test", + **kwargs, + ) + + class _FakePayload: + class accepted: + amount = "1000000" + pay_to = "GsbwXfJraMomNxBcpR3DBNxnKwZbyq7YCoDdSLDwzxdV" + + client._x402_client = mock.MagicMock() + client._x402_client.create_payment_payload.return_value = _FakePayload() + client._client = httpx.Client(transport=transport) + # Pre-seed the wallet address so billing metadata doesn't try to base58 + # decode the patched-out bogus key (system program address — valid b58). + client._address = "11111111111111111111111111111111" + return client + + +def _payment_required(request: httpx.Request) -> httpx.Response: + return httpx.Response( + 402, + headers={"content-type": "application/json", "payment-required": "stub-header"}, + json={"error": "Payment Required"}, + ) + + +_CHAT_OK = { + "id": "chatcmpl-1", + "created": 1700000000, + "model": "openai/gpt-5.2", + "choices": [ + {"index": 0, "message": {"role": "assistant", "content": "ok"}, "finish_reason": "stop"} + ], +} +_IMAGE_OK = {"created": 1700000000, "data": [{"url": "https://blockrun.ai/i.png"}]} +_SEARCH_OK = {"query": "q", "summary": "s"} + + +def _json_flow(calls: list[httpx.Request], ok_body: dict) -> httpx.MockTransport: + """402 on the unsigned probe, then the success body once signed.""" + + def handler(request: httpx.Request) -> httpx.Response: + calls.append(request) + if "PAYMENT-SIGNATURE" not in request.headers: + return _payment_required(request) + return httpx.Response(200, json=ok_body, headers={"content-type": "application/json"}) + + return httpx.MockTransport(handler) + + +# --------------------------------------------------------------------------- +# Per-use-case defaults +# --------------------------------------------------------------------------- + + +def test_chat_uses_chat_timeout_default() -> None: + calls: list[httpx.Request] = [] + client = _make_client(_json_flow(calls, _CHAT_OK)) + client.chat_completion("openai/gpt-5.2", [{"role": "user", "content": "hi"}]) + assert _read_timeout(calls[-1]) == sol.DEFAULT_CHAT_TIMEOUT + + +def test_image_uses_image_timeout_default() -> None: + calls: list[httpx.Request] = [] + client = _make_client(_json_flow(calls, _IMAGE_OK)) + client.image("a cat", model="openai/gpt-image-2") + # Probe + signed submit both carry the image budget, not the chat one. + assert _read_timeout(calls[0]) == sol.DEFAULT_IMAGE_TIMEOUT + assert _read_timeout(calls[-1]) == sol.DEFAULT_IMAGE_TIMEOUT + assert sol.DEFAULT_IMAGE_TIMEOUT != sol.DEFAULT_CHAT_TIMEOUT # regression: not chat + + +def test_search_uses_search_timeout_default() -> None: + calls: list[httpx.Request] = [] + client = _make_client(_json_flow(calls, _SEARCH_OK)) + client.search("deep query") + assert _read_timeout(calls[-1]) == sol.DEFAULT_SEARCH_TIMEOUT + + +def test_exa_uses_search_timeout_default() -> None: + calls: list[httpx.Request] = [] + client = _make_client(_json_flow(calls, {"results": []})) + client.exa_search("latest AI papers") + assert _read_timeout(calls[-1]) == sol.DEFAULT_SEARCH_TIMEOUT + + +# --------------------------------------------------------------------------- +# Per-call override wins over every default +# --------------------------------------------------------------------------- + + +def test_chat_per_call_override() -> None: + calls: list[httpx.Request] = [] + client = _make_client(_json_flow(calls, _CHAT_OK)) + client.chat_completion("openai/gpt-5.2", [{"role": "user", "content": "hi"}], timeout=7.0) + assert _read_timeout(calls[0]) == 7.0 + assert _read_timeout(calls[-1]) == 7.0 + + +def test_image_per_call_override() -> None: + calls: list[httpx.Request] = [] + client = _make_client(_json_flow(calls, _IMAGE_OK)) + client.image("a cat", model="openai/gpt-image-2", timeout=9.0) + assert _read_timeout(calls[-1]) == 9.0 + + +def test_search_per_call_override() -> None: + calls: list[httpx.Request] = [] + client = _make_client(_json_flow(calls, _SEARCH_OK)) + client.search("q", timeout=11.0) + assert _read_timeout(calls[-1]) == 11.0 + + +# --------------------------------------------------------------------------- +# Constructor-level overrides +# --------------------------------------------------------------------------- + + +def test_constructor_image_and_search_timeout_respected() -> None: + img_calls: list[httpx.Request] = [] + img_client = _make_client(_json_flow(img_calls, _IMAGE_OK), image_timeout=42.0) + img_client.image("a cat", model="openai/gpt-image-2") + assert _read_timeout(img_calls[-1]) == 42.0 + + s_calls: list[httpx.Request] = [] + s_client = _make_client(_json_flow(s_calls, _SEARCH_OK), search_timeout=99.0) + s_client.search("q") + assert _read_timeout(s_calls[-1]) == 99.0 + + +def test_legacy_flat_timeout_still_governs_chat() -> None: + """Backwards-compat: old ``SolanaLLMClient(timeout=...)`` callers keep + controlling the chat budget through the single keyword.""" + calls: list[httpx.Request] = [] + client = _make_client(_json_flow(calls, _CHAT_OK), timeout=33.0) + client.chat_completion("openai/gpt-5.2", [{"role": "user", "content": "hi"}]) + assert _read_timeout(calls[-1]) == 33.0 + + +# --------------------------------------------------------------------------- +# Async mirror — the threading is symmetric, so cover chat both ways. +# --------------------------------------------------------------------------- + + +def _make_async_client(transport: httpx.MockTransport, **kwargs: float): + from blockrun_llm.solana_client import AsyncSolanaLLMClient + + with ( + mock.patch("blockrun_llm.solana_client.register_exact_svm_client"), + mock.patch("blockrun_llm.solana_client._create_signer"), + mock.patch("x402.x402Client"), + ): + client = AsyncSolanaLLMClient( + private_key="bogus_signer_is_patched", + api_url="https://sol.blockrun.ai/api", + rpc_url="http://test", + **kwargs, + ) + + class _FakePayload: + class accepted: + amount = "1000000" + pay_to = "GsbwXfJraMomNxBcpR3DBNxnKwZbyq7YCoDdSLDwzxdV" + + client._x402_client = mock.MagicMock() + client._x402_client.create_payment_payload = mock.AsyncMock(return_value=_FakePayload()) + client._client = httpx.AsyncClient(transport=transport) + client._address = "11111111111111111111111111111111" + return client + + +async def test_async_chat_uses_chat_timeout_default() -> None: + calls: list[httpx.Request] = [] + client = _make_async_client(_json_flow(calls, _CHAT_OK)) + await client.chat_completion("openai/gpt-5.2", [{"role": "user", "content": "hi"}]) + assert _read_timeout(calls[-1]) == sol.DEFAULT_CHAT_TIMEOUT + await client.close() + + +async def test_async_chat_per_call_override() -> None: + calls: list[httpx.Request] = [] + client = _make_async_client(_json_flow(calls, _CHAT_OK)) + await client.chat_completion( + "openai/gpt-5.2", [{"role": "user", "content": "hi"}], timeout=13.0 + ) + assert _read_timeout(calls[0]) == 13.0 + assert _read_timeout(calls[-1]) == 13.0 + await client.close() diff --git a/tests/unit/test_solana_wallet.py b/tests/unit/test_solana_wallet.py new file mode 100644 index 0000000..6bc8929 --- /dev/null +++ b/tests/unit/test_solana_wallet.py @@ -0,0 +1,108 @@ +"""Unit tests for Solana wallet utilities.""" + +import pytest + +from blockrun_llm.solana_wallet import ( + create_solana_wallet, + get_solana_public_key, + solana_key_to_bytes, +) + +# A valid test bs58 secret key (64 bytes, valid keypair from deterministic seed) +TEST_BS58_KEY = ( + "433C7KFcM4y1ZEVdZYSH7wheSNAM384UcbgXEyD5FV7Q2HsQ1BwjEDx4GbBZUqPkZTVhFPyLyuZnzK8wCeAkU7wG" +) + + +class TestCreateSolanaWallet: + def test_returns_address_and_key(self): + wallet = create_solana_wallet() + assert "address" in wallet + assert "private_key" in wallet + assert len(wallet["address"]) >= 32 # base58 pubkey + assert len(wallet["private_key"]) >= 86 # bs58 64-byte key + + def test_unique_wallets(self): + w1 = create_solana_wallet() + w2 = create_solana_wallet() + assert w1["address"] != w2["address"] + assert w1["private_key"] != w2["private_key"] + + +class TestSolanaKeyToBytes: + def test_valid_key(self): + b = solana_key_to_bytes(TEST_BS58_KEY) + assert isinstance(b, bytes) + assert len(b) == 64 + + def test_invalid_key_raises(self): + with pytest.raises(ValueError, match="Invalid Solana private key"): + solana_key_to_bytes("not-a-valid-key!!!") + + def test_accepts_solana_cli_json_array(self): + import json + + canonical = solana_key_to_bytes(TEST_BS58_KEY) + as_json = json.dumps(list(canonical)) + assert solana_key_to_bytes(as_json) == canonical + + def test_accepts_64_byte_hex_with_or_without_0x(self): + canonical = solana_key_to_bytes(TEST_BS58_KEY) + hex_key = canonical.hex() + assert solana_key_to_bytes(hex_key) == canonical + assert solana_key_to_bytes("0x" + hex_key) == canonical + + def test_evm_key_gets_explicit_hint(self): + evm_key = "0x" + "ab" * 32 # 32-byte hex — Base/EVM format + with pytest.raises(ValueError, match="EVM"): + solana_key_to_bytes(evm_key) + + def test_unrecognized_input_lists_accepted_formats(self): + with pytest.raises(ValueError, match="base58"): + solana_key_to_bytes("not/a/key!!") + + +class TestGetOrCreateSolanaWalletErrorSources: + def test_bad_env_key_names_the_env_var(self, monkeypatch, tmp_path): + from blockrun_llm import solana_wallet + + monkeypatch.setattr(solana_wallet, "SOLANA_WALLET_FILE", tmp_path / ".solana-session") + monkeypatch.setenv("SOLANA_WALLET_KEY", "not/a/key!!") + with pytest.raises(ValueError, match="SOLANA_WALLET_KEY"): + solana_wallet.get_or_create_solana_wallet() + + def test_bad_session_file_names_the_path(self, monkeypatch, tmp_path): + from blockrun_llm import solana_wallet + + session = tmp_path / ".solana-session" + session.write_text("not/a/key!!") + monkeypatch.setattr(solana_wallet, "SOLANA_WALLET_FILE", session) + monkeypatch.delenv("SOLANA_WALLET_KEY", raising=False) + with pytest.raises(ValueError, match="solana-session"): + solana_wallet.get_or_create_solana_wallet() + + def test_adopts_session_file_in_json_array_format(self, monkeypatch, tmp_path): + import json + + from blockrun_llm import solana_wallet + + wallet = create_solana_wallet() + canonical = solana_key_to_bytes(wallet["private_key"]) + session = tmp_path / ".solana-session" + session.write_text(json.dumps(list(canonical))) + monkeypatch.setattr(solana_wallet, "SOLANA_WALLET_FILE", session) + monkeypatch.delenv("SOLANA_WALLET_KEY", raising=False) + result = solana_wallet.get_or_create_solana_wallet() + assert result["address"] == wallet["address"] + assert result["is_new"] is False + + +class TestGetSolanaPublicKey: + def test_returns_base58_address(self): + addr = get_solana_public_key(TEST_BS58_KEY) + assert isinstance(addr, str) + assert len(addr) >= 32 + # Should be valid base58 (only alphanumeric, no 0/O/I/l) + import re + + assert re.match(r"^[1-9A-HJ-NP-Za-km-z]+$", addr) diff --git a/tests/unit/test_speech.py b/tests/unit/test_speech.py new file mode 100644 index 0000000..3042b47 --- /dev/null +++ b/tests/unit/test_speech.py @@ -0,0 +1,95 @@ +"""Unit tests for SpeechClient request construction and response parsing.""" + +import os + +import pytest + +from blockrun_llm import SpeechClient, SpeechResponse + + +@pytest.fixture +def client(): + # Deterministic dummy key — never actually signs against a live endpoint + # in unit tests; we only exercise local request/response paths. + os.environ.setdefault("BLOCKRUN_WALLET_KEY", "0x" + "11" * 32) + return SpeechClient() + + +def test_generate_builds_speech_body(client, monkeypatch): + captured = {} + + def fake_request(endpoint, body): + captured["endpoint"] = endpoint + captured["body"] = body + return SpeechResponse(created=1, model=body["model"], data=[]) + + monkeypatch.setattr(client, "_request_with_payment", fake_request) + + client.generate("Hello", voice="george", response_format="wav", speed=1.1) + + assert captured["endpoint"] == "/v1/audio/speech" + assert captured["body"] == { + "model": "elevenlabs/flash-v2.5", + "input": "Hello", + "voice": "george", + "response_format": "wav", + "speed": 1.1, + } + + +def test_generate_omits_optional_fields(client, monkeypatch): + captured = {} + + def fake_request(endpoint, body): + captured["body"] = body + return SpeechResponse(created=1, model=body["model"], data=[]) + + monkeypatch.setattr(client, "_request_with_payment", fake_request) + + client.generate("Hi", model="elevenlabs/v3") + + assert captured["body"] == {"model": "elevenlabs/v3", "input": "Hi"} + + +def test_speak_is_generate_alias(client): + assert SpeechClient.speak is SpeechClient.generate + + +def test_sound_effect_builds_body(client, monkeypatch): + captured = {} + + def fake_request(endpoint, body): + captured["endpoint"] = endpoint + captured["body"] = body + return SpeechResponse(created=1, model=body["model"], data=[]) + + monkeypatch.setattr(client, "_request_with_payment", fake_request) + + client.sound_effect("rain on a tin roof", duration_seconds=5, prompt_influence=0.7) + + assert captured["endpoint"] == "/v1/audio/sound-effects" + assert captured["body"] == { + "model": "elevenlabs/sound-effects", + "text": "rain on a tin roof", + "duration_seconds": 5, + "prompt_influence": 0.7, + } + + +def test_speech_response_parses_payload(): + resp = SpeechResponse( + created=1749000000, + model="elevenlabs/flash-v2.5", + data=[{"url": "https://cdn.example/a.mp3", "format": "mp3", "characters": 42}], + txHash="0xabc", + ) + assert resp.data[0].url == "https://cdn.example/a.mp3" + assert resp.data[0].characters == 42 + assert resp.data[0].credits is None + assert resp.txHash == "0xabc" + + +def test_get_wallet_address(client): + addr = client.get_wallet_address() + assert addr.startswith("0x") + assert len(addr) == 42 diff --git a/tests/unit/test_spend_limits.py b/tests/unit/test_spend_limits.py new file mode 100644 index 0000000..a1d948c --- /dev/null +++ b/tests/unit/test_spend_limits.py @@ -0,0 +1,203 @@ +"""Client-side spend limits. + +Before 1.9.0 there was no ceiling anywhere: `client.py` computed `cost_usd` and +signed the quote in the next statement, with nothing compared against anything. +`chat_completion` even documented a `PaymentError: If budget is set and would be +exceeded` for a `budget` parameter that did not exist. + +The rule these tests encode: when a limit refuses a quote, **no paid request is +ever sent**. Signing alone moves no money — the gateway submitting the signed +authorization does — so a refusal before the send costs the caller nothing. +""" + +import httpx +import pytest + +from blockrun_llm import LLMClient +from blockrun_llm.types import PaymentError, SpendLimitError +from blockrun_llm.validation import check_spend_limits, resolve_spend_limit + +from ..helpers import ( + TEST_PRIVATE_KEY, + build_chat_response, + build_payment_required_response, +) + +MESSAGES = [{"role": "user", "content": "hi"}] + +# build_payment_required_response defaults to amount "1000000" = 1 USDC. +QUOTED_USD = 1.0 + + +def _client(**kwargs): + """A client whose paid leg fails the test if it is ever reached.""" + signed = [] + + def handler(request: httpx.Request) -> httpx.Response: + if "PAYMENT-SIGNATURE" in request.headers: + signed.append(request) + return httpx.Response(200, json=build_chat_response()) + return httpx.Response( + 402, + json={"error": "Payment Required"}, + headers={"payment-required": build_payment_required_response()}, + ) + + client = LLMClient(private_key=TEST_PRIVATE_KEY, **kwargs) + client._client = httpx.Client(transport=httpx.MockTransport(handler)) + return client, signed + + +class TestNoLimitsIsUnchanged: + def test_default_client_still_pays(self): + """Limits are opt-in. Omitting them must behave exactly as before.""" + client, signed = _client() + client.chat_completion("a/b", MESSAGES) + assert len(signed) == 1 + + def test_helper_is_a_noop_without_limits(self): + check_spend_limits( + 999.0, max_cost_per_call=None, max_session_cost=None, session_spent_usd=0.0 + ) + + +class TestPerCallLimit: + def test_refuses_over_limit_quote_without_sending(self): + client, signed = _client(max_cost_per_call=0.10) + with pytest.raises(SpendLimitError) as exc: + client.chat_completion("a/b", MESSAGES) + assert signed == [], "a refused quote must never be sent" + assert exc.value.scope == "call" + assert exc.value.quoted_usd == pytest.approx(QUOTED_USD) + assert exc.value.limit_usd == pytest.approx(0.10) + + def test_allows_quote_at_or_under_limit(self): + client, signed = _client(max_cost_per_call=QUOTED_USD) + client.chat_completion("a/b", MESSAGES) + assert len(signed) == 1, "the limit is inclusive" + + def test_message_names_both_numbers_and_the_model(self): + client, _ = _client(max_cost_per_call=0.10) + with pytest.raises(SpendLimitError) as exc: + client.chat_completion("anthropic/claude-opus-4.8", MESSAGES) + msg = str(exc.value) + assert "1.000000" in msg and "0.100000" in msg + assert "claude-opus-4.8" in msg + assert "nothing was charged" in msg.lower() + + def test_nothing_is_recorded_as_spent(self): + client, _ = _client(max_cost_per_call=0.10) + with pytest.raises(SpendLimitError): + client.chat_completion("a/b", MESSAGES) + assert client.get_spending()["total_usd"] == 0.0 + assert client.get_spending()["calls"] == 0 + + +class TestSessionLimit: + def test_refuses_the_call_that_would_breach_the_total(self): + client, signed = _client(max_session_cost=1.5) + client.chat_completion("a/b", MESSAGES) # 1.0 spent, 0.5 left + assert len(signed) == 1 + with pytest.raises(SpendLimitError) as exc: + client.chat_completion("a/b", MESSAGES) # would reach 2.0 + assert len(signed) == 1, "the second quote must not be sent" + assert exc.value.scope == "session" + + def test_message_reports_what_is_left(self): + client, _ = _client(max_session_cost=1.5) + client.chat_completion("a/b", MESSAGES) + with pytest.raises(SpendLimitError) as exc: + client.chat_completion("a/b", MESSAGES) + assert "0.500000" in str(exc.value) + + +class TestAsyncClient: + """The async handler computes its cost_usd only after the paid POST returns, + so the guard has to read the quote off the 402 instead. Without a test the + limit silently ran too late to refuse anything.""" + + def _run(self, **kwargs): + from blockrun_llm import AsyncLLMClient + + signed = [] + + def handler(request: httpx.Request) -> httpx.Response: + if "PAYMENT-SIGNATURE" in request.headers: + signed.append(request) + return httpx.Response(200, json=build_chat_response()) + return httpx.Response( + 402, + json={"error": "Payment Required"}, + headers={"payment-required": build_payment_required_response()}, + ) + + async def go(): + client = AsyncLLMClient(private_key=TEST_PRIVATE_KEY, **kwargs) + client._client = httpx.AsyncClient(transport=httpx.MockTransport(handler)) + return await client.chat_completion("a/b", MESSAGES) + + return go, signed + + def test_refuses_over_limit_without_sending(self): + import asyncio + + go, signed = self._run(max_cost_per_call=0.10) + with pytest.raises(SpendLimitError): + asyncio.run(go()) + assert signed == [], "the async path must refuse before the paid POST" + + def test_allows_quote_under_limit(self): + import asyncio + + go, signed = self._run(max_cost_per_call=QUOTED_USD) + asyncio.run(go()) + assert len(signed) == 1 + + +class TestErrorContract: + def test_is_a_payment_error(self): + """Existing `except PaymentError` handlers must keep working.""" + assert issubclass(SpendLimitError, PaymentError) + + def test_does_not_trigger_model_fallback(self): + """Falling back to another model after refusing on cost would defeat + the limit, and would sign a second quote.""" + from blockrun_llm.client import _should_fallback + + exc = SpendLimitError("x", quoted_usd=1.0, limit_usd=0.1, scope="call") + assert _should_fallback(exc) is False + + def test_fallback_chain_refuses_rather_than_shopping_for_a_cheaper_model(self): + client, signed = _client(max_cost_per_call=0.10) + with pytest.raises(SpendLimitError): + client.chat_completion("a/b", MESSAGES, fallback_models=["c/d", "e/f"]) + assert signed == [], "must not try the next model looking for a cheaper quote" + + +class TestLimitResolution: + def test_explicit_argument_wins_over_env(self, monkeypatch): + monkeypatch.setenv("BLOCKRUN_MAX_COST_PER_CALL", "5.0") + assert resolve_spend_limit(0.25, "BLOCKRUN_MAX_COST_PER_CALL") == 0.25 + + def test_env_var_applies_when_no_argument(self, monkeypatch): + monkeypatch.setenv("BLOCKRUN_MAX_COST_PER_CALL", "0.25") + assert resolve_spend_limit(None, "BLOCKRUN_MAX_COST_PER_CALL") == 0.25 + + def test_env_var_reaches_a_real_client(self, monkeypatch): + monkeypatch.setenv("BLOCKRUN_MAX_COST_PER_CALL", "0.10") + client, signed = _client() + with pytest.raises(SpendLimitError): + client.chat_completion("a/b", MESSAGES) + assert signed == [] + + def test_malformed_env_is_ignored_not_fatal(self, monkeypatch): + """A bad env var must not brick every client in a deployment.""" + for bad in ("abc", "", "-1", "0"): + monkeypatch.setenv("BLOCKRUN_MAX_COST_PER_CALL", bad) + assert resolve_spend_limit(None, "BLOCKRUN_MAX_COST_PER_CALL") is None + + def test_explicit_non_positive_is_a_programming_error(self): + with pytest.raises(ValueError, match="positive"): + resolve_spend_limit(0, "BLOCKRUN_MAX_COST_PER_CALL") + with pytest.raises(ValueError, match="positive"): + resolve_spend_limit(-1.0, "BLOCKRUN_MAX_COST_PER_CALL") diff --git a/tests/unit/test_streaming.py b/tests/unit/test_streaming.py new file mode 100644 index 0000000..572f7f6 --- /dev/null +++ b/tests/unit/test_streaming.py @@ -0,0 +1,790 @@ +""" +Unit tests for chat_completion_stream (sync + async). + +We use httpx.MockTransport so no real network call ever happens. Two +scenarios are covered for each variant: + +1. **Free-model path** — first POST returns 200 + ``text/event-stream``; + we should iterate chunks without ever invoking the x402 signer. + +2. **Paid-model path** — first POST returns 402 with payment requirements; + we verify the signer fires, the retry sends a ``PAYMENT-SIGNATURE`` + header, and chunks come back on the second response. + +Plus a couple of robustness tests: ``[DONE]`` terminator, malformed +chunks skipped, finish_reason on the final chunk. +""" + +from __future__ import annotations + +import json + +import httpx +import pytest + +from blockrun_llm import AsyncLLMClient, ChatCompletionChunk, LLMClient +from blockrun_llm.types import PaymentError + +from ..helpers import TEST_PRIVATE_KEY, build_payment_required_response + +# --------------------------------------------------------------------------- +# Synthetic SSE bodies +# --------------------------------------------------------------------------- + + +def _sse_events(deltas: list[str], finish: str = "stop", model: str = "test/model") -> bytes: + """Render a list of content deltas as raw SSE bytes ending with [DONE].""" + lines: list[str] = [] + # First chunk — role only. + lines.append( + "data: " + + json.dumps( + { + "id": "chatcmpl-test", + "object": "chat.completion.chunk", + "created": 1700000000, + "model": model, + "choices": [{"index": 0, "delta": {"role": "assistant"}, "finish_reason": None}], + } + ) + ) + # Content chunks. + for i, d in enumerate(deltas): + lines.append( + "data: " + + json.dumps( + { + "id": "chatcmpl-test", + "object": "chat.completion.chunk", + "created": 1700000000, + "model": model, + "choices": [{"index": 0, "delta": {"content": d}, "finish_reason": None}], + } + ) + ) + # Final chunk with finish_reason. + lines.append( + "data: " + + json.dumps( + { + "id": "chatcmpl-test", + "object": "chat.completion.chunk", + "created": 1700000000, + "model": model, + "choices": [{"index": 0, "delta": {}, "finish_reason": finish}], + } + ) + ) + lines.append("data: [DONE]") + body = "\n\n".join(lines) + "\n\n" + return body.encode("utf-8") + + +def _sse_with_garbage(deltas: list[str]) -> bytes: + """Same as ``_sse_events`` but with a couple of malformed lines mixed in + to verify the parser is tolerant.""" + base = _sse_events(deltas).decode("utf-8") + # Insert a malformed chunk after the first content event. + parts = base.split("\n\n") + parts.insert(2, "data: {this is not valid json}") + parts.insert(3, ": this is an SSE comment heartbeat") + return ("\n\n".join(parts)).encode("utf-8") + + +# --------------------------------------------------------------------------- +# Mock transports +# --------------------------------------------------------------------------- + + +def _make_free_model_transport(sse_body: bytes, calls: list[httpx.Request]) -> httpx.MockTransport: + def handler(request: httpx.Request) -> httpx.Response: + calls.append(request) + return httpx.Response( + 200, + headers={"content-type": "text/event-stream"}, + content=sse_body, + ) + + return httpx.MockTransport(handler) + + +def _make_paid_model_transport(sse_body: bytes, calls: list[httpx.Request]) -> httpx.MockTransport: + """First call → 402 with valid payment-required header; second → 200 SSE.""" + + def handler(request: httpx.Request) -> httpx.Response: + calls.append(request) + if "PAYMENT-SIGNATURE" not in request.headers: + return httpx.Response( + 402, + headers={ + "content-type": "application/json", + "payment-required": build_payment_required_response(), + }, + json={"error": "Payment Required", "price": {"amount": "0.001"}}, + ) + return httpx.Response( + 200, + headers={"content-type": "text/event-stream"}, + content=sse_body, + ) + + return httpx.MockTransport(handler) + + +# --------------------------------------------------------------------------- +# Sync tests +# --------------------------------------------------------------------------- + + +class TestSyncStreaming: + def test_free_model_streams_without_payment(self): + calls: list[httpx.Request] = [] + client = LLMClient(private_key=TEST_PRIVATE_KEY) + client._client = httpx.Client( + transport=_make_free_model_transport(_sse_events(["Hello", " world"]), calls) + ) + + chunks: list[ChatCompletionChunk] = list( + client.chat_completion_stream( + "nvidia/deepseek-v4-flash", + [{"role": "user", "content": "hi"}], + max_tokens=32, + ) + ) + + # Free path = one HTTP request total. No PAYMENT-SIGNATURE seen. + assert len(calls) == 1 + assert "PAYMENT-SIGNATURE" not in calls[0].headers + + # First chunk carries role; subsequent carry content; last carries finish. + roles = [c.choices[0].delta.role for c in chunks] + contents = [c.choices[0].delta.content for c in chunks if c.choices[0].delta.content] + finishes = [c.choices[0].finish_reason for c in chunks if c.choices[0].finish_reason] + + assert roles[0] == "assistant" + assert "".join(contents) == "Hello world" + assert finishes == ["stop"] + + def test_paid_model_signs_and_retries(self): + calls: list[httpx.Request] = [] + client = LLMClient(private_key=TEST_PRIVATE_KEY) + client._client = httpx.Client( + transport=_make_paid_model_transport(_sse_events(["Paid"]), calls) + ) + + chunks = list( + client.chat_completion_stream( + "openai/gpt-5.5", + [{"role": "user", "content": "hi"}], + max_tokens=16, + ) + ) + + # 402 dance = exactly two HTTP requests. + assert len(calls) == 2 + assert "PAYMENT-SIGNATURE" not in calls[0].headers + assert "PAYMENT-SIGNATURE" in calls[1].headers + # Session cost was tracked. + assert client._session_calls == 1 + assert client._session_total_usd > 0 + # Streamed content arrives. + assert ( + "".join(c.choices[0].delta.content for c in chunks if c.choices[0].delta.content) + == "Paid" + ) + + def test_paid_stream_chunks_carry_real_cost(self): + """Every paid-path chunk carries the real per-call x402 charge as + ``chunk.cost_usd`` (race-free, vs the shared ``_last_call_cost``), so a + streaming consumer can report the actual wallet deduction.""" + calls: list[httpx.Request] = [] + client = LLMClient(private_key=TEST_PRIVATE_KEY) + client._client = httpx.Client( + transport=_make_paid_model_transport(_sse_events(["Paid"]), calls) + ) + + chunks = list( + client.chat_completion_stream( + "openai/gpt-5.5", + [{"role": "user", "content": "hi"}], + max_tokens=16, + ) + ) + + charge = client._last_call_cost + assert charge > 0 + assert chunks, "expected at least one chunk" + assert all(getattr(c, "cost_usd", None) == charge for c in chunks) + + def test_free_stream_chunks_have_no_cost(self): + """Free models skip the 402/sign path (and the archive), so chunks + carry no ``cost_usd`` — consumers treat that as 'no real charge'.""" + calls: list[httpx.Request] = [] + client = LLMClient(private_key=TEST_PRIVATE_KEY) + client._client = httpx.Client( + transport=_make_free_model_transport(_sse_events(["hi"]), calls) + ) + + chunks = list( + client.chat_completion_stream( + "nvidia/deepseek-v4-flash", + [{"role": "user", "content": "hi"}], + ) + ) + assert all(getattr(c, "cost_usd", None) is None for c in chunks) + + def test_malformed_chunks_dont_abort_stream(self): + calls: list[httpx.Request] = [] + client = LLMClient(private_key=TEST_PRIVATE_KEY) + client._client = httpx.Client( + transport=_make_free_model_transport(_sse_with_garbage(["A", "B"]), calls) + ) + + chunks = list( + client.chat_completion_stream( + "nvidia/deepseek-v4-flash", + [{"role": "user", "content": "hi"}], + ) + ) + # We should have gotten both deltas through, despite the garbage chunk. + joined = "".join(c.choices[0].delta.content for c in chunks if c.choices[0].delta.content) + assert joined == "AB" + + def test_paid_path_propagates_payment_rejected(self): + """If the retry also returns 402, surface PaymentError.""" + calls: list[httpx.Request] = [] + + def handler(request: httpx.Request) -> httpx.Response: + calls.append(request) + return httpx.Response( + 402, + headers={ + "content-type": "application/json", + "payment-required": build_payment_required_response(), + }, + json={"error": "Payment Required", "price": {"amount": "0.001"}}, + ) + + client = LLMClient(private_key=TEST_PRIVATE_KEY) + client._client = httpx.Client(transport=httpx.MockTransport(handler)) + + with pytest.raises(PaymentError): + list( + client.chat_completion_stream( + "openai/gpt-5.5", + [{"role": "user", "content": "hi"}], + ) + ) + # Probe + retry both got 402. + assert len(calls) == 2 + + +# --------------------------------------------------------------------------- +# Async tests +# --------------------------------------------------------------------------- + + +class TestAsyncStreaming: + @pytest.mark.asyncio + async def test_async_free_model(self): + calls: list[httpx.Request] = [] + client = AsyncLLMClient(private_key=TEST_PRIVATE_KEY) + # Swap in mock transport (same pattern as sync). + await client._client.aclose() + client._client = httpx.AsyncClient( + transport=_make_free_model_transport(_sse_events(["Hi", "!"]), calls) + ) + + chunks: list[ChatCompletionChunk] = [] + async for chunk in client.chat_completion_stream( + "nvidia/deepseek-v4-flash", + [{"role": "user", "content": "hi"}], + ): + chunks.append(chunk) + + assert len(calls) == 1 + assert ( + "".join(c.choices[0].delta.content for c in chunks if c.choices[0].delta.content) + == "Hi!" + ) + await client.close() + + @pytest.mark.asyncio + async def test_async_paid_model_signs_and_retries(self): + calls: list[httpx.Request] = [] + client = AsyncLLMClient(private_key=TEST_PRIVATE_KEY) + await client._client.aclose() + client._client = httpx.AsyncClient( + transport=_make_paid_model_transport(_sse_events(["X"]), calls) + ) + + chunks: list[ChatCompletionChunk] = [] + async for chunk in client.chat_completion_stream( + "openai/gpt-5.5", + [{"role": "user", "content": "hi"}], + ): + chunks.append(chunk) + + assert len(calls) == 2 + assert "PAYMENT-SIGNATURE" not in calls[0].headers + assert "PAYMENT-SIGNATURE" in calls[1].headers + await client.close() + + +# --------------------------------------------------------------------------- +# 5xx retry tests +# --------------------------------------------------------------------------- + + +def _make_flaky_free_transport( + sse_body: bytes, + fail_count: int, + calls: list[httpx.Request], + status: int = 503, +) -> httpx.MockTransport: + """Returns ``status`` (default 503) for the first ``fail_count`` requests, + then 200 + SSE on the next one. Used to verify retry-with-backoff logic.""" + + def handler(request: httpx.Request) -> httpx.Response: + calls.append(request) + if len(calls) <= fail_count: + return httpx.Response( + status, headers={"content-type": "application/json"}, json={"error": "transient"} + ) + return httpx.Response(200, headers={"content-type": "text/event-stream"}, content=sse_body) + + return httpx.MockTransport(handler) + + +class TestStreamingRetries: + """LLMClient._STREAM_5XX_BACKOFFS controls the retry policy. With three + backoffs the SDK tries up to 4 times per phase before raising.""" + + def test_recovers_after_two_503s(self, monkeypatch): + # Zero out sleeps to keep tests fast. + monkeypatch.setattr("time.sleep", lambda _s: None) + + calls: list[httpx.Request] = [] + client = LLMClient(private_key=TEST_PRIVATE_KEY) + client._client = httpx.Client( + transport=_make_flaky_free_transport(_sse_events(["OK"]), fail_count=2, calls=calls) + ) + chunks = list( + client.chat_completion_stream( + "nvidia/deepseek-v4-flash", + [{"role": "user", "content": "hi"}], + ) + ) + # 2 failed + 1 success + assert len(calls) == 3 + assert any(c.choices[0].delta.content == "OK" for c in chunks) + + def test_raises_after_exhausting_retries(self, monkeypatch): + monkeypatch.setattr("time.sleep", lambda _s: None) + + calls: list[httpx.Request] = [] + + def handler(request: httpx.Request) -> httpx.Response: + calls.append(request) + return httpx.Response( + 503, + headers={"content-type": "application/json"}, + json={"error": "persistent"}, + ) + + client = LLMClient(private_key=TEST_PRIVATE_KEY) + client._client = httpx.Client(transport=httpx.MockTransport(handler)) + + from blockrun_llm.types import APIError + + with pytest.raises(APIError): + list( + client.chat_completion_stream( + "nvidia/deepseek-v4-flash", + [{"role": "user", "content": "hi"}], + ) + ) + # 1 + 3 backoffs == 4 probe attempts before raising. + assert len(calls) == 1 + len(LLMClient._STREAM_5XX_BACKOFFS) + + def test_5xx_retry_also_works_after_payment(self, monkeypatch): + """After signing a 402, subsequent 5xx on the retry stream should + also trigger in-band retries before raising.""" + monkeypatch.setattr("time.sleep", lambda _s: None) + + calls: list[httpx.Request] = [] + body = _sse_events(["paid-OK"]) + + def handler(request: httpx.Request) -> httpx.Response: + calls.append(request) + sig = request.headers.get("PAYMENT-SIGNATURE") + if not sig: + # Probe → 402 with payment requirements. + return httpx.Response( + 402, + headers={ + "content-type": "application/json", + "payment-required": build_payment_required_response(), + }, + json={"error": "Payment Required", "price": {"amount": "0.001"}}, + ) + # After payment: fail twice with 503, then succeed. + paid_calls = sum(1 for c in calls if c.headers.get("PAYMENT-SIGNATURE")) + if paid_calls <= 2: + return httpx.Response(503, json={"error": "transient"}) + return httpx.Response(200, headers={"content-type": "text/event-stream"}, content=body) + + client = LLMClient(private_key=TEST_PRIVATE_KEY) + client._client = httpx.Client(transport=httpx.MockTransport(handler)) + + chunks = list( + client.chat_completion_stream( + "openai/gpt-5.5", + [{"role": "user", "content": "hi"}], + ) + ) + # 1 probe (402) + 2 paid-503 + 1 paid-200 == 4 total + assert len(calls) == 4 + assert any(c.choices[0].delta.content == "paid-OK" for c in chunks) + + +# --------------------------------------------------------------------------- +# Fallback chain tests +# --------------------------------------------------------------------------- + + +class TestStreamingFallback: + """``fallback_models`` walks the chain only on retriable pre-stream + errors. Once a chunk is yielded, the upstream is committed.""" + + def test_falls_back_to_next_model_on_503(self, monkeypatch): + monkeypatch.setattr("time.sleep", lambda _s: None) + + calls: list[httpx.Request] = [] + + def handler(request: httpx.Request) -> httpx.Response: + calls.append(request) + body = request.read() + import json as _json + + payload = _json.loads(body) + if payload["model"] == "primary/bad": + return httpx.Response(503, json={"error": "down"}) + # Fallback model succeeds. + return httpx.Response( + 200, + headers={"content-type": "text/event-stream"}, + content=_sse_events(["FALLBACK"]), + ) + + client = LLMClient(private_key=TEST_PRIVATE_KEY) + client._client = httpx.Client(transport=httpx.MockTransport(handler)) + + chunks = list( + client.chat_completion_stream( + "primary/bad", + [{"role": "user", "content": "hi"}], + fallback_models=["fallback/good"], + ) + ) + + # 4 hits on primary (1 + 3 retries) all 503 → swap to fallback → 1 success + assert len(calls) >= 5 + assert any(c.choices[0].delta.content == "FALLBACK" for c in chunks) + + def test_no_fallback_after_first_chunk(self, monkeypatch): + """If the upstream successfully streams a few chunks then drops, we + must NOT fall back — partial output has already gone to the caller.""" + monkeypatch.setattr("time.sleep", lambda _s: None) + + # Build SSE that's truncated (no [DONE]) so iter_lines simulates a + # mid-stream connection drop via httpx parsing exception. + truncated = ( + b'data: {"id":"x","object":"chat.completion.chunk","created":1,' + b'"model":"primary/bad","choices":[{"index":0,"delta":{"role":"assistant"},' + b'"finish_reason":null}]}\n\n' + b'data: {"id":"x","object":"chat.completion.chunk","created":1,' + b'"model":"primary/bad","choices":[{"index":0,"delta":{"content":"par"},' + b'"finish_reason":null}]}\n\n' + ) + + calls: list[httpx.Request] = [] + + def handler(request: httpx.Request) -> httpx.Response: + calls.append(request) + return httpx.Response( + 200, + headers={"content-type": "text/event-stream"}, + content=truncated, + ) + + client = LLMClient(private_key=TEST_PRIVATE_KEY) + client._client = httpx.Client(transport=httpx.MockTransport(handler)) + + # Even with a fallback configured, the partial stream completes + # naturally — no exception, no fallback. The fallback handler should + # NEVER be invoked because we got valid chunks before the stream + # ended. + chunks = list( + client.chat_completion_stream( + "primary/bad", + [{"role": "user", "content": "hi"}], + fallback_models=["fallback/good"], + ) + ) + # Exactly one upstream call: no fallback because partial chunks were + # already yielded. + assert len(calls) == 1 + contents = [c.choices[0].delta.content for c in chunks if c.choices[0].delta.content] + assert "par" in contents + + def test_non_retriable_error_does_not_fall_back(self, monkeypatch): + """4xx (other than 402) must NOT trigger fallback — those are + permanent client errors, not transient upstream issues.""" + monkeypatch.setattr("time.sleep", lambda _s: None) + + calls: list[httpx.Request] = [] + + def handler(request: httpx.Request) -> httpx.Response: + calls.append(request) + return httpx.Response(400, json={"error": "bad request"}) + + client = LLMClient(private_key=TEST_PRIVATE_KEY) + client._client = httpx.Client(transport=httpx.MockTransport(handler)) + + from blockrun_llm.types import APIError + + with pytest.raises(APIError): + list( + client.chat_completion_stream( + "primary/bad", + [{"role": "user", "content": "hi"}], + fallback_models=["fallback/good"], + ) + ) + # Single attempt; no retries (400 isn't 5xx), no fallback (400 isn't retriable). + assert len(calls) == 1 + + +# --------------------------------------------------------------------------- +# Streamed tool calls — regression for the archive-loop crash +# --------------------------------------------------------------------------- + + +def _sse_with_tool_call(model: str = "anthropic/claude-haiku-4-5") -> bytes: + """SSE for a streamed tool call: role frame, a name frame, then argument- + fragment frames (id/name absent — these used to fail the strict ToolCall + schema), and a final finish=tool_calls frame with usage.""" + frames = [ + {"choices": [{"index": 0, "delta": {"role": "assistant"}, "finish_reason": None}]}, + { + "choices": [ + { + "index": 0, + "delta": { + "tool_calls": [ + { + "index": 0, + "id": "call_1", + "type": "function", + "function": {"name": "get_weather", "arguments": ""}, + } + ] + }, + "finish_reason": None, + } + ] + }, + { + "choices": [ + { + "index": 0, + "delta": {"tool_calls": [{"index": 0, "function": {"arguments": '{"city":'}}]}, + "finish_reason": None, + } + ] + }, + { + "choices": [ + { + "index": 0, + "delta": {"tool_calls": [{"index": 0, "function": {"arguments": '"Paris"}'}}]}, + "finish_reason": None, + } + ] + }, + { + "choices": [{"index": 0, "delta": {}, "finish_reason": "tool_calls"}], + "usage": {"prompt_tokens": 10, "completion_tokens": 5, "total_tokens": 15}, + }, + ] + lines = [] + for f in frames: + f = { + "id": "chatcmpl-tc", + "object": "chat.completion.chunk", + "created": 1700000000, + "model": model, + **f, + } + lines.append("data: " + json.dumps(f)) + lines.append("data: [DONE]") + return ("\n\n".join(lines) + "\n\n").encode("utf-8") + + +def _collect_tool_args(chunks: list[ChatCompletionChunk]) -> str: + out: list[str] = [] + for c in chunks: + if not c.choices: + continue + for tc in c.choices[0].delta.tool_calls or []: + if tc.function and tc.function.arguments: + out.append(tc.function.arguments) + return "".join(out) + + +class TestStreamedToolCalls: + """Streamed tool calls must parse + archive without crashing. + + The argument-fragment frames (id/name absent) used to fail the strict + ToolCall schema, fall back to model_construct (leaving choices as dicts), + then crash the archive loop with "'dict' object has no attribute 'delta'". + The PAID path is used so cost_usd > 0 and the archive loop actually runs. + """ + + def test_sync_streamed_tool_call(self): + calls: list[httpx.Request] = [] + client = LLMClient(private_key=TEST_PRIVATE_KEY) + client._client = httpx.Client( + transport=_make_paid_model_transport(_sse_with_tool_call(), calls) + ) + chunks = list( + client.chat_completion_stream( + "openai/gpt-5.5", + [{"role": "user", "content": "weather?"}], + max_tokens=64, + ) + ) + tool_frames = [c for c in chunks if c.choices and c.choices[0].delta.tool_calls] + assert tool_frames, "expected streamed tool_call deltas" + for c in tool_frames: + assert hasattr(c.choices[0], "delta") # parsed object, not a raw dict + assert _collect_tool_args(chunks) == '{"city":"Paris"}' + finishes = [ + c.choices[0].finish_reason for c in chunks if c.choices and c.choices[0].finish_reason + ] + assert finishes == ["tool_calls"] + + @pytest.mark.asyncio + async def test_async_streamed_tool_call(self): + calls: list[httpx.Request] = [] + client = AsyncLLMClient(private_key=TEST_PRIVATE_KEY) + await client._client.aclose() + client._client = httpx.AsyncClient( + transport=_make_paid_model_transport(_sse_with_tool_call(), calls) + ) + chunks: list[ChatCompletionChunk] = [] + async for chunk in client.chat_completion_stream( + "openai/gpt-5.5", + [{"role": "user", "content": "weather?"}], + ): + chunks.append(chunk) + assert _collect_tool_args(chunks) == '{"city":"Paris"}' + await client.close() + + def test_sync_streamed_tool_call_non_function_type(self): + """A non-"function" tool ``type`` must still parse into a real object + rather than re-trigger the strict-validation -> model_construct fallback + (which would leave choices as raw dicts and crash consumers).""" + frames = [ + { + "id": "chatcmpl-tc", + "object": "chat.completion.chunk", + "created": 1700000000, + "model": "anthropic/claude-haiku-4-5", + "choices": [ + { + "index": 0, + "delta": { + "tool_calls": [ + { + "index": 0, + "id": "call_1", + "type": "custom", # non-"function" type + "function": { + "name": "get_weather", + "arguments": '{"city":"Paris"}', + }, + } + ] + }, + "finish_reason": None, + } + ], + }, + { + "id": "chatcmpl-tc", + "object": "chat.completion.chunk", + "created": 1700000000, + "model": "anthropic/claude-haiku-4-5", + "choices": [{"index": 0, "delta": {}, "finish_reason": "tool_calls"}], + "usage": {"prompt_tokens": 10, "completion_tokens": 5, "total_tokens": 15}, + }, + ] + sse = ( + "\n\n".join("data: " + json.dumps(f) for f in frames) + "\n\ndata: [DONE]\n\n" + ).encode() + calls: list[httpx.Request] = [] + client = LLMClient(private_key=TEST_PRIVATE_KEY) + client._client = httpx.Client(transport=_make_paid_model_transport(sse, calls)) + chunks = list( + client.chat_completion_stream( + "openai/gpt-5.5", + [{"role": "user", "content": "weather?"}], + max_tokens=64, + ) + ) + tool_frames = [c for c in chunks if c.choices and c.choices[0].delta.tool_calls] + assert tool_frames, "expected the non-'function' tool_call delta to parse" + assert tool_frames[0].choices[0].delta.tool_calls[0].type == "custom" + assert _collect_tool_args(chunks) == '{"city":"Paris"}' + + def test_sync_stream_archive_survives_model_construct_chunk_missing_id(self): + """Archive-loop hardening: a frame that omits the required top-level + ``id`` fails strict validation and is yielded via ``model_construct`` + (no ``id`` attribute). The stream-archiving loop must not crash reading + ``chunk.id`` (old behaviour: ``AttributeError``); draining the generator + runs the paid archive path end to end.""" + frames = [ + # Missing "id" -> model_construct, no .id attribute on the chunk. + { + "object": "chat.completion.chunk", + "created": 1700000000, + "model": "anthropic/claude-haiku-4-5", + "choices": [{"index": 0, "delta": {"content": "hi"}, "finish_reason": None}], + }, + { + "id": "chatcmpl-tc", + "object": "chat.completion.chunk", + "created": 1700000000, + "model": "anthropic/claude-haiku-4-5", + "choices": [{"index": 0, "delta": {"content": " there"}, "finish_reason": "stop"}], + "usage": {"prompt_tokens": 10, "completion_tokens": 5, "total_tokens": 15}, + }, + ] + sse = ( + "\n\n".join("data: " + json.dumps(f) for f in frames) + "\n\ndata: [DONE]\n\n" + ).encode() + calls: list[httpx.Request] = [] + client = LLMClient(private_key=TEST_PRIVATE_KEY) + client._client = httpx.Client(transport=_make_paid_model_transport(sse, calls)) + # Must not raise: the archive loop reads chunk.id via the dict/attr-tolerant + # accessor, so a model_construct'd chunk missing id is skipped, not fatal. + chunks = list( + client.chat_completion_stream( + "openai/gpt-5.5", + [{"role": "user", "content": "hi"}], + max_tokens=64, + ) + ) + assert len(chunks) == 2 # both frames yielded, stream drained cleanly diff --git a/tests/unit/test_streaming_solana.py b/tests/unit/test_streaming_solana.py new file mode 100644 index 0000000..386db30 --- /dev/null +++ b/tests/unit/test_streaming_solana.py @@ -0,0 +1,300 @@ +""" +Unit tests for SolanaLLMClient.chat_completion_stream (sync only — async +isn't implemented for the Solana client yet). + +We mock at the httpx transport level so no real wallet / RPC / network +is needed. The Solana payment-signing path uses the x402 SDK's SVM +client, which we patch with a small fake that returns a static encoded +payload — this isolates the SSE/retry logic from the cryptography. +""" + +from __future__ import annotations + +import json + +import httpx +import pytest + +# Skip the whole module if the solana extras aren't installed. +pytest.importorskip("x402") +pytest.importorskip("solders") + +from blockrun_llm import SolanaLLMClient +from blockrun_llm.types import APIError, PaymentError + +# --------------------------------------------------------------------------- +# Helpers — synthetic SSE bodies (same shape Base tests use) +# --------------------------------------------------------------------------- + + +def _sse_events(deltas: list[str], finish: str = "stop", model: str = "test/model") -> bytes: + lines: list[str] = [] + lines.append( + "data: " + + json.dumps( + { + "id": "chatcmpl-test", + "object": "chat.completion.chunk", + "created": 1700000000, + "model": model, + "choices": [{"index": 0, "delta": {"role": "assistant"}, "finish_reason": None}], + } + ) + ) + for d in deltas: + lines.append( + "data: " + + json.dumps( + { + "id": "chatcmpl-test", + "object": "chat.completion.chunk", + "created": 1700000000, + "model": model, + "choices": [{"index": 0, "delta": {"content": d}, "finish_reason": None}], + } + ) + ) + lines.append( + "data: " + + json.dumps( + { + "id": "chatcmpl-test", + "object": "chat.completion.chunk", + "created": 1700000000, + "model": model, + "choices": [{"index": 0, "delta": {}, "finish_reason": finish}], + } + ) + ) + lines.append("data: [DONE]") + return ("\n\n".join(lines) + "\n\n").encode("utf-8") + + +# A valid Solana keypair seed (32 bytes, base58-encoded). Hardcoded test value; +# never use in production. +TEST_SOLANA_KEY = "AQEBAQEBAQEBAQEBAQEBAQEBAQEBAQEBAQEBAQEBAQE" # 32 1-bytes + + +@pytest.fixture +def solana_client(): + """Build a SolanaLLMClient without going through the x402 SDK signer + init (which needs real keys + an RPC). We monkey-patch the signer + after construction by replacing the x402_client with a fake.""" + from unittest import mock + + with ( + mock.patch("blockrun_llm.solana_client.register_exact_svm_client"), + mock.patch("blockrun_llm.solana_client._create_signer"), + ): + client = SolanaLLMClient( + private_key="bogus_not_used_because_signer_is_patched", + api_url="https://sol.blockrun.ai/api", + rpc_url="http://test", + ) + + # Stub the signing path: x402_client.create_payment_payload returns an + # object with the right shape for the rest of the code. + class _FakePayload: + class accepted: + amount = "1000000" # 1 USDC in micro-units + + client._x402_client = mock.MagicMock() + client._x402_client.create_payment_payload.return_value = _FakePayload() + return client + + +def _patch_sse_helpers(monkeypatch): + """Replace the x402 SDK's decode/encode functions with identity-ish + stubs so our handler code can run without real x402 payloads.""" + monkeypatch.setattr( + "blockrun_llm.solana_client.decode_payment_required_header", + lambda header: {"stub": True}, + ) + monkeypatch.setattr( + "blockrun_llm.solana_client.encode_payment_signature_header", + lambda payload: "stub-signature", + ) + + +# --------------------------------------------------------------------------- +# Transport builders +# --------------------------------------------------------------------------- + + +def _free_transport(sse_body: bytes, calls: list[httpx.Request]) -> httpx.MockTransport: + def handler(request: httpx.Request) -> httpx.Response: + calls.append(request) + return httpx.Response(200, headers={"content-type": "text/event-stream"}, content=sse_body) + + return httpx.MockTransport(handler) + + +def _paid_transport(sse_body: bytes, calls: list[httpx.Request]) -> httpx.MockTransport: + """First call → 402; second call (with PAYMENT-SIGNATURE) → 200 SSE.""" + + def handler(request: httpx.Request) -> httpx.Response: + calls.append(request) + if "PAYMENT-SIGNATURE" not in request.headers: + return httpx.Response( + 402, + headers={ + "content-type": "application/json", + "payment-required": "stub-payment-required-base64", + }, + json={"error": "Payment Required"}, + ) + return httpx.Response(200, headers={"content-type": "text/event-stream"}, content=sse_body) + + return httpx.MockTransport(handler) + + +def _flaky_transport( + sse_body: bytes, fail_count: int, calls: list[httpx.Request], status: int = 503 +) -> httpx.MockTransport: + def handler(request: httpx.Request) -> httpx.Response: + calls.append(request) + if len(calls) <= fail_count: + return httpx.Response(status, json={"error": "transient"}) + return httpx.Response(200, headers={"content-type": "text/event-stream"}, content=sse_body) + + return httpx.MockTransport(handler) + + +# --------------------------------------------------------------------------- +# Tests +# --------------------------------------------------------------------------- + + +class TestSolanaStreaming: + def test_free_model_streams_directly(self, solana_client, monkeypatch): + calls: list[httpx.Request] = [] + solana_client._client = httpx.Client( + transport=_free_transport(_sse_events(["Hello", " world"]), calls) + ) + + chunks = list( + solana_client.chat_completion_stream( + "nvidia/deepseek-v4-flash", + [{"role": "user", "content": "hi"}], + ) + ) + + assert len(calls) == 1 + assert "PAYMENT-SIGNATURE" not in calls[0].headers + content = "".join(c.choices[0].delta.content for c in chunks if c.choices[0].delta.content) + assert content == "Hello world" + + def test_paid_model_signs_and_retries(self, solana_client, monkeypatch): + _patch_sse_helpers(monkeypatch) + calls: list[httpx.Request] = [] + solana_client._client = httpx.Client( + transport=_paid_transport(_sse_events(["Paid"]), calls) + ) + + chunks = list( + solana_client.chat_completion_stream( + "openai/gpt-5.5", + [{"role": "user", "content": "hi"}], + ) + ) + # 1 probe (402) + 1 paid (200) == 2 total + assert len(calls) == 2 + assert "PAYMENT-SIGNATURE" not in calls[0].headers + assert calls[1].headers["PAYMENT-SIGNATURE"] == "stub-signature" + content = "".join(c.choices[0].delta.content for c in chunks if c.choices[0].delta.content) + assert content == "Paid" + assert solana_client._session_calls == 1 + assert solana_client._last_call_cost > 0 + + def test_retries_5xx_with_backoff(self, solana_client, monkeypatch): + monkeypatch.setattr("time.sleep", lambda _s: None) + calls: list[httpx.Request] = [] + solana_client._client = httpx.Client( + transport=_flaky_transport(_sse_events(["OK"]), fail_count=2, calls=calls) + ) + + chunks = list( + solana_client.chat_completion_stream( + "nvidia/deepseek-v4-flash", + [{"role": "user", "content": "hi"}], + ) + ) + # 2 failed + 1 success + assert len(calls) == 3 + assert any(c.choices[0].delta.content == "OK" for c in chunks) + + def test_raises_after_exhausting_retries(self, solana_client, monkeypatch): + monkeypatch.setattr("time.sleep", lambda _s: None) + calls: list[httpx.Request] = [] + + def handler(request: httpx.Request) -> httpx.Response: + calls.append(request) + return httpx.Response(503, json={"error": "persistent"}) + + solana_client._client = httpx.Client(transport=httpx.MockTransport(handler)) + + with pytest.raises(APIError): + list( + solana_client.chat_completion_stream( + "nvidia/deepseek-v4-flash", + [{"role": "user", "content": "hi"}], + ) + ) + # 1 + 3 backoffs == 4 attempts + assert len(calls) == 1 + len(SolanaLLMClient._STREAM_5XX_BACKOFFS) + + def test_fallback_models_walks_chain(self, solana_client, monkeypatch): + _patch_sse_helpers(monkeypatch) + monkeypatch.setattr("time.sleep", lambda _s: None) + calls: list[httpx.Request] = [] + + def handler(request: httpx.Request) -> httpx.Response: + calls.append(request) + body = json.loads(request.read()) + if body["model"] == "primary/bad": + return httpx.Response(503, json={"error": "down"}) + return httpx.Response( + 200, + headers={"content-type": "text/event-stream"}, + content=_sse_events(["FALLBACK"]), + ) + + solana_client._client = httpx.Client(transport=httpx.MockTransport(handler)) + + chunks = list( + solana_client.chat_completion_stream( + "primary/bad", + [{"role": "user", "content": "hi"}], + fallback_models=["fallback/good"], + ) + ) + # 4 calls to primary/bad all 503, then 1 to fallback/good + assert len(calls) >= 5 + assert any(c.choices[0].delta.content == "FALLBACK" for c in chunks) + + def test_payment_rejected_raises_payment_error(self, solana_client, monkeypatch): + _patch_sse_helpers(monkeypatch) + monkeypatch.setattr("time.sleep", lambda _s: None) + calls: list[httpx.Request] = [] + + def handler(request: httpx.Request) -> httpx.Response: + calls.append(request) + # Always 402, even after signing. + return httpx.Response( + 402, + headers={ + "content-type": "application/json", + "payment-required": "stub-payment-required-base64", + }, + json={"error": "Payment Required"}, + ) + + solana_client._client = httpx.Client(transport=httpx.MockTransport(handler)) + + with pytest.raises(PaymentError): + list( + solana_client.chat_completion_stream( + "openai/gpt-5.5", + [{"role": "user", "content": "hi"}], + ) + ) diff --git a/tests/unit/test_tx_log.py b/tests/unit/test_tx_log.py new file mode 100644 index 0000000..e39a243 --- /dev/null +++ b/tests/unit/test_tx_log.py @@ -0,0 +1,227 @@ +"""Unit tests for the opt-in project-local transaction log. + +Covered surface: +* ``TransactionLogger.log`` formats one plain-text row per call. +* The on-chain ``tx_hash`` is pulled from a decoded ``X-PAYMENT-RESPONSE`` + payload (both EVM ``transaction`` and Solana ``signature`` field names). +* ``format_row`` matches the column layout shown in the README so future + reformat regressions get caught immediately. +* ``_resolve_log_dir`` honors the constructor argument + ``BLOCKRUN_TX_LOG`` + env var fallback. +""" + +from __future__ import annotations + +import base64 +import json +from pathlib import Path + +from blockrun_llm.tx_log import ( + DEFAULT_LOG_DIR, + TransactionLogger, + _resolve_log_dir, + decode_settlement_header, + format_row, +) + +# --------------------------------------------------------------------------- +# Logger writes +# --------------------------------------------------------------------------- + + +def test_log_writes_one_row(tmp_path): + logger = TransactionLogger(tmp_path) + logger.log( + endpoint="/v1/chat/completions", + request={"model": "openai/gpt-5.5", "messages": []}, + response={"usage": {"prompt_tokens": 14, "completion_tokens": 18}}, + cost_usd=0.001, + settlement={"tx_hash": "0x421796a3deadbeef"}, + ) + rows = logger.entries() + assert len(rows) == 1 + row = rows[0] + assert "chat" in row + assert "openai/gpt-5.5" in row + assert "in= 14" in row + assert "out=18" in row + assert "$0.001000" in row + assert "0x421796a3" in row # truncated to 10 chars + ellipsis + + +def test_log_appends(tmp_path): + """Two calls → two lines, oldest first.""" + logger = TransactionLogger(tmp_path) + for i in range(2): + logger.log( + endpoint="/v1/chat/completions", + request={"model": "openai/gpt-5.5"}, + response={"usage": {"prompt_tokens": i, "completion_tokens": i}}, + cost_usd=0.001, + settlement={"tx_hash": f"0x{i:064x}"}, + ) + rows = logger.entries() + assert len(rows) == 2 + # Second row should have the second tx hash + assert "0x00000000" in rows[0] + assert rows[0] != rows[1] + + +def test_log_without_settlement_emits_placeholder(tmp_path): + """A free / cached call → no tx hash → ``(no-tx)``.""" + logger = TransactionLogger(tmp_path) + logger.log( + endpoint="/v1/chat/completions", + request={"model": "free/model"}, + response={"usage": {"prompt_tokens": 1, "completion_tokens": 1}}, + cost_usd=0.0, + ) + assert "(no-tx)" in logger.entries()[0] + + +# --------------------------------------------------------------------------- +# Settlement header decoding +# --------------------------------------------------------------------------- + + +def _b64(obj): + return base64.b64encode(json.dumps(obj).encode()).decode() + + +def test_decode_evm_settlement(): + settlement = decode_settlement_header( + _b64( + { + "success": True, + "transaction": "0xdeadbeef", + "network": "eip155:8453", + "payer": "0xabc", + "payee": "0xdef", + "amount": "1000", + } + ) + ) + assert settlement["tx_hash"] == "0xdeadbeef" + assert settlement["amount_micro_usdc"] == "1000" + assert settlement["network"] == "eip155:8453" + + +def test_decode_solana_settlement_uses_signature(): + settlement = decode_settlement_header( + _b64({"signature": "5h7Kabc…", "network": "solana:mainnet", "amount": 500}) + ) + assert settlement["tx_hash"] == "5h7Kabc…" + assert settlement["amount_micro_usdc"] == "500" + + +def test_decode_returns_none_for_missing_or_garbage(): + assert decode_settlement_header(None) is None + assert decode_settlement_header("not-base64") is None + # Valid base64 but not JSON + assert decode_settlement_header(base64.b64encode(b"not-json").decode()) is None + + +# --------------------------------------------------------------------------- +# Row formatting (regression guard for the README example) +# --------------------------------------------------------------------------- + + +def test_format_row_matches_readme_layout(): + row = format_row( + ts=1747842286.0, # arbitrary + endpoint="/v1/chat/completions", + model="anthropic/claude-sonnet-4.6", + in_tokens=3, + out_tokens=4, + cost_usd=0.034137, + tx_hash="0x6513d12812345", + ) + assert "chat" in row + assert "anthropic/claude-sonnet-4.6" in row + assert "in= 3" in row + assert "out=4" in row + assert "$0.034137" in row + assert "0x6513d128" in row + assert row.endswith("0x6513d128…") + + +# --------------------------------------------------------------------------- +# Path resolution / env-var fallback +# --------------------------------------------------------------------------- + + +def test_resolve_log_dir_truthy_returns_default(): + assert _resolve_log_dir(True) == DEFAULT_LOG_DIR + + +def test_resolve_log_dir_path_passes_through(tmp_path): + assert _resolve_log_dir(str(tmp_path)) == Path(str(tmp_path)) + + +def test_resolve_log_dir_false_is_disabled(): + assert _resolve_log_dir(False) is None + + +def test_resolve_log_dir_env_enables_default(monkeypatch): + monkeypatch.setenv("BLOCKRUN_TX_LOG", "1") + assert _resolve_log_dir(None) == DEFAULT_LOG_DIR + + +def test_resolve_log_dir_env_path(monkeypatch, tmp_path): + monkeypatch.setenv("BLOCKRUN_TX_LOG", str(tmp_path)) + assert _resolve_log_dir(None) == Path(str(tmp_path)) + + +def test_resolve_log_dir_env_missing_is_disabled(monkeypatch): + monkeypatch.delenv("BLOCKRUN_TX_LOG", raising=False) + assert _resolve_log_dir(None) is None + + +# --------------------------------------------------------------------------- +# Best-effort behaviour +# --------------------------------------------------------------------------- + + +def test_log_into_unwritable_dir_returns_none(tmp_path): + """A read-only parent must not crash a paid call.""" + parent = tmp_path / "ro" + parent.mkdir() + parent.chmod(0o500) # read+execute, no write + try: + logger = TransactionLogger(parent / "log") + result = logger.log( + endpoint="/v1/chat/completions", + request={"model": "x"}, + response={}, + cost_usd=0.0, + ) + assert result is None + finally: + parent.chmod(0o700) # restore so pytest can clean up + + +def test_pydantic_usage_object_is_handled(): + """Pydantic response objects with usage attrs should not crash the row. + + Mirrors how ``LLMClient`` passes a ``ChatResponse`` in for chat calls.""" + + class Usage: + prompt_tokens = 7 + completion_tokens = 3 + + class Resp: + usage = Usage() + + row = format_row( + endpoint="/v1/chat/completions", + model="openai/gpt-5.5", + in_tokens=7, + out_tokens=3, + cost_usd=0.001, + tx_hash="0xabc", + ) + assert "in= 7" in row and "out=3" in row + # _extract_tokens path: feed via TransactionLogger + from blockrun_llm.tx_log import _extract_tokens + + assert _extract_tokens(Resp()) == (7, 3) diff --git a/tests/unit/test_validation.py b/tests/unit/test_validation.py index 9bb2e04..3ae142e 100644 --- a/tests/unit/test_validation.py +++ b/tests/unit/test_validation.py @@ -1,279 +1,406 @@ -"""Unit tests for validation module.""" - -import pytest -from blockrun_llm.validation import ( - validate_private_key, - validate_api_url, - validate_model, - validate_max_tokens, - validate_temperature, - validate_top_p, - sanitize_error_response, - validate_resource_url, -) - - -class TestValidatePrivateKey: - def test_valid_private_key(self): - """Should accept valid private key.""" - key = "0xac0974bec39a17e36ba4a6b4d238ff944bacb478cbed5efcae784d7bf4f2ff80" - validate_private_key(key) # Should not raise - - def test_reject_non_string(self): - """Should reject non-string input.""" - with pytest.raises(ValueError, match="must be a string"): - validate_private_key(123) # type: ignore - - def test_reject_no_prefix(self): - """Should reject key without 0x prefix.""" - with pytest.raises(ValueError, match="must start with 0x"): - validate_private_key( - "ac0974bec39a17e36ba4a6b4d238ff944bacb478cbed5efcae784d7bf4f2ff80" - ) - - def test_reject_short_key(self): - """Should reject short key.""" - with pytest.raises(ValueError, match="66 characters"): - validate_private_key("0x123") - - def test_reject_long_key(self): - """Should reject long key.""" - with pytest.raises(ValueError, match="66 characters"): - validate_private_key( - "0xac0974bec39a17e36ba4a6b4d238ff944bacb478cbed5efcae784d7bf4f2ff80123" - ) - - def test_reject_non_hex(self): - """Should reject non-hexadecimal characters.""" - with pytest.raises(ValueError, match="hexadecimal"): - validate_private_key( - "0xGGGG74bec39a17e36ba4a6b4d238ff944bacb478cbed5efcae784d7bf4f2ff80" - ) - - def test_accept_uppercase(self): - """Should accept uppercase hex.""" - key = "0xAC0974BEC39A17E36BA4A6B4D238FF944BACB478CBED5EFCAE784D7BF4F2FF80" - validate_private_key(key) # Should not raise - - def test_accept_mixed_case(self): - """Should accept mixed case hex.""" - key = "0xAc0974Bec39a17e36ba4a6b4d238ff944bacb478cbed5efcae784d7bf4f2ff80" - validate_private_key(key) # Should not raise - - -class TestValidateApiUrl: - def test_accept_https(self): - """Should accept HTTPS URLs.""" - validate_api_url("https://api.blockrun.ai") - validate_api_url("https://example.com:8443") - - def test_accept_localhost_http(self): - """Should accept localhost HTTP.""" - validate_api_url("http://localhost") - validate_api_url("http://localhost:3000") - validate_api_url("http://127.0.0.1") - validate_api_url("http://127.0.0.1:8080") - - def test_reject_http_production(self): - """Should reject HTTP for non-localhost.""" - with pytest.raises(ValueError, match="HTTPS"): - validate_api_url("http://api.example.com") - with pytest.raises(ValueError, match="HTTPS"): - validate_api_url("http://192.168.1.1") - - def test_reject_invalid_url(self): - """Should reject invalid URL format.""" - with pytest.raises(ValueError, match="Invalid"): - validate_api_url("not-a-url") - with pytest.raises(ValueError, match="Invalid"): - validate_api_url("") - - -class TestValidateModel: - def test_accept_valid_model(self): - """Should accept valid model IDs.""" - validate_model("openai/gpt-4o") - validate_model("anthropic/claude-sonnet-4.5") - validate_model("google/gemini-2.5-flash") - - def test_reject_empty_string(self): - """Should reject empty string.""" - with pytest.raises(ValueError, match="non-empty string"): - validate_model("") - - def test_reject_non_string(self): - """Should reject non-string.""" - with pytest.raises(ValueError, match="non-empty string"): - validate_model(None) # type: ignore - - -class TestValidateMaxTokens: - def test_accept_valid_values(self): - """Should accept valid max_tokens.""" - validate_max_tokens(1) - validate_max_tokens(100) - validate_max_tokens(1000) - validate_max_tokens(100000) - - def test_accept_none(self): - """Should accept None.""" - validate_max_tokens(None) - - def test_reject_negative(self): - """Should reject negative values.""" - with pytest.raises(ValueError, match="positive"): - validate_max_tokens(-1) - - def test_reject_zero(self): - """Should reject zero.""" - with pytest.raises(ValueError, match="positive"): - validate_max_tokens(0) - - def test_reject_too_large(self): - """Should reject values too large.""" - with pytest.raises(ValueError, match="too large"): - validate_max_tokens(200000) - - def test_reject_non_integer(self): - """Should reject non-integer.""" - with pytest.raises(ValueError, match="integer"): - validate_max_tokens(100.5) # type: ignore - - -class TestValidateTemperature: - def test_accept_valid_values(self): - """Should accept valid temperature.""" - validate_temperature(0.0) - validate_temperature(0.7) - validate_temperature(1.0) - validate_temperature(2.0) - - def test_accept_none(self): - """Should accept None.""" - validate_temperature(None) - - def test_reject_negative(self): - """Should reject negative values.""" - with pytest.raises(ValueError, match="between 0 and 2"): - validate_temperature(-0.1) - - def test_reject_too_large(self): - """Should reject values > 2.""" - with pytest.raises(ValueError, match="between 0 and 2"): - validate_temperature(2.1) - - def test_reject_non_number(self): - """Should reject non-numeric.""" - with pytest.raises(ValueError, match="number"): - validate_temperature("0.7") # type: ignore - - -class TestValidateTopP: - def test_accept_valid_values(self): - """Should accept valid top_p.""" - validate_top_p(0.0) - validate_top_p(0.5) - validate_top_p(0.9) - validate_top_p(1.0) - - def test_accept_none(self): - """Should accept None.""" - validate_top_p(None) - - def test_reject_negative(self): - """Should reject negative values.""" - with pytest.raises(ValueError, match="between 0 and 1"): - validate_top_p(-0.1) - - def test_reject_too_large(self): - """Should reject values > 1.""" - with pytest.raises(ValueError, match="between 0 and 1"): - validate_top_p(1.1) - - def test_reject_non_number(self): - """Should reject non-numeric.""" - with pytest.raises(ValueError, match="number"): - validate_top_p("0.9") # type: ignore - - -class TestSanitizeErrorResponse: - def test_extract_safe_fields(self): - """Should extract only safe error fields.""" - result = sanitize_error_response( - { - "error": "User-facing error", - "internal_stack": "/var/app/sensitive.py:123", - "api_key": "sk-secret", - "database_url": "postgres://user:pass@host/db", - } - ) - assert result == {"message": "User-facing error", "code": None} - - def test_include_code_if_present(self): - """Should include code if present.""" - result = sanitize_error_response( - {"error": "Invalid request", "code": "invalid_request_error"} - ) - assert result == {"message": "Invalid request", "code": "invalid_request_error"} - - def test_handle_non_dict(self): - """Should handle non-dict input.""" - assert sanitize_error_response("error") == { - "message": "API request failed", - "code": None, - } - assert sanitize_error_response(None) == { - "message": "API request failed", - "code": None, - } - assert sanitize_error_response(123) == { - "message": "API request failed", - "code": None, - } - - def test_handle_missing_error_field(self): - """Should handle missing error field.""" - result = sanitize_error_response({"something": "else"}) - assert result == {"message": "API request failed", "code": None} - - -class TestValidateResourceUrl: - def test_allow_matching_domain(self): - """Should allow matching domain.""" - result = validate_resource_url( - "https://api.blockrun.ai/v1/chat", "https://api.blockrun.ai" - ) - assert result == "https://api.blockrun.ai/v1/chat" - - def test_allow_different_path(self): - """Should allow different path on same domain.""" - result = validate_resource_url( - "https://api.blockrun.ai/v2/models", "https://api.blockrun.ai" - ) - assert result == "https://api.blockrun.ai/v2/models" - - def test_reject_different_domain(self): - """Should reject different domain.""" - result = validate_resource_url( - "https://malicious.com/steal", "https://api.blockrun.ai" - ) - assert result == "https://api.blockrun.ai/v1/chat/completions" - - def test_reject_different_protocol(self): - """Should reject different protocol.""" - result = validate_resource_url( - "http://api.blockrun.ai/v1/chat", "https://api.blockrun.ai" - ) - assert result == "https://api.blockrun.ai/v1/chat/completions" - - def test_handle_invalid_url(self): - """Should handle invalid URL format.""" - result = validate_resource_url("not-a-url", "https://api.blockrun.ai") - assert result == "https://api.blockrun.ai/v1/chat/completions" - - def test_reject_subdomain_difference(self): - """Should reject subdomain differences.""" - result = validate_resource_url( - "https://evil.api.blockrun.ai/v1/chat", "https://api.blockrun.ai" - ) - assert result == "https://api.blockrun.ai/v1/chat/completions" +"""Unit tests for validation module.""" + +import pytest + +from blockrun_llm.validation import ( + MAX_TOKENS_SANITY_LIMIT, + sanitize_error_response, + validate_api_url, + validate_max_tokens, + validate_model, + validate_private_key, + validate_resource_url, + validate_temperature, + validate_top_p, +) + + +class TestValidatePrivateKey: + def test_valid_private_key(self): + """Should accept valid private key.""" + key = "0xac0974bec39a17e36ba4a6b4d238ff944bacb478cbed5efcae784d7bf4f2ff80" + validate_private_key(key) # Should not raise + + def test_reject_non_string(self): + """Should reject non-string input.""" + with pytest.raises(ValueError, match="must be a string"): + validate_private_key(123) # type: ignore + + def test_reject_no_prefix(self): + """Should reject key without 0x prefix.""" + with pytest.raises(ValueError, match="must start with 0x"): + validate_private_key("ac0974bec39a17e36ba4a6b4d238ff944bacb478cbed5efcae784d7bf4f2ff80") + + def test_reject_short_key(self): + """Should reject short key.""" + with pytest.raises(ValueError, match="66 characters"): + validate_private_key("0x123") + + def test_reject_long_key(self): + """Should reject long key.""" + with pytest.raises(ValueError, match="66 characters"): + validate_private_key( + "0xac0974bec39a17e36ba4a6b4d238ff944bacb478cbed5efcae784d7bf4f2ff80123" + ) + + def test_reject_non_hex(self): + """Should reject non-hexadecimal characters.""" + with pytest.raises(ValueError, match="hexadecimal"): + validate_private_key( + "0xGGGG74bec39a17e36ba4a6b4d238ff944bacb478cbed5efcae784d7bf4f2ff80" + ) + + def test_accept_uppercase(self): + """Should accept uppercase hex.""" + key = "0xAC0974BEC39A17E36BA4A6B4D238FF944BACB478CBED5EFCAE784D7BF4F2FF80" + validate_private_key(key) # Should not raise + + def test_accept_mixed_case(self): + """Should accept mixed case hex.""" + key = "0xAc0974Bec39a17e36ba4a6b4d238ff944bacb478cbed5efcae784d7bf4f2ff80" + validate_private_key(key) # Should not raise + + def test_reject_solana_base58_keypair_with_helpful_message(self): + """A 64-byte base58 Solana keypair should point users to SolanaLLMClient.""" + key = "3zZXZ37shyzxw7ZUePxwvJk8wkab8vPjHY6AWwE7CTJzZSP6zp8hnYNSsL6U4FgkacrbMhq2c1BZwgoKu17tdUa8" + with pytest.raises(ValueError, match="SolanaLLMClient"): + validate_private_key(key) + + def test_reject_solana_base58_seed_with_helpful_message(self): + """A 32-byte base58 Solana seed should point users to SolanaLLMClient.""" + key = "B5Fx69Nhu21vhFotKkFsURy554TqSo5ESN7ew4M6yjvH" + with pytest.raises(ValueError, match="SolanaLLMClient"): + validate_private_key(key) + + def test_reject_solana_base58_keypair_with_0x_prefix(self): + """Even after a caller prepends 0x, a Solana key should be detected.""" + key = "0x3zZXZ37shyzxw7ZUePxwvJk8wkab8vPjHY6AWwE7CTJzZSP6zp8hnYNSsL6U4FgkacrbMhq2c1BZwgoKu17tdUa8" + with pytest.raises(ValueError, match="Solana"): + validate_private_key(key) + + +class TestValidateApiUrl: + def test_accept_https(self): + """Should accept HTTPS URLs.""" + validate_api_url("https://api.blockrun.ai") + validate_api_url("https://example.com:8443") + + def test_accept_localhost_http(self): + """Should accept localhost HTTP.""" + validate_api_url("http://localhost") + validate_api_url("http://localhost:3000") + validate_api_url("http://127.0.0.1") + validate_api_url("http://127.0.0.1:8080") + + def test_reject_http_production(self): + """Should reject HTTP for non-localhost.""" + with pytest.raises(ValueError, match="HTTPS"): + validate_api_url("http://api.example.com") + with pytest.raises(ValueError, match="HTTPS"): + validate_api_url("http://192.168.1.1") + + def test_reject_invalid_url(self): + """Should reject invalid URL format.""" + with pytest.raises(ValueError, match="scheme"): + validate_api_url("not-a-url") + with pytest.raises(ValueError, match="scheme"): + validate_api_url("") + + +class TestValidateModel: + def test_accept_valid_model(self): + """Should accept valid model IDs.""" + validate_model("openai/gpt-5.2") + validate_model("anthropic/claude-sonnet-4.5") + validate_model("google/gemini-2.5-flash") + + def test_reject_empty_string(self): + """Should reject empty string.""" + with pytest.raises(ValueError, match="non-empty string"): + validate_model("") + + def test_reject_non_string(self): + """Should reject non-string.""" + with pytest.raises(ValueError, match="non-empty string"): + validate_model(None) # type: ignore + + +class TestValidateMaxTokens: + def test_accept_valid_values(self): + """Should accept valid max_tokens.""" + validate_max_tokens(1) + validate_max_tokens(100) + validate_max_tokens(1000) + validate_max_tokens(100000) + + def test_accept_none(self): + """Should accept None.""" + validate_max_tokens(None) + + def test_reject_negative(self): + """Should reject negative values.""" + with pytest.raises(ValueError, match="positive"): + validate_max_tokens(-1) + + def test_reject_zero(self): + """Should reject zero.""" + with pytest.raises(ValueError, match="positive"): + validate_max_tokens(0) + + def test_reject_implausible(self): + """Should reject values no model could mean (typo guard, not a limit).""" + with pytest.raises(ValueError, match="implausibly large") as exc: + validate_max_tokens(MAX_TOKENS_SANITY_LIMIT * 2) + # The message must name the SDK's own number, so a caller can tell it + # apart from a provider ceiling. + assert str(MAX_TOKENS_SANITY_LIMIT) in str(exc.value) + + def test_boundary_is_inclusive(self): + """Pin the exact edge: the limit passes, one above it fails. + + Without this, flipping ``>`` to ``>=`` keeps the suite green while + rejecting a legal value. + """ + validate_max_tokens(MAX_TOKENS_SANITY_LIMIT) + with pytest.raises(ValueError, match="implausibly large"): + validate_max_tokens(MAX_TOKENS_SANITY_LIMIT + 1) + + def test_does_not_cap_below_real_ceilings(self): + """The bound must never be the binding constraint on a real request. + + Regression: this was 100000, which rejected every ceiling above it + client-side — the caller saw a ValueError naming a limit no provider + had set, and the request never reached the network. 128000 is the + common ceiling (opus-4.8 / sonnet-5 / gpt-5.6 / glm-5); 262144 is + zai/glm-5.2, the highest any model serves. + """ + for real_ceiling in (128_000, 262_144): + validate_max_tokens(real_ceiling) + + def test_reject_non_integer(self): + """Should reject non-integer.""" + with pytest.raises(ValueError, match="integer"): + validate_max_tokens(100.5) # type: ignore + + def test_reject_bool(self): + """bool is an int subclass, so `isinstance(True, int)` is True and a + flag threaded into the wrong keyword reached the wire as + `"max_tokens": true`. It is not a token count.""" + for bad in (True, False): + with pytest.raises(ValueError, match="bool"): + validate_max_tokens(bad) # type: ignore + + +class TestValidateTemperature: + def test_accept_valid_values(self): + """Should accept valid temperature.""" + validate_temperature(0.0) + validate_temperature(0.7) + validate_temperature(1.0) + validate_temperature(2.0) + + def test_accept_none(self): + """Should accept None.""" + validate_temperature(None) + + def test_reject_negative(self): + """Should reject negative values.""" + with pytest.raises(ValueError, match="between 0 and 2"): + validate_temperature(-0.1) + + def test_reject_too_large(self): + """Should reject values > 2.""" + with pytest.raises(ValueError, match="between 0 and 2"): + validate_temperature(2.1) + + def test_reject_non_number(self): + """Should reject non-numeric.""" + with pytest.raises(ValueError, match="number"): + validate_temperature("0.7") # type: ignore + + def test_reject_bool(self): + """bool is an int subclass, so `isinstance(True, (int, float))` is True + and `temperature=True` serialized to the wire as JSON `true`.""" + for bad in (True, False): + with pytest.raises(ValueError, match="bool"): + validate_temperature(bad) # type: ignore + + def test_values_next_to_the_bools_still_pass(self): + """The guard must reject the type, not the neighbouring numbers.""" + validate_temperature(0) + validate_temperature(1) + validate_temperature(1.5) + + +class TestValidateTopP: + def test_accept_valid_values(self): + """Should accept valid top_p.""" + validate_top_p(0.0) + validate_top_p(0.5) + validate_top_p(0.9) + validate_top_p(1.0) + + def test_accept_none(self): + """Should accept None.""" + validate_top_p(None) + + def test_reject_negative(self): + """Should reject negative values.""" + with pytest.raises(ValueError, match="between 0 and 1"): + validate_top_p(-0.1) + + def test_reject_too_large(self): + """Should reject values > 1.""" + with pytest.raises(ValueError, match="between 0 and 1"): + validate_top_p(1.1) + + def test_reject_non_number(self): + """Should reject non-numeric.""" + with pytest.raises(ValueError, match="number"): + validate_top_p("0.9") # type: ignore + + def test_reject_bool(self): + """Same hole as temperature: `top_p=False` reached the gateway as JSON + `false` instead of being caught as the wrong type.""" + for bad in (True, False): + with pytest.raises(ValueError, match="bool"): + validate_top_p(bad) # type: ignore + + def test_values_next_to_the_bools_still_pass(self): + validate_top_p(0) + validate_top_p(1) + validate_top_p(0.5) + + +class TestSanitizeErrorResponse: + def test_extract_safe_fields(self): + """Should extract only safe error fields.""" + result = sanitize_error_response( + { + "error": "User-facing error", + "internal_stack": "/var/app/sensitive.py:123", + "api_key": "sk-secret", + "database_url": "postgres://user:pass@host/db", + } + ) + assert result == {"message": "User-facing error", "code": None} + + def test_include_code_if_present(self): + """Should include code if present.""" + result = sanitize_error_response( + {"error": "Invalid request", "code": "invalid_request_error"} + ) + assert result == {"message": "Invalid request", "code": "invalid_request_error"} + + def test_handle_non_dict(self): + """Should handle non-dict input.""" + assert sanitize_error_response("error") == { + "message": "API request failed", + "code": None, + } + assert sanitize_error_response(None) == { + "message": "API request failed", + "code": None, + } + assert sanitize_error_response(123) == { + "message": "API request failed", + "code": None, + } + + def test_handle_missing_error_field(self): + """Should handle missing error field.""" + result = sanitize_error_response({"something": "else"}) + assert result == {"message": "API request failed", "code": None} + + def test_nested_openai_error_shape(self): + """Should pass through the gateway's OpenAI-compatible nested error.""" + result = sanitize_error_response( + { + "error": { + "message": "Conversation too long — Message @bc1max on Telegram", + "type": "invalid_request_error", + "code": "CONTEXT_LENGTH_EXCEEDED", + "param": None, + }, + "message": "Message @bc1max on Telegram", + "code": "CONTEXT_LENGTH_EXCEEDED", + "debug": "/var/app/handler.py:123 SECRET_KEY=xyz", + } + ) + assert result["message"] == "Conversation too long — Message @bc1max on Telegram" + assert result["code"] == "CONTEXT_LENGTH_EXCEEDED" + assert result["type"] == "invalid_request_error" + # Raw upstream debug text must never be surfaced. + assert "debug" not in result + + def test_nested_error_falls_back_to_top_level_code(self): + """Should use top-level code when the nested object omits it.""" + result = sanitize_error_response( + { + "error": {"message": "Rate limited", "type": "rate_limit_error"}, + "code": "RATE_LIMITED", + } + ) + assert result == { + "message": "Rate limited", + "code": "RATE_LIMITED", + "type": "rate_limit_error", + } + + def test_nested_error_passes_param(self): + """Should pass through the OpenAI `param` field when present.""" + result = sanitize_error_response( + { + "error": { + "message": "Set stream: false", + "type": "invalid_request_error", + "code": "STREAM_UNSUPPORTED", + "param": "stream", + } + } + ) + assert result["param"] == "stream" + + def test_flat_string_error_still_supported(self): + """Should keep supporting the legacy flat string `error` shape.""" + result = sanitize_error_response({"error": "Unknown model: foo. Available models: gpt-5.2"}) + assert result == { + "message": "Unknown model: foo. Available models: gpt-5.2", + "code": None, + } + + +class TestValidateResourceUrl: + def test_allow_matching_domain(self): + """Should allow matching domain.""" + result = validate_resource_url("https://api.blockrun.ai/v1/chat", "https://api.blockrun.ai") + assert result == "https://api.blockrun.ai/v1/chat" + + def test_allow_different_path(self): + """Should allow different path on same domain.""" + result = validate_resource_url( + "https://api.blockrun.ai/v2/models", "https://api.blockrun.ai" + ) + assert result == "https://api.blockrun.ai/v2/models" + + def test_reject_different_domain(self): + """Should reject different domain.""" + result = validate_resource_url("https://malicious.com/steal", "https://api.blockrun.ai") + assert result == "https://api.blockrun.ai/v1/chat/completions" + + def test_reject_different_protocol(self): + """Should reject different protocol.""" + result = validate_resource_url("http://api.blockrun.ai/v1/chat", "https://api.blockrun.ai") + assert result == "https://api.blockrun.ai/v1/chat/completions" + + def test_handle_invalid_url(self): + """Should handle invalid URL format.""" + result = validate_resource_url("not-a-url", "https://api.blockrun.ai") + assert result == "https://api.blockrun.ai/v1/chat/completions" + + def test_reject_subdomain_difference(self): + """Should reject subdomain differences.""" + result = validate_resource_url( + "https://evil.api.blockrun.ai/v1/chat", "https://api.blockrun.ai" + ) + assert result == "https://api.blockrun.ai/v1/chat/completions" diff --git a/tests/unit/test_version_consistency.py b/tests/unit/test_version_consistency.py new file mode 100644 index 0000000..75e5ad5 --- /dev/null +++ b/tests/unit/test_version_consistency.py @@ -0,0 +1,49 @@ +"""The package version must be declared identically in both places. + +Releases bump the version in two files — ``pyproject.toml`` (what PyPI ships) +and ``blockrun_llm/__init__.py`` (what ``blockrun_llm.__version__`` reports). +These drifted once (1.4.6 bumped pyproject but not __init__, so installed +copies under-reported as 1.4.5). This test fails CI if they ever diverge +again, instead of the mismatch shipping silently to PyPI. + +Parsed with a regex rather than tomllib so it runs on Python 3.9 (no stdlib +TOML parser before 3.11) without adding a tomli dependency. +""" + +import re +from pathlib import Path + +import blockrun_llm + +_PYPROJECT = Path(__file__).resolve().parents[2] / "pyproject.toml" + + +def _pyproject_version() -> str: + text = _PYPROJECT.read_text(encoding="utf-8") + # First top-level `version = "..."` under [project] / [tool.poetry] etc. + match = re.search(r'(?m)^\s*version\s*=\s*"([^"]+)"', text) + assert match, "no version declared in pyproject.toml" + return match.group(1) + + +def test_version_matches_pyproject(): + assert blockrun_llm.__version__ == _pyproject_version(), ( + f"version drift: __init__.py={blockrun_llm.__version__!r} " + f"!= pyproject.toml={_pyproject_version()!r} — bump BOTH on release" + ) + + +_VERSION_FILE = Path(__file__).resolve().parents[2] / "VERSION" + + +def test_version_matches_version_file(): + """The VERSION file is the third declaration and drifts the most quietly. + + It sat at 1.4.0 while pyproject.toml and __init__.py were both on 1.7.0, + because the two-way check above cannot see it. + """ + declared = _VERSION_FILE.read_text(encoding="utf-8").strip() + assert declared == _pyproject_version(), ( + f"version drift: VERSION={declared!r} " + f"!= pyproject.toml={_pyproject_version()!r} — bump ALL THREE on release" + ) diff --git a/tests/unit/test_video_params.py b/tests/unit/test_video_params.py new file mode 100644 index 0000000..cfbb641 --- /dev/null +++ b/tests/unit/test_video_params.py @@ -0,0 +1,218 @@ +"""Unit tests for VideoClient.generate() parameter validation and body construction.""" + +import os + +import pytest + +from blockrun_llm import VideoClient +from blockrun_llm.types import VideoResponse + + +@pytest.fixture +def client(): + # Deterministic dummy key — never signs against a live endpoint in unit + # tests; we only exercise local request/response paths. + os.environ.setdefault("BLOCKRUN_WALLET_KEY", "0x" + "11" * 32) + return VideoClient() + + +@pytest.fixture +def captured(client, monkeypatch): + captured = {} + + def fake_submit(body, budget_seconds): + captured["body"] = body + captured["budget"] = budget_seconds + return VideoResponse(created=1, model=body["model"], data=[]) + + monkeypatch.setattr(client, "_submit_and_poll", fake_submit) + return captured + + +def test_first_last_frame_body(client, captured): + client.generate( + "the flower blooms", + model="bytedance/seedance-1.5-pro", + image_url="https://example.com/bud.jpg", + last_frame_url="https://example.com/bloom.jpg", + ) + assert captured["body"]["image_url"] == "https://example.com/bud.jpg" + assert captured["body"]["last_frame_url"] == "https://example.com/bloom.jpg" + + +def test_reference_images_body(client, captured): + urls = ["https://example.com/1.jpg", "https://example.com/2.jpg"] + client.generate( + "the character from image 1 in the city from image 2", + model="bytedance/seedance-2.0", + reference_image_urls=urls, + ) + assert captured["body"]["reference_image_urls"] == urls + assert "image_url" not in captured["body"] + + +def test_token360_passthroughs(client, captured): + client.generate( + "a calm lake at dawn", + model="bytedance/seedance-2.0", + aspect_ratio="16:9", + seed=42, + watermark=False, + return_last_frame=True, + ) + body = captured["body"] + assert body["aspect_ratio"] == "16:9" + assert body["seed"] == 42 + assert body["watermark"] is False + assert body["return_last_frame"] is True + + +def test_last_frame_requires_image_url(client): + with pytest.raises(ValueError, match="requires image_url"): + client.generate("x", last_frame_url="https://example.com/last.jpg") + + +def test_last_frame_excludes_real_face(client): + with pytest.raises(ValueError, match="mutually exclusive"): + client.generate( + "x", + image_url="https://example.com/first.jpg", + last_frame_url="https://example.com/last.jpg", + real_face_asset_id="ta_abc123", + ) + + +def test_reference_images_exclude_other_image_inputs(client): + with pytest.raises(ValueError, match="mutually exclusive"): + client.generate( + "x", + image_url="https://example.com/seed.jpg", + reference_image_urls=["https://example.com/r.jpg"], + ) + with pytest.raises(ValueError, match="mutually exclusive"): + client.generate( + "x", + real_face_asset_id="ta_abc123", + reference_image_urls=["https://example.com/r.jpg"], + ) + + +def test_reference_images_max_nine(client): + with pytest.raises(ValueError, match="at most 9"): + client.generate( + "x", + reference_image_urls=[f"https://example.com/{i}.jpg" for i in range(10)], + ) + + +def test_image_url_and_real_face_still_exclusive(client): + with pytest.raises(ValueError, match="mutually exclusive"): + client.generate( + "x", + image_url="https://example.com/a.jpg", + real_face_asset_id="ta_abc123", + ) + + +# --- input_type ------------------------------------------------------------ +# A declared seed mode the gateway cross-checks against the fields actually +# sent. Only the spelling is validated locally; the match is the gateway's +# call (400, unbilled) so the two can't drift. + + +def test_input_type_forwarded(client, captured): + client.generate( + "the flower blooms", + model="bytedance/seedance-1.5-pro", + image_url="https://example.com/bud.jpg", + last_frame_url="https://example.com/bloom.jpg", + input_type="first_last_frame", + ) + assert captured["body"]["input_type"] == "first_last_frame" + + +def test_input_type_omitted_when_unset(client, captured): + client.generate("a calm lake at dawn") + assert "input_type" not in captured["body"] + + +@pytest.mark.parametrize("value", ["text", "image", "first_last_frame", "reference"]) +def test_input_type_accepts_every_gateway_mode(client, captured, value): + client.generate("x", input_type=value) + assert captured["body"]["input_type"] == value + + +def test_input_type_rejects_unknown_value(client): + with pytest.raises(ValueError, match="input_type must be one of"): + client.generate("x", input_type="img") + + +def test_input_type_mismatch_is_left_to_the_gateway(client, captured): + """Declaring a mode that contradicts the seed fields must still be sent. + + The gateway owns that check and answers 400 before charging; rejecting it + here would fork the inference into a second copy that drifts. + """ + client.generate("x", input_type="image") # no image_url — gateway's call + assert captured["body"]["input_type"] == "image" + + +def test_mixed_references_and_controls_reach_body(client, captured): + client.generate( + "follow the motion", + model="bytedance/seedance-2.0", + reference_image_urls=["https://example.com/person.png"], + reference_videos=[{"url": "https://example.com/motion.mp4"}], + reference_audios=[{"url": "https://example.com/music.mp3"}], + bitrate_mode="high", + safety_identifier="test", + return_last_frame=True, + input_type="reference", + ) + body = captured["body"] + assert body["reference_videos"] == [{"url": "https://example.com/motion.mp4"}] + assert body["reference_audios"] == [{"url": "https://example.com/music.mp3"}] + assert body["reference_image_urls"] == ["https://example.com/person.png"] + assert body["bitrate_mode"] == "high" + assert body["safety_identifier"] == "test" + assert body["input_type"] == "reference" + + +def test_25_reference_limit_and_output_controls(client, captured): + images = ["https://example.com/person.png"] * 30 + client.generate( + "test", model="bytedance/seedance-2.5", reference_image_urls=images, output_format="mov" + ) + assert captured["body"]["reference_image_urls"] == images + assert captured["body"]["output_format"] == "mov" + with pytest.raises(ValueError, match="at most 30"): + client.generate( + "test", model="bytedance/seedance-2.5", reference_image_urls=images + images + ) + client.generate("test", model="bytedance/seedance-1.5-pro", camera_fixed=False) + assert captured["body"]["camera_fixed"] is False + + +def test_reference_media_cannot_be_frame_seeds(client): + with pytest.raises(ValueError, match="mutually exclusive"): + client.generate( + "test", + image_url="https://example.com/frame.png", + reference_videos=[{"url": "https://example.com/motion.mp4"}], + ) + + +def test_last_frame_response_is_not_dropped(): + result = VideoResponse( + created=1, + model="bytedance/seedance-2.0", + data=[ + { + "url": "https://example.com/movie.mp4", + "last_frame_url": "https://example.com/last.png", + "last_frame_backed_up": True, + } + ], + ) + assert result.data[0].last_frame_url == "https://example.com/last.png" + assert result.data[0].last_frame_backed_up is True diff --git a/tests/unit/test_wallet_selection.py b/tests/unit/test_wallet_selection.py new file mode 100644 index 0000000..5445494 --- /dev/null +++ b/tests/unit/test_wallet_selection.py @@ -0,0 +1,294 @@ +"""Regression tests for canonical wallet selection.""" + +import os +from pathlib import Path + +import pytest +from eth_account import Account + +from blockrun_llm import solana_wallet, wallet + +PROVIDER_KEY = "0x" + "2" * 64 +CANONICAL_KEY = "0x" + "1" * 64 +LEGACY_KEY = "0x" + "3" * 64 + + +def _fake_home(monkeypatch, home: Path) -> None: + """Point Path.home() at a temp dir so scan_wallets() reads real fixtures.""" + monkeypatch.setattr(Path, "home", classmethod(lambda cls: home)) + + +def _write_provider_wallet(home: Path, filename: str = "wallet.json") -> Path: + """Create a provider wallet whose "address" field is a lie.""" + provider_dir = home / ".agentcash" + provider_dir.mkdir(exist_ok=True) + target = provider_dir / filename + target.write_text('{"privateKey":"' + PROVIDER_KEY + '","address":"0xNotTheRealAddress"}') + return target + + +def _blockrun_dir(home: Path) -> Path: + blockrun_dir = home / ".blockrun" + blockrun_dir.mkdir(exist_ok=True) + return blockrun_dir + + +def test_load_wallet_prefers_blockrun_session_over_provider_wallet(monkeypatch, tmp_path): + """A newer wallet.json from another app must not replace BlockRun's wallet.""" + blockrun_dir = tmp_path / ".blockrun" + blockrun_dir.mkdir() + canonical_file = blockrun_dir / ".session" + canonical_key = "0x" + "1" * 64 + canonical_file.write_text(canonical_key) + + provider_dir = tmp_path / ".agentcash" + provider_dir.mkdir() + (provider_dir / "wallet.json").write_text( + '{"privateKey":"0x' + "2" * 64 + '","address":"0xprovider"}' + ) + + monkeypatch.setattr(wallet, "WALLET_DIR", blockrun_dir) + monkeypatch.setattr(wallet, "WALLET_FILE", canonical_file) + monkeypatch.setattr( + wallet, + "scan_wallets", + lambda: (_ for _ in ()).throw(AssertionError("automatic scan must not run")), + ) + + assert wallet.load_wallet() == canonical_key + + +def test_load_solana_wallet_prefers_blockrun_session_over_provider_wallet(monkeypatch, tmp_path): + """A provider Solana wallet must not replace BlockRun's active wallet.""" + blockrun_dir = tmp_path / ".blockrun" + blockrun_dir.mkdir() + canonical_file = blockrun_dir / ".solana-session" + canonical_key = "canonical-solana-key" + canonical_file.write_text(canonical_key) + + monkeypatch.setattr(solana_wallet, "WALLET_DIR", blockrun_dir) + monkeypatch.setattr(solana_wallet, "SOLANA_WALLET_FILE", canonical_file) + monkeypatch.setattr( + solana_wallet, + "scan_solana_wallets", + lambda: (_ for _ in ()).throw(AssertionError("automatic scan must not run")), + ) + + assert solana_wallet.load_solana_wallet() == canonical_key + + +def test_get_or_create_wallet_does_not_adopt_provider_wallet(monkeypatch, tmp_path): + """get_or_create_wallet() is what callers hit — it must not scan either. + + load_wallet() and get_or_create_wallet() carry separate resolution logic, + so covering only load_wallet() would let the bug return through the door + users actually walk through. + """ + monkeypatch.delenv("BLOCKRUN_WALLET_KEY", raising=False) + monkeypatch.delenv("BASE_CHAIN_WALLET_KEY", raising=False) + _fake_home(monkeypatch, tmp_path) + blockrun_dir = _blockrun_dir(tmp_path) + _write_provider_wallet(tmp_path) + + monkeypatch.setattr(wallet, "WALLET_DIR", blockrun_dir) + monkeypatch.setattr(wallet, "WALLET_FILE", blockrun_dir / ".session") + + # The provider wallet is genuinely on disk and discoverable... + assert len(wallet.scan_wallets()) == 1 + + address, key, is_new = wallet.get_or_create_wallet() + + # ...but a brand new wallet is minted instead of adopting it. + assert is_new is True + assert key != PROVIDER_KEY + assert address != Account.from_key(PROVIDER_KEY).address + + +def test_get_or_create_wallet_honours_legacy_wallet_key(monkeypatch, tmp_path): + """A funded ~/.blockrun/wallet.key must not be replaced by a new wallet. + + The TypeScript SDK resolves this file via loadWallet(); Python must match, + otherwise a legacy user silently loses access to their funds. + """ + monkeypatch.delenv("BLOCKRUN_WALLET_KEY", raising=False) + monkeypatch.delenv("BASE_CHAIN_WALLET_KEY", raising=False) + _fake_home(monkeypatch, tmp_path) + blockrun_dir = _blockrun_dir(tmp_path) + (blockrun_dir / "wallet.key").write_text(LEGACY_KEY) + + monkeypatch.setattr(wallet, "WALLET_DIR", blockrun_dir) + monkeypatch.setattr(wallet, "WALLET_FILE", blockrun_dir / ".session") + + address, key, is_new = wallet.get_or_create_wallet() + + assert is_new is False + assert key == LEGACY_KEY + assert address == Account.from_key(LEGACY_KEY).address + + +def test_provider_wallet_cannot_hijack_even_when_written_last(monkeypatch, tmp_path): + """The core invariant: another app must never take over the active wallet. + + Discovery sorts by modification time, so the hijack is simply "write a + wallet.json last". Here the provider wallet is the newest file on disk and + claims a plausible address; BlockRun's own session must still win. + """ + monkeypatch.delenv("BLOCKRUN_WALLET_KEY", raising=False) + monkeypatch.delenv("BASE_CHAIN_WALLET_KEY", raising=False) + _fake_home(monkeypatch, tmp_path) + blockrun_dir = _blockrun_dir(tmp_path) + (blockrun_dir / ".session").write_text(CANONICAL_KEY) + + monkeypatch.setattr(wallet, "WALLET_DIR", blockrun_dir) + monkeypatch.setattr(wallet, "WALLET_FILE", blockrun_dir / ".session") + + # Provider writes after us, from two different directories. + provider = _write_provider_wallet(tmp_path) + second_dir = tmp_path / ".someotherprovider" + second_dir.mkdir() + (second_dir / "wallet.json").write_text( + '{"privateKey":"' + LEGACY_KEY + '","address":"0xAlsoNotOurs"}' + ) + os.utime(provider, (2**31, 2**31)) + + assert len(wallet.scan_wallets()) == 2 + + address, key, is_new = wallet.get_or_create_wallet() + + assert key == CANONICAL_KEY + assert address == Account.from_key(CANONICAL_KEY).address + assert is_new is False + + +def test_import_wallet_adopts_by_derived_address(monkeypatch, tmp_path): + """The deliberate migration path: opt in to a discovered wallet's funds.""" + monkeypatch.delenv("BLOCKRUN_WALLET_KEY", raising=False) + monkeypatch.delenv("BASE_CHAIN_WALLET_KEY", raising=False) + _fake_home(monkeypatch, tmp_path) + blockrun_dir = _blockrun_dir(tmp_path) + (blockrun_dir / ".session").write_text(CANONICAL_KEY) + _write_provider_wallet(tmp_path) + + monkeypatch.setattr(wallet, "WALLET_DIR", blockrun_dir) + monkeypatch.setattr(wallet, "WALLET_FILE", blockrun_dir / ".session") + + provider_address = Account.from_key(PROVIDER_KEY).address + adopted = wallet.import_wallet(provider_address) + + assert adopted == provider_address + # It is now the active wallet, through the normal selection path. + address, key, is_new = wallet.get_or_create_wallet() + assert key == PROVIDER_KEY + assert address == provider_address + assert is_new is False + + +def test_import_wallet_backs_up_the_replaced_wallet(monkeypatch, tmp_path): + """Adopting must not strand funds sitting in the outgoing wallet.""" + _fake_home(monkeypatch, tmp_path) + blockrun_dir = _blockrun_dir(tmp_path) + (blockrun_dir / ".session").write_text(CANONICAL_KEY) + _write_provider_wallet(tmp_path) + + monkeypatch.setattr(wallet, "WALLET_DIR", blockrun_dir) + monkeypatch.setattr(wallet, "WALLET_FILE", blockrun_dir / ".session") + + wallet.import_wallet(Account.from_key(PROVIDER_KEY).address) + + backups = list(blockrun_dir.glob(".session.backup-*")) + assert len(backups) == 1 + assert backups[0].read_text() == CANONICAL_KEY + + +def test_import_wallet_rejects_an_address_no_discovered_key_controls(monkeypatch, tmp_path): + """A wallet file claiming someone else's address cannot be adopted by it. + + This is the whole point of matching on the derived address: a planted file + that names an attacker's receiving address must not be selectable by it. + """ + _fake_home(monkeypatch, tmp_path) + blockrun_dir = _blockrun_dir(tmp_path) + (blockrun_dir / ".session").write_text(CANONICAL_KEY) + _write_provider_wallet(tmp_path) # claims "0xNotTheRealAddress" + + monkeypatch.setattr(wallet, "WALLET_DIR", blockrun_dir) + monkeypatch.setattr(wallet, "WALLET_FILE", blockrun_dir / ".session") + + with pytest.raises(ValueError, match="No discovered wallet controls"): + wallet.import_wallet("0xNotTheRealAddress") + + # The active wallet is untouched. + assert (blockrun_dir / ".session").read_text() == CANONICAL_KEY + assert not list(blockrun_dir.glob(".session.backup-*")) + + +def test_list_discovered_wallets_never_returns_secrets(monkeypatch, tmp_path): + """Safe to print: derived addresses and provenance, no keys.""" + _fake_home(monkeypatch, tmp_path) + _blockrun_dir(tmp_path) + _write_provider_wallet(tmp_path) + + listed = wallet.list_discovered_wallets() + + assert len(listed) == 1 + assert listed[0]["address"] == Account.from_key(PROVIDER_KEY).address + assert ".agentcash" in listed[0]["source"] + assert "private_key" not in listed[0] + assert PROVIDER_KEY not in str(listed) + + +def test_migration_notice_derives_address_from_key_not_file_claim(monkeypatch, tmp_path): + """The notice must name the address the discovered key actually controls.""" + _fake_home(monkeypatch, tmp_path) + _blockrun_dir(tmp_path) + _write_provider_wallet(tmp_path) + + new_address = Account.from_key(CANONICAL_KEY).address + notice = wallet.format_wallet_migration_notice(new_address) + + assert notice is not None + # The real address derived from the key, never the file's bogus claim. + assert Account.from_key(PROVIDER_KEY).address in notice + assert "0xNotTheRealAddress" not in notice + assert new_address in notice + # Never leak the discovered private key. + assert PROVIDER_KEY not in notice + + +def test_migration_notice_is_silent_when_nothing_discovered(monkeypatch, tmp_path): + """No provider wallets means no scary notice.""" + _fake_home(monkeypatch, tmp_path) + _blockrun_dir(tmp_path) + + assert wallet.format_wallet_migration_notice("0xabc") is None + + +def test_solana_migration_notice_lists_discovered_wallets(monkeypatch, tmp_path): + """Solana counterpart surfaces discovered wallets the same way.""" + # CI runs this file on Python 3.9, where the solana extra is not installed. + pytest.importorskip("solders") + _fake_home(monkeypatch, tmp_path) + _blockrun_dir(tmp_path) + provider_dir = tmp_path / ".agentcash" + provider_dir.mkdir(exist_ok=True) + + discovered = solana_wallet.create_solana_wallet() + (provider_dir / "solana-wallet.json").write_text( + '{"privateKey":"' + discovered["private_key"] + '","address":"NotTheRealAddress"}' + ) + + notice = solana_wallet.format_solana_wallet_migration_notice("NewWalletAddress") + + assert notice is not None + assert discovered["address"] in notice + assert "NotTheRealAddress" not in notice + assert discovered["private_key"] not in notice + + +def test_solana_migration_notice_is_silent_when_nothing_discovered(monkeypatch, tmp_path): + """No provider Solana wallets means no notice.""" + pytest.importorskip("solders") + _fake_home(monkeypatch, tmp_path) + _blockrun_dir(tmp_path) + + assert solana_wallet.format_solana_wallet_migration_notice("NewWalletAddress") is None diff --git a/tests/unit/test_x402.py b/tests/unit/test_x402.py index 656b48b..0ccb26d 100644 --- a/tests/unit/test_x402.py +++ b/tests/unit/test_x402.py @@ -1,223 +1,320 @@ -"""Unit tests for x402 payment protocol.""" - -import pytest -import base64 -import json -from blockrun_llm.x402 import ( - create_nonce, - create_payment_payload, - parse_payment_required, - extract_payment_details, -) -from ..helpers import TEST_ACCOUNT, TEST_RECIPIENT - - -class TestCreateNonce: - def test_nonce_format(self): - """Should generate nonce with correct format.""" - nonce = create_nonce() - - assert nonce.startswith("0x") - assert len(nonce) == 66 # 0x + 64 hex chars - - def test_nonce_uniqueness(self): - """Should generate unique nonces.""" - nonce1 = create_nonce() - nonce2 = create_nonce() - - assert nonce1 != nonce2 - - def test_nonce_is_hex(self): - """Should contain only hex characters.""" - nonce = create_nonce() - # Remove 0x prefix and check if valid hex - hex_part = nonce[2:] - assert all(c in "0123456789abcdef" for c in hex_part.lower()) - - -class TestCreatePaymentPayload: - def test_create_valid_payload(self): - """Should create valid payment payload.""" - payload = create_payment_payload( - account=TEST_ACCOUNT, - recipient=TEST_RECIPIENT, - amount="1000000", - network="eip155:8453", - ) - - assert isinstance(payload, str) - - # Decode and verify structure - decoded = json.loads(base64.b64decode(payload)) - assert decoded["x402Version"] == 2 - assert "payload" in decoded - assert "signature" in decoded["payload"] - assert decoded["payload"]["signature"].startswith("0x") - - def test_payload_includes_authorization(self): - """Should include authorization details.""" - payload = create_payment_payload( - account=TEST_ACCOUNT, - recipient=TEST_RECIPIENT, - amount="1000000", - ) - - decoded = json.loads(base64.b64decode(payload)) - auth = decoded["payload"]["authorization"] - - assert auth["from"] == TEST_ACCOUNT.address - assert auth["to"] == TEST_RECIPIENT - assert auth["value"] == "1000000" - assert "validAfter" in auth - assert "validBefore" in auth - assert "nonce" in auth - - def test_payload_includes_resource_info(self): - """Should include resource information.""" - payload = create_payment_payload( - account=TEST_ACCOUNT, - recipient=TEST_RECIPIENT, - amount="1000000", - resource_url="https://api.blockrun.ai/v1/test", - resource_description="Test Resource", - ) - - decoded = json.loads(base64.b64decode(payload)) - assert decoded["resource"]["url"] == "https://api.blockrun.ai/v1/test" - assert decoded["resource"]["description"] == "Test Resource" - - def test_payload_time_windows(self): - """Should set valid time windows.""" - import time - - before = int(time.time()) - payload = create_payment_payload( - account=TEST_ACCOUNT, recipient=TEST_RECIPIENT, amount="1000000" - ) - after = int(time.time()) - - decoded = json.loads(base64.b64decode(payload)) - auth = decoded["payload"]["authorization"] - - # Valid after should be in the past (allows clock skew) - assert int(auth["validAfter"]) < before - - # Valid before should be in the future - assert int(auth["validBefore"]) > after - - def test_custom_timeout(self): - """Should use custom max timeout.""" - payload = create_payment_payload( - account=TEST_ACCOUNT, - recipient=TEST_RECIPIENT, - amount="1000000", - max_timeout_seconds=600, - ) - - decoded = json.loads(base64.b64decode(payload)) - assert decoded["accepted"]["maxTimeoutSeconds"] == 600 - - -class TestParsePaymentRequired: - def test_parse_valid_header(self): - """Should parse valid payment required header.""" - data = { - "x402Version": 2, - "accepts": [ - { - "scheme": "exact", - "network": "eip155:8453", - "amount": "1000000", - "asset": "0x833589fCD6eDb6E08f4c7C32D4f71b54bdA02913", - "payTo": TEST_RECIPIENT, - "maxTimeoutSeconds": 300, - } - ], - } - - encoded = base64.b64encode(json.dumps(data).encode()).decode() - result = parse_payment_required(encoded) - - assert result["x402Version"] == 2 - assert len(result["accepts"]) == 1 - - def test_invalid_base64(self): - """Should raise ValueError on invalid base64.""" - with pytest.raises(ValueError, match="invalid format"): - parse_payment_required("invalid!!!") - - def test_invalid_json(self): - """Should raise ValueError on invalid JSON.""" - with pytest.raises(ValueError, match="invalid format"): - parse_payment_required(base64.b64encode(b"not json").decode()) - - -class TestExtractPaymentDetails: - def test_extract_details(self): - """Should extract payment details.""" - payment_required = { - "x402Version": 2, - "accepts": [ - { - "scheme": "exact", - "network": "eip155:8453", - "amount": "1000000", - "asset": "0x833589fCD6eDb6E08f4c7C32D4f71b54bdA02913", - "payTo": TEST_RECIPIENT, - "maxTimeoutSeconds": 300, - } - ], - } - - details = extract_payment_details(payment_required) - - assert details["amount"] == "1000000" - assert details["recipient"] == TEST_RECIPIENT - assert details["network"] == "eip155:8453" - assert details["maxTimeoutSeconds"] == 300 - - def test_empty_accepts(self): - """Should raise ValueError on empty accepts.""" - with pytest.raises(ValueError, match="No payment options"): - extract_payment_details({"x402Version": 2, "accepts": []}) - - def test_default_timeout(self): - """Should use default timeout if not specified.""" - payment_required = { - "x402Version": 2, - "accepts": [ - { - "scheme": "exact", - "network": "eip155:8453", - "amount": "1000000", - "asset": "0x833589fCD6eDb6E08f4c7C32D4f71b54bdA02913", - "payTo": TEST_RECIPIENT, - # maxTimeoutSeconds not specified - } - ], - } - - details = extract_payment_details(payment_required) - assert details["maxTimeoutSeconds"] == 300 - - def test_include_resource(self): - """Should include resource if present.""" - payment_required = { - "x402Version": 2, - "accepts": [ - { - "scheme": "exact", - "network": "eip155:8453", - "amount": "1000000", - "asset": "0x833589fCD6eDb6E08f4c7C32D4f71b54bdA02913", - "payTo": TEST_RECIPIENT, - } - ], - "resource": { - "url": "https://api.blockrun.ai/test", - "description": "Test", - }, - } - - details = extract_payment_details(payment_required) - assert details["resource"]["url"] == "https://api.blockrun.ai/test" +"""Unit tests for x402 payment protocol.""" + +import base64 +import json + +import pytest + +from blockrun_llm.x402 import ( + create_nonce, + create_payment_payload, + extract_payment_details, + parse_payment_required, +) + +from ..helpers import TEST_ACCOUNT, TEST_RECIPIENT + + +class TestCreateNonce: + def test_nonce_format(self): + """Should generate nonce with correct format.""" + nonce = create_nonce() + + assert nonce.startswith("0x") + assert len(nonce) == 66 # 0x + 64 hex chars + + def test_nonce_uniqueness(self): + """Should generate unique nonces.""" + nonce1 = create_nonce() + nonce2 = create_nonce() + + assert nonce1 != nonce2 + + def test_nonce_is_hex(self): + """Should contain only hex characters.""" + nonce = create_nonce() + # Remove 0x prefix and check if valid hex + hex_part = nonce[2:] + assert all(c in "0123456789abcdef" for c in hex_part.lower()) + + +class TestCreatePaymentPayload: + def test_create_valid_payload(self): + """Should create valid payment payload.""" + payload = create_payment_payload( + account=TEST_ACCOUNT, + recipient=TEST_RECIPIENT, + amount="1000000", + network="eip155:8453", + ) + + assert isinstance(payload, str) + + # Decode and verify structure + decoded = json.loads(base64.b64decode(payload)) + assert decoded["x402Version"] == 2 + assert "payload" in decoded + assert "signature" in decoded["payload"] + assert decoded["payload"]["signature"].startswith("0x") + + def test_payload_includes_authorization(self): + """Should include authorization details.""" + payload = create_payment_payload( + account=TEST_ACCOUNT, + recipient=TEST_RECIPIENT, + amount="1000000", + ) + + decoded = json.loads(base64.b64decode(payload)) + auth = decoded["payload"]["authorization"] + + assert auth["from"] == TEST_ACCOUNT.address + assert auth["to"] == TEST_RECIPIENT + assert auth["value"] == "1000000" + assert "validAfter" in auth + assert "validBefore" in auth + assert "nonce" in auth + + def test_payload_attaches_builder_code_service_code(self): + """Should tag every payment with the BlockRun service code (s).""" + payload = create_payment_payload( + account=TEST_ACCOUNT, + recipient=TEST_RECIPIENT, + amount="1000000", + ) + + decoded = json.loads(base64.b64decode(payload)) + assert decoded["extensions"]["builder-code"]["info"]["s"] == ["blockrun"] + + def test_payload_preserves_echoed_app_code(self): + """Should keep the server-echoed app code (a) when adding service code (s).""" + payload = create_payment_payload( + account=TEST_ACCOUNT, + recipient=TEST_RECIPIENT, + amount="1000000", + extensions={"builder-code": {"info": {"a": "blockrun"}}}, + ) + + decoded = json.loads(base64.b64decode(payload)) + info = decoded["extensions"]["builder-code"]["info"] + assert info["a"] == "blockrun" + assert info["s"] == ["blockrun"] + + def test_payload_includes_resource_info(self): + """Should include resource information.""" + payload = create_payment_payload( + account=TEST_ACCOUNT, + recipient=TEST_RECIPIENT, + amount="1000000", + resource_url="https://api.blockrun.ai/v1/test", + resource_description="Test Resource", + ) + + decoded = json.loads(base64.b64decode(payload)) + assert decoded["resource"]["url"] == "https://api.blockrun.ai/v1/test" + assert decoded["resource"]["description"] == "Test Resource" + + def test_payload_time_windows(self): + """Should set valid time windows.""" + import time + + before = int(time.time()) + payload = create_payment_payload( + account=TEST_ACCOUNT, recipient=TEST_RECIPIENT, amount="1000000" + ) + after = int(time.time()) + + decoded = json.loads(base64.b64decode(payload)) + auth = decoded["payload"]["authorization"] + + # Valid after should be in the past (allows clock skew) + assert int(auth["validAfter"]) < before + + # Valid before should be in the future + assert int(auth["validBefore"]) > after + + def test_custom_timeout(self): + """Should use custom max timeout.""" + payload = create_payment_payload( + account=TEST_ACCOUNT, + recipient=TEST_RECIPIENT, + amount="1000000", + max_timeout_seconds=600, + ) + + decoded = json.loads(base64.b64decode(payload)) + assert decoded["accepted"]["maxTimeoutSeconds"] == 600 + + +class TestParsePaymentRequired: + def test_parse_valid_header(self): + """Should parse valid payment required header.""" + data = { + "x402Version": 2, + "accepts": [ + { + "scheme": "exact", + "network": "eip155:8453", + "amount": "1000000", + "asset": "0x833589fCD6eDb6E08f4c7C32D4f71b54bdA02913", + "payTo": TEST_RECIPIENT, + "maxTimeoutSeconds": 300, + } + ], + } + + encoded = base64.b64encode(json.dumps(data).encode()).decode() + result = parse_payment_required(encoded) + + assert result["x402Version"] == 2 + assert len(result["accepts"]) == 1 + + def test_invalid_base64(self): + """Should raise ValueError on invalid base64.""" + with pytest.raises(ValueError, match="invalid format"): + parse_payment_required("invalid!!!") + + def test_invalid_json(self): + """Should raise ValueError on invalid JSON.""" + with pytest.raises(ValueError, match="invalid format"): + parse_payment_required(base64.b64encode(b"not json").decode()) + + +class TestExtractPaymentDetails: + def test_extract_details(self): + """Should extract payment details.""" + payment_required = { + "x402Version": 2, + "accepts": [ + { + "scheme": "exact", + "network": "eip155:8453", + "amount": "1000000", + "asset": "0x833589fCD6eDb6E08f4c7C32D4f71b54bdA02913", + "payTo": TEST_RECIPIENT, + "maxTimeoutSeconds": 300, + } + ], + } + + details = extract_payment_details(payment_required) + + assert details["amount"] == "1000000" + assert details["recipient"] == TEST_RECIPIENT + assert details["network"] == "eip155:8453" + assert details["maxTimeoutSeconds"] == 300 + + def test_empty_accepts(self): + """Should raise ValueError on empty accepts.""" + with pytest.raises(ValueError, match="No payment options"): + extract_payment_details({"x402Version": 2, "accepts": []}) + + def test_default_timeout(self): + """Should use default timeout if not specified.""" + payment_required = { + "x402Version": 2, + "accepts": [ + { + "scheme": "exact", + "network": "eip155:8453", + "amount": "1000000", + "asset": "0x833589fCD6eDb6E08f4c7C32D4f71b54bdA02913", + "payTo": TEST_RECIPIENT, + # maxTimeoutSeconds not specified + } + ], + } + + details = extract_payment_details(payment_required) + assert details["maxTimeoutSeconds"] == 300 + + def test_include_resource(self): + """Should include resource if present.""" + payment_required = { + "x402Version": 2, + "accepts": [ + { + "scheme": "exact", + "network": "eip155:8453", + "amount": "1000000", + "asset": "0x833589fCD6eDb6E08f4c7C32D4f71b54bdA02913", + "payTo": TEST_RECIPIENT, + } + ], + "resource": { + "url": "https://api.blockrun.ai/test", + "description": "Test", + }, + } + + details = extract_payment_details(payment_required) + assert details["resource"]["url"] == "https://api.blockrun.ai/test" + + +class TestSolanaX402SdkIntegration: + """Tests for Solana x402 SDK integration.""" + + USDC_SOLANA = "EPjFWdd5AufqSSqeM2qN1xzybapC8G4wEGGkZwyTDt1v" + TOKEN_PROGRAM_ID = "TokenkegQfeZyiNwAJbNbGKPFXCWuBvf9Ss623VQ5DA" + TEST_FEE_PAYER = "2wKupLR9q6wXYppw8Gr2NvWxKBUqm4PPJKkQfoxHDBg4" + TEST_SOL_RECIPIENT = "AQqnMFBwGZEoti85aTVRy8XYpKrho7GaMDx9ZB3CEeKA" + + def test_decode_solana_payment_required(self): + """Should decode a Solana 402 PaymentRequired header.""" + from x402.http.utils import decode_payment_required_header + + data = { + "x402Version": 2, + "accepts": [ + { + "scheme": "exact", + "network": "solana:5eykt4UsFv8P8NJdTREpY1vzqKqZKvdp", + "amount": "1000", + "asset": self.USDC_SOLANA, + "payTo": self.TEST_SOL_RECIPIENT, + "maxTimeoutSeconds": 300, + "extra": {"feePayer": self.TEST_FEE_PAYER}, + } + ], + } + encoded = base64.b64encode(json.dumps(data).encode()).decode() + result = decode_payment_required_header(encoded) + + assert result.x402_version == 2 + assert len(result.accepts) == 1 + assert str(result.accepts[0].network).startswith("solana:") + assert result.accepts[0].pay_to == self.TEST_SOL_RECIPIENT + assert result.accepts[0].amount == "1000" + assert result.accepts[0].extra["feePayer"] == self.TEST_FEE_PAYER + + def test_keypair_signer_address(self): + """KeypairSigner should derive correct public key from bs58 secret.""" + from solders.keypair import Keypair + from x402.mechanisms.svm import KeypairSigner + + # Generate a valid keypair and get its base58 representation + kp = Keypair() + expected_address = str(kp.pubkey()) + + signer = KeypairSigner.from_base58(str(kp)) + assert signer.address == expected_address + + def test_ata_derivation_uses_correct_program_id(self): + """ATA derivation must use the correct Associated Token Program ID.""" + from x402.mechanisms.svm import derive_ata + + # Known wallet -> known USDC ATA (verified on-chain) + owner = "CtJTYWPQSL5jw9B2JRHmpQjYCSSgUX3LRvmMBhq55HmQ" + expected_ata = "HZPPxg9ZyoHu4f2pj5uEEXsArLA2rnL9FtDgC8rrAp5Q" + + result = derive_ata(owner, self.USDC_SOLANA, self.TOKEN_PROGRAM_ID) + assert result == expected_ata + + def test_is_solana_network(self): + """Should correctly identify Solana networks.""" + from blockrun_llm.x402 import is_solana_network + + assert is_solana_network("solana:5eykt4UsFv8P8NJdTREpY1vzqKqZKvdp") + assert is_solana_network("solana:EtWTRABZaYq6iMfeYKouRu166VU2xqa1") + assert not is_solana_network("eip155:8453") + assert not is_solana_network("base-sepolia") diff --git a/tests/unit/test_x402_evm_networks.py b/tests/unit/test_x402_evm_networks.py new file mode 100644 index 0000000..31cf774 --- /dev/null +++ b/tests/unit/test_x402_evm_networks.py @@ -0,0 +1,113 @@ +"""The EIP-712 domain a payment is signed against follows the 402's network. + +Until 2.x's Arc release the chain table knew Base and Base Sepolia and fell +back to Base for anything else, while `asset` and `extra` were taken from the +402 as given. Against arc.blockrun.ai (eip155:5042, USDC at 0x3600…, domain +name "USDC") that produced a signature over chainId 8453 with Arc's contract +— invalid; the facilitator recovers a different signer and answers 401 after +the SDK has reported a payment. + +The 402 now SELECTS a network from the SDK's own table, which supplies the +chainId, the USDC address and the domain; a hostile 402's `extra` cannot +steer a signature onto another contract, an unknown network is refused +naming what is supported, and a 402 whose `asset` is not that network's USDC +is refused before anything is signed. Mirrors @blockrun/llm 3.16.0. +""" + +import base64 +import json + +import pytest +from eth_account import Account +from eth_account.messages import encode_typed_data + +from blockrun_llm.x402 import EVM_NETWORKS, create_payment_payload, evm_network + +from ..helpers import TEST_ACCOUNT, TEST_RECIPIENT + +TYPES = { + "TransferWithAuthorization": [ + {"name": "from", "type": "address"}, + {"name": "to", "type": "address"}, + {"name": "value", "type": "uint256"}, + {"name": "validAfter", "type": "uint256"}, + {"name": "validBefore", "type": "uint256"}, + {"name": "nonce", "type": "bytes32"}, + ], +} + + +def sign_and_decode(network: str, **kwargs) -> dict: + payload = create_payment_payload( + account=TEST_ACCOUNT, recipient=TEST_RECIPIENT, amount="2000", network=network, **kwargs + ) + return json.loads(base64.b64decode(payload)) + + +def recovered_signer(decoded: dict, domain: dict) -> str: + a = decoded["payload"]["authorization"] + message = { + "from": a["from"], + "to": a["to"], + "value": int(a["value"]), + "validAfter": int(a["validAfter"]), + "validBefore": int(a["validBefore"]), + "nonce": bytes.fromhex(a["nonce"][2:]), + } + signable = encode_typed_data(domain_data=domain, message_types=TYPES, message_data=message) + return Account.recover_message(signable, signature=decoded["payload"]["signature"]) + + +class TestSignedDomainFollowsNetwork: + def test_knows_arc_base_and_base_sepolia(self): + arc = evm_network("eip155:5042") + assert arc["chain_id"] == 5042 + assert arc["usdc"] == "0x3600000000000000000000000000000000000000" + assert arc["domain"]["name"] == "USDC" + base = evm_network("eip155:8453") + assert base["chain_id"] == 8453 + assert base["domain"]["name"] == "USD Coin" + assert evm_network("eip155:84532")["domain"]["name"] == "USDC" + # The old alias still resolves. + assert evm_network("base-sepolia")["chain_id"] == 84532 + + def test_arc_payment_signs_arc_domain_not_base(self): + decoded = sign_and_decode("eip155:5042") + assert ( + recovered_signer(decoded, EVM_NETWORKS["eip155:5042"]["domain"]) == TEST_ACCOUNT.address + ) + assert ( + recovered_signer(decoded, EVM_NETWORKS["eip155:8453"]["domain"]) != TEST_ACCOUNT.address + ) + assert decoded["accepted"]["network"] == "eip155:5042" + assert decoded["accepted"]["asset"] == "0x3600000000000000000000000000000000000000" + assert decoded["accepted"]["extra"] == {"name": "USDC", "version": "2"} + + def test_base_payment_unchanged(self): + decoded = sign_and_decode("eip155:8453") + assert ( + recovered_signer(decoded, EVM_NETWORKS["eip155:8453"]["domain"]) == TEST_ACCOUNT.address + ) + assert decoded["accepted"]["asset"] == "0x833589fCD6eDb6E08f4c7C32D4f71b54bdA02913" + assert decoded["accepted"]["extra"] == {"name": "USD Coin", "version": "2"} + + def test_unknown_network_is_refused_not_signed_as_base(self): + with pytest.raises(ValueError, match="eip155:1") as e: + sign_and_decode("eip155:1") + assert "eip155:5042" in str(e.value) # names what it does know + + def test_asset_not_that_networks_usdc_is_refused(self): + with pytest.raises(ValueError, match="(?i)asset"): + sign_and_decode("eip155:5042", asset="0x833589fCD6eDb6E08f4c7C32D4f71b54bdA02913") + # Case-insensitive on the address; the gateway checksums, wallets often do not. + ok = sign_and_decode( + "eip155:5042", asset="0x3600000000000000000000000000000000000000".lower() + ) + assert ok["accepted"]["asset"] == "0x3600000000000000000000000000000000000000" + + def test_402_extra_is_ignored_for_the_domain(self): + decoded = sign_and_decode("eip155:5042", extra={"name": "USD Coin", "version": "9"}) + assert ( + recovered_signer(decoded, EVM_NETWORKS["eip155:5042"]["domain"]) == TEST_ACCOUNT.address + ) + assert decoded["accepted"]["extra"] == {"name": "USDC", "version": "2"}