From 418a96f456e5eda36791a306e3b48e7e56861ee4 Mon Sep 17 00:00:00 2001 From: MK Date: Sun, 27 Sep 2026 22:30:47 -0400 Subject: [PATCH 1/4] feat(llm): PDF document input to document-capable models (#255 Phase 3) Forge now forwards an inbound PDF file part to the model as a native document block, alongside the Phase 2 image path. Builds on the Phase 2 gate + DoS controls. - Capability: ModelSupportsPDF (Anthropic Claude 3.5+; the bare claude-3.0 models are correctly excluded) + IsDocumentMIME (application/pdf). - Anthropic provider: anthropicBlocksFromParts emits a {type:document, source:{base64,application/pdf}} block for document parts; the shared source struct is renamed anthropicMediaSource (image + document). - Projection (a2aMessageToLLM): PDF file parts -> ContentPartDocument, next to the image projection; text stays the text-of-record for the scanners. - Gate (checkInboundMedia): PDFs pass on a document-capable model; a PDF on a non-document model rejects (model_not_document_capable); non-PDF documents and other unsupported types reject (unsupported_media_type). Per-document byte cap (32 MiB) + count (5) + a %PDF- format sniff, with reason codes too_many_document_parts / document_limit_exceeded. No new dependency (native document blocks, no extraction). OpenAI Responses input_file, Gemini documents, and text-extraction fallback for non-native models are deferred follow-ups. Tests: ModelSupportsPDF / IsDocumentMIME; CheckDocumentLimits (valid/empty/ oversized/mislabeled); Anthropic document-block serialization; projection (PDF->document, unsupported not projected); gate matrix (pdf accept on claude, reject on gpt-4o, mislabeled, unsupported type). Docs: runtime-engine multimodal + DoS table, audit reason codes, forge.md skill (synced). --- .claude/skills/forge.md | 2 +- docs/core-concepts/runtime-engine.md | 15 +++- docs/security/audit-logging.md | 2 +- forge-cli/internal/surface/knowledge/forge.md | 2 +- forge-cli/runtime/media_gate_test.go | 33 +++++-- forge-cli/runtime/runner.go | 90 ++++++++++++------- forge-core/llm/providers/anthropic.go | 25 ++++-- forge-core/llm/providers/multimodal_test.go | 31 +++++++ forge-core/runtime/loop.go | 41 +++++---- forge-core/runtime/loop_projection_test.go | 36 ++++++-- forge-core/runtime/media_limits.go | 27 ++++++ forge-core/runtime/media_limits_test.go | 25 ++++++ forge-core/runtime/model_capabilities.go | 32 ++++++- forge-core/runtime/model_capabilities_test.go | 26 ++++++ 14 files changed, 313 insertions(+), 74 deletions(-) diff --git a/.claude/skills/forge.md b/.claude/skills/forge.md index c469f3cc..acf38d7c 100644 --- a/.claude/skills/forge.md +++ b/.claude/skills/forge.md @@ -1200,7 +1200,7 @@ when OTel tracing is enabled (OTel v1 / Phase 4 / #105). Both use | `AuditScheduleModify` | `schedule_modify` | Schedule mutated at runtime | | `EventAuthVerify` | `auth_verify` | Inbound request authenticated (`provider`, `user_id`, `org_id`, `token_kind`; `email` when the identity carries one). **Channel invoker:** for a channel-originated request the transport credential is the loopback token (`provider:internal`/`user_id:forge-internal`, recorded truthfully) and the human sender is stamped as `channel`/`channel_user`/`channel_email` from the `X-Forge-Channel*` headers — honored only for the runtime-internal identity (same trust gate as `applyChannelOnBehalfOf`). Slack/Teams resolve `channel_email`; Telegram (numeric id) & WhatsApp (msisdn) carry `channel_user` only | | `EventAuthFail` | `auth_fail` | Inbound request rejected (`reason`, `token_kind`) | -| `AuditInputMediaRejected` | `input_media_rejected` | Inbound `file` parts the runtime can't forward to the model → rejected 4xx, not silently dropped (#255). Fields: `dropped` (`["file:"]`), `count`, `reason` (`model_not_vision_capable` \| `unsupported_media_type` \| `too_many_image_parts` \| `image_limit_exceeded`). Gate: `Runner.checkInboundMedia`. **Images** (png/jpeg/gif/webp) on a **vision model** (`coreruntime.ModelSupportsVision`), within the DoS bounds, are NOT rejected — `a2aMessageToLLM` projects them into `llm.ChatMessage.Parts` → Anthropic `image` source blocks / OpenAI `image_url` data URLs. Docs/video still rejected (Phase 3+). **DoS controls** (`media_limits.go` + `mediaSem`): body cap 32 MiB both transports; per-image ≤5 MiB & ≤50 MP (`CheckImageLimits`, header-only decode defuses bombs); ≤20 images/msg; ≤4 concurrent media requests (excess shed with 429/unavailable) | +| `AuditInputMediaRejected` | `input_media_rejected` | Inbound `file` parts the runtime can't forward to the model → rejected 4xx, not silently dropped (#255). Fields: `dropped` (`["file:"]`), `count`, `reason` (`model_not_vision_capable` \| `model_not_document_capable` \| `unsupported_media_type` \| `too_many_image_parts` \| `too_many_document_parts` \| `image_limit_exceeded` \| `document_limit_exceeded`). Gate: `Runner.checkInboundMedia`. **Images** (png/jpeg/gif/webp) on a **vision model** (`coreruntime.ModelSupportsVision`) and **PDFs** on a **doc model** (`ModelSupportsPDF` = Anthropic Claude 3.5+), within the DoS bounds, are NOT rejected — `a2aMessageToLLM` projects them into `llm.ChatMessage.Parts` → Anthropic `image`/`document` source blocks, OpenAI `image_url` data URLs. Non-PDF docs/video still rejected; OpenAI-Responses PDF + extraction fallback are follow-ups. **DoS controls** (`media_limits.go` + `mediaSem`): body cap 32 MiB both transports; per-image ≤5 MiB & ≤50 MP/≤100k-per-side (`CheckImageLimits`); per-PDF ≤32 MiB + `%PDF-` sniff (`CheckDocumentLimits`); ≤20 images & ≤5 docs/msg; ≤4 concurrent media requests (excess shed with 429/unavailable) | | `EventMCPServerStarted` | `mcp_server_started` | MCP server handshake succeeded | | `EventMCPServerFailed` | `mcp_server_failed` | MCP server dial / handshake failed | | `EventMCPServerDegraded` | `mcp_server_degraded` | MCP server in soft-fail | diff --git a/docs/core-concepts/runtime-engine.md b/docs/core-concepts/runtime-engine.md index 820dbcd3..b862681a 100644 --- a/docs/core-concepts/runtime-engine.md +++ b/docs/core-concepts/runtime-engine.md @@ -33,19 +33,26 @@ An inbound A2A message is a list of typed parts. `a2a.Message.PromptText()` proj The same projection feeds the inbound guardrail and intent-alignment scanners, so the security checks see exactly what the model sees — a payload carried in a data part can't reach the LLM while bypassing them. Each data part's projected block is capped (~16KiB, rune-safe) — the cap applies identically to the scanners and the prompt, so truncation can't open a divergence. -#### Image input (multimodal) +#### Image and document input (multimodal) -An image `file` part (`image/png`, `image/jpeg`, `image/gif`, `image/webp`) is forwarded to the model as native vision input when the resolved model is **vision-capable** (`runtime.ModelSupportsVision` — OpenAI `gpt-4o`/`gpt-4.1`/`gpt-5`/`o1`/`o3`/`o4`, Anthropic Claude 3+, Gemini 1.5/2). `a2aMessageToLLM` projects such parts into `llm.ChatMessage.Parts` (the flattened text stays in `Content` as the text-of-record for the scanners), and each provider serializes them natively — Anthropic `image` source blocks, OpenAI/Gemini `image_url` data URLs. A text-only message keeps `Parts` empty and marshals byte-identically to before. +A media `file` part is forwarded to the model as native input when the resolved model supports that modality: -Media the model can't consume is **rejected loudly, never silently dropped** (the `checkInboundMedia` ingest gate): an image on a text-only model, or a document/video part (not yet supported), returns a 4xx and emits the `input_media_rejected` audit event. Note the image **bytes** themselves are not text-scannable, so guardrail/intent scanning still applies only to the text/data projection; this is an accepted limitation. +- **Images** (`image/png`, `image/jpeg`, `image/gif`, `image/webp`) → **vision-capable** models (`runtime.ModelSupportsVision` — OpenAI `gpt-4o`/`gpt-4.1`/`gpt-5`/`o1`/`o3`/`o4`, Anthropic Claude 3+, Gemini 1.5/2). Serialized as Anthropic `image` source blocks / OpenAI/Gemini `image_url` data URLs. +- **PDFs** (`application/pdf`) → **document-capable** models (`runtime.ModelSupportsPDF` — Anthropic Claude 3.5+). Serialized as Anthropic `document` source blocks. OpenAI Responses `input_file`, Gemini documents, and text-extraction fallback for non-native models are deferred follow-ups. -**DoS bounds.** Because inline images raise the inbound-body cap to 32 MiB (both transports), the gate also enforces per-image and per-message limits, and a concurrency semaphore bounds how many media-bearing requests run at once — a flat body cap alone is not media DoS protection: +`a2aMessageToLLM` projects supported parts into `llm.ChatMessage.Parts` (the flattened text stays in `Content` as the text-of-record for the scanners). A text-only message keeps `Parts` empty and marshals byte-identically to before. + +Media the model can't consume is **rejected loudly, never silently dropped** (the `checkInboundMedia` ingest gate): an image on a text-only model, a PDF on a non-document model, or an unsupported type (other documents/video) returns a 4xx and emits the `input_media_rejected` audit event. Note media **bytes** are not text-scannable, so guardrail/intent scanning still applies only to the text/data projection; this is an accepted limitation. + +**DoS bounds.** Because inline media raises the inbound-body cap to 32 MiB (both transports), the gate also enforces per-part and per-message limits, and a concurrency semaphore bounds how many media-bearing requests run at once — a flat body cap alone is not media DoS protection: | Bound | Limit | On breach | |-------|-------|-----------| | Per-image bytes | 5 MiB (`MaxImagePartBytes`) | 4xx `image_limit_exceeded` | | Decoded dimensions | 100 000 px per side **and** 50 MP total (`MaxImagePixels`, png/jpeg/gif via header-only `DecodeConfig`; webp bounded by bytes) | 4xx `image_limit_exceeded` (defuses decompression bombs; the per-side bound also keeps the pixel product from overflowing int64) | | Images per message | 20 (`MaxImagePartsPerMessage`) | 4xx `too_many_image_parts` | +| Per-document bytes | 32 MiB (`MaxDocumentPartBytes`), plus a `%PDF-` format sniff | 4xx `document_limit_exceeded` | +| Documents per message | 5 (`MaxDocumentPartsPerMessage`) | 4xx `too_many_document_parts` | | Concurrent media requests | 4 (`maxConcurrentMediaRequests`) | `429`/unavailable — request is shed, not queued | **Note for guardrail pattern authors:** parts join with **newlines** (matching what the model sees). A pattern intended to match content that may span a part boundary should use `\s+` rather than a literal space — a payload split across two text parts joins as `…end\nstart…`. diff --git a/docs/security/audit-logging.md b/docs/security/audit-logging.md index b664c8d7..f1346ce6 100644 --- a/docs/security/audit-logging.md +++ b/docs/security/audit-logging.md @@ -32,7 +32,7 @@ All runtime security events are emitted as structured NDJSON to stderr with corr | `mcp_auth_resolved` | The parked call's consent arrived (or the wait was canceled) and it resumed (#330). Carries `server`, `subject`, `wait_ms`. Emitted **once**, attributed to the parked invocation (#366). | | `mcp_auth_timeout` | No consent within the window; the parked MCP call fails `no_token` (#330). Carries `server`, `subject`, `wait_ms`, `decision`. | | `auth_fail` | Inbound request rejected (with `reason`, `token_kind`). No `task_id` (none is ever created), but carries `workflow_execution_id` when the request had the execution header — so a rejected request is still attributable to its workflow run (#278). | -| `input_media_rejected` | An inbound `tasks/send`/`sendSubscribe` message carried `file` parts the runtime cannot forward to the model, so the request was rejected with a 4xx (JSON-RPC invalid-params / HTTP 400) instead of silently dropping the attachment and returning a plausible answer that ignored it (#255). Carries `fields.dropped` (`["file:", …]`), `fields.count`, and `fields.reason` — `model_not_vision_capable` (image sent to a text-only model), `unsupported_media_type` (documents/video — not yet accepted), `too_many_image_parts` (over the per-message image limit), or `image_limit_exceeded` (an image over the per-image byte or pixel bound). **Images** (`png`/`jpeg`/`gif`/`webp`) within those bounds, sent to a **vision-capable model**, are NOT rejected: they are projected into the model request as inline vision input and do not emit this event. A separate load-shedding path returns an unavailable/`429` (no audit event) when the runner is already at its concurrent-media-request cap. | +| `input_media_rejected` | An inbound `tasks/send`/`sendSubscribe` message carried `file` parts the runtime cannot forward to the model, so the request was rejected with a 4xx (JSON-RPC invalid-params / HTTP 400) instead of silently dropping the attachment and returning a plausible answer that ignored it (#255). Carries `fields.dropped` (`["file:", …]`), `fields.count`, and `fields.reason` — `model_not_vision_capable` (image on a text-only model), `model_not_document_capable` (PDF on a model without native document support), `unsupported_media_type` (a non-PDF document, video, … — not yet accepted), `too_many_image_parts` / `too_many_document_parts` (over the per-message count), or `image_limit_exceeded` / `document_limit_exceeded` (a part over its byte/pixel bound or failing its format sniff). **Images** (`png`/`jpeg`/`gif`/`webp`) on a **vision-capable model**, and **PDFs** on a **document-capable model** (Anthropic Claude 3.5+), within those bounds, are NOT rejected: they are projected into the model request as native vision / document blocks and do not emit this event. A separate load-shedding path returns an unavailable/`429` (no audit event) when the runner is already at its concurrent-media-request cap. | | `agent_card_published` | Agent Card finalized at startup or hot-reload (with `name`, `version`, `protocol_version`, `url`, `skill_count`, `capabilities`, `security_schemes`, `card_size_bytes`, `card_sha256`). See [Agent Card reference](../reference/a2a-agent-card.md). | | `policy_loaded` | One per non-empty policy layer at startup (system / user / workspace). Carries `fields.layer`, `source` (file path), deny-list size counts, and max bounds. See [Platform Policy](platform-policy.md). | | `policy_violation_at_build_time` | One per violation when `forge.yaml` conflicts with any policy layer. Agent refuses to start. Carries `fields.violation_kind` / `offending_value` / `forge_yaml_field` plus `layer` + `source` identifying the enforcing file. See [Platform Policy](platform-policy.md). | diff --git a/forge-cli/internal/surface/knowledge/forge.md b/forge-cli/internal/surface/knowledge/forge.md index c469f3cc..acf38d7c 100644 --- a/forge-cli/internal/surface/knowledge/forge.md +++ b/forge-cli/internal/surface/knowledge/forge.md @@ -1200,7 +1200,7 @@ when OTel tracing is enabled (OTel v1 / Phase 4 / #105). Both use | `AuditScheduleModify` | `schedule_modify` | Schedule mutated at runtime | | `EventAuthVerify` | `auth_verify` | Inbound request authenticated (`provider`, `user_id`, `org_id`, `token_kind`; `email` when the identity carries one). **Channel invoker:** for a channel-originated request the transport credential is the loopback token (`provider:internal`/`user_id:forge-internal`, recorded truthfully) and the human sender is stamped as `channel`/`channel_user`/`channel_email` from the `X-Forge-Channel*` headers — honored only for the runtime-internal identity (same trust gate as `applyChannelOnBehalfOf`). Slack/Teams resolve `channel_email`; Telegram (numeric id) & WhatsApp (msisdn) carry `channel_user` only | | `EventAuthFail` | `auth_fail` | Inbound request rejected (`reason`, `token_kind`) | -| `AuditInputMediaRejected` | `input_media_rejected` | Inbound `file` parts the runtime can't forward to the model → rejected 4xx, not silently dropped (#255). Fields: `dropped` (`["file:"]`), `count`, `reason` (`model_not_vision_capable` \| `unsupported_media_type` \| `too_many_image_parts` \| `image_limit_exceeded`). Gate: `Runner.checkInboundMedia`. **Images** (png/jpeg/gif/webp) on a **vision model** (`coreruntime.ModelSupportsVision`), within the DoS bounds, are NOT rejected — `a2aMessageToLLM` projects them into `llm.ChatMessage.Parts` → Anthropic `image` source blocks / OpenAI `image_url` data URLs. Docs/video still rejected (Phase 3+). **DoS controls** (`media_limits.go` + `mediaSem`): body cap 32 MiB both transports; per-image ≤5 MiB & ≤50 MP (`CheckImageLimits`, header-only decode defuses bombs); ≤20 images/msg; ≤4 concurrent media requests (excess shed with 429/unavailable) | +| `AuditInputMediaRejected` | `input_media_rejected` | Inbound `file` parts the runtime can't forward to the model → rejected 4xx, not silently dropped (#255). Fields: `dropped` (`["file:"]`), `count`, `reason` (`model_not_vision_capable` \| `model_not_document_capable` \| `unsupported_media_type` \| `too_many_image_parts` \| `too_many_document_parts` \| `image_limit_exceeded` \| `document_limit_exceeded`). Gate: `Runner.checkInboundMedia`. **Images** (png/jpeg/gif/webp) on a **vision model** (`coreruntime.ModelSupportsVision`) and **PDFs** on a **doc model** (`ModelSupportsPDF` = Anthropic Claude 3.5+), within the DoS bounds, are NOT rejected — `a2aMessageToLLM` projects them into `llm.ChatMessage.Parts` → Anthropic `image`/`document` source blocks, OpenAI `image_url` data URLs. Non-PDF docs/video still rejected; OpenAI-Responses PDF + extraction fallback are follow-ups. **DoS controls** (`media_limits.go` + `mediaSem`): body cap 32 MiB both transports; per-image ≤5 MiB & ≤50 MP/≤100k-per-side (`CheckImageLimits`); per-PDF ≤32 MiB + `%PDF-` sniff (`CheckDocumentLimits`); ≤20 images & ≤5 docs/msg; ≤4 concurrent media requests (excess shed with 429/unavailable) | | `EventMCPServerStarted` | `mcp_server_started` | MCP server handshake succeeded | | `EventMCPServerFailed` | `mcp_server_failed` | MCP server dial / handshake failed | | `EventMCPServerDegraded` | `mcp_server_degraded` | MCP server in soft-fail | diff --git a/forge-cli/runtime/media_gate_test.go b/forge-cli/runtime/media_gate_test.go index 194d3fcb..bfa4c2c2 100644 --- a/forge-cli/runtime/media_gate_test.go +++ b/forge-cli/runtime/media_gate_test.go @@ -66,11 +66,34 @@ func TestCheckInboundMedia_Phase2(t *testing.T) { } }) - t.Run("document rejected even on vision model", func(t *testing.T) { - r := runnerWithModel("gpt-4o") - got := r.checkInboundMedia(ctx, fileMsg("application/pdf", []byte("%PDF-1.7")), nil) - if got == "" || !strings.Contains(got, "application/pdf") { - t.Errorf("a document part must still be rejected in Phase 2; got %q", got) + t.Run("pdf accepted on a document-capable model", func(t *testing.T) { + r := runnerWithModel("claude-sonnet-5") + if got := r.checkInboundMedia(ctx, fileMsg("application/pdf", []byte("%PDF-1.7\nbody")), nil); got != "" { + t.Errorf("pdf on a document-capable model must be accepted, got reject: %q", got) + } + }) + + t.Run("pdf rejected on a non-document model", func(t *testing.T) { + r := runnerWithModel("gpt-4o") // vision-capable but not PDF-capable + got := r.checkInboundMedia(ctx, fileMsg("application/pdf", []byte("%PDF-1.7\nbody")), nil) + if got == "" || !strings.Contains(got, "application/pdf") || !strings.Contains(got, "does not support PDF") { + t.Errorf("pdf on a non-document model must be rejected with a clear reason; got %q", got) + } + }) + + t.Run("mislabeled pdf rejected", func(t *testing.T) { + r := runnerWithModel("claude-sonnet-5") + got := r.checkInboundMedia(ctx, fileMsg("application/pdf", []byte("PK\x03\x04 zip")), nil) + if got == "" { + t.Error("a non-PDF payload labeled application/pdf must be rejected") + } + }) + + t.Run("unsupported document type rejected", func(t *testing.T) { + r := runnerWithModel("claude-sonnet-5") + got := r.checkInboundMedia(ctx, fileMsg("application/vnd.ms-excel", []byte("x")), nil) + if got == "" { + t.Error("a non-PDF document type must be rejected (only PDF supported)") } }) diff --git a/forge-cli/runtime/runner.go b/forge-cli/runtime/runner.go index c1eb4bf6..de451619 100644 --- a/forge-cli/runtime/runner.go +++ b/forge-cli/runtime/runner.go @@ -2354,15 +2354,21 @@ func (r *Runner) checkInboundMedia(ctx context.Context, msg a2a.Message, auditLo if len(files) == 0 { return "" } - // Image parts are forwarded to the model as inline vision input when the - // resolved model is vision-capable (#255 Phase 2). Everything else — images - // on a non-vision model, and documents/video (later phases) — is still - // rejected loudly rather than silently dropped. - visionCapable := r.modelConfig != nil && coreruntime.ModelSupportsVision(r.modelConfig.Client.Model) + // Images are forwarded as inline vision input on a vision-capable model + // (#255 Phase 2); PDFs as native document blocks on a document-capable model + // (Phase 3). Everything else — media on a model that can't consume it, or an + // unsupported type (other documents/video) — is rejected loudly rather than + // silently dropped. + model := "" + if r.modelConfig != nil { + model = r.modelConfig.Client.Model + } + visionCapable := coreruntime.ModelSupportsVision(model) + pdfCapable := coreruntime.ModelSupportsPDF(model) var dropped []string reason := "unsupported_media_type" - imageCount := 0 + imageCount, docCount := 0, 0 for _, p := range msg.Parts { if p.Kind != a2a.PartKindFile { continue @@ -2375,32 +2381,50 @@ func (r *Runner) checkInboundMedia(ctx context.Context, msg a2a.Message, auditLo } data = p.File.Bytes } - if !coreruntime.IsImageMIME(mt) { + switch { + case coreruntime.IsImageMIME(mt): + if !visionCapable { + reason = "model_not_vision_capable" + dropped = append(dropped, "file:"+mt) + continue + } + imageCount++ + if imageCount > coreruntime.MaxImagePartsPerMessage { + reason = "too_many_image_parts" + dropped = append(dropped, fmt.Sprintf("file:%s (over the %d-image limit)", mt, coreruntime.MaxImagePartsPerMessage)) + continue + } + if lim := coreruntime.CheckImageLimits(mt, data); lim != "" { + reason = "image_limit_exceeded" + dropped = append(dropped, fmt.Sprintf("file:%s (%s)", mt, lim)) + continue + } + // accepted → projected as vision input + case coreruntime.IsDocumentMIME(mt): + if !pdfCapable { + reason = "model_not_document_capable" + dropped = append(dropped, "file:"+mt) + continue + } + docCount++ + if docCount > coreruntime.MaxDocumentPartsPerMessage { + reason = "too_many_document_parts" + dropped = append(dropped, fmt.Sprintf("file:%s (over the %d-document limit)", mt, coreruntime.MaxDocumentPartsPerMessage)) + continue + } + if lim := coreruntime.CheckDocumentLimits(mt, data); lim != "" { + reason = "document_limit_exceeded" + dropped = append(dropped, fmt.Sprintf("file:%s (%s)", mt, lim)) + continue + } + // accepted → projected as a native document block + default: reason = "unsupported_media_type" dropped = append(dropped, "file:"+mt) - continue - } - if !visionCapable { - reason = "model_not_vision_capable" - dropped = append(dropped, "file:"+mt) - continue } - // Image on a vision model: enforce the DoS bounds (#255). - imageCount++ - if imageCount > coreruntime.MaxImagePartsPerMessage { - reason = "too_many_image_parts" - dropped = append(dropped, fmt.Sprintf("file:%s (over the %d-image limit)", mt, coreruntime.MaxImagePartsPerMessage)) - continue - } - if lim := coreruntime.CheckImageLimits(mt, data); lim != "" { - reason = "image_limit_exceeded" - dropped = append(dropped, fmt.Sprintf("file:%s (%s)", mt, lim)) - continue - } - // accepted → projected to the model as vision input } if len(dropped) == 0 { - return "" // all file parts are images the model can consume + return "" // all file parts are media the model can consume } if auditLogger != nil { auditLogger.EmitFromContext(ctx, coreruntime.AuditEvent{ @@ -2415,17 +2439,19 @@ func (r *Runner) checkInboundMedia(ctx context.Context, msg a2a.Message, auditLo var detail string switch reason { case "model_not_vision_capable": - model := "" - if r.modelConfig != nil { - model = r.modelConfig.Client.Model - } detail = fmt.Sprintf("the configured model %q does not support image input", model) case "too_many_image_parts": detail = fmt.Sprintf("a message may carry at most %d image parts", coreruntime.MaxImagePartsPerMessage) case "image_limit_exceeded": detail = fmt.Sprintf("each image must be a decodable png/jpeg/gif/webp under %d bytes and %d pixels", coreruntime.MaxImagePartBytes, coreruntime.MaxImagePixels) + case "model_not_document_capable": + detail = fmt.Sprintf("the configured model %q does not support PDF document input", model) + case "too_many_document_parts": + detail = fmt.Sprintf("a message may carry at most %d document parts", coreruntime.MaxDocumentPartsPerMessage) + case "document_limit_exceeded": + detail = fmt.Sprintf("each document must be a valid PDF under %d bytes", coreruntime.MaxDocumentPartBytes) default: - detail = "only image input (png/jpeg/gif/webp) is accepted; documents and video are not yet supported" + detail = "only image (png/jpeg/gif/webp) and PDF input are accepted; other documents and video are not yet supported" } return fmt.Sprintf( "%d media part(s) rejected (%s): %s.", diff --git a/forge-core/llm/providers/anthropic.go b/forge-core/llm/providers/anthropic.go index 3add08b2..810bce19 100644 --- a/forge-core/llm/providers/anthropic.go +++ b/forge-core/llm/providers/anthropic.go @@ -238,12 +238,13 @@ type anthropicContentBlock struct { Input json.RawMessage `json:"input,omitempty"` ToolUseID string `json:"tool_use_id,omitempty"` Content string `json:"content,omitempty"` - Source *anthropicImageSource `json:"source,omitempty"` // type=="image" (#255) + Source *anthropicMediaSource `json:"source,omitempty"` // type=="image" | "document" (#255) } -// anthropicImageSource is the source of an image content block: -// {"type":"image","source":{"type":"base64","media_type":"image/png","data":"…"}}. -type anthropicImageSource struct { +// anthropicMediaSource is the base64 source of an image or document content +// block, e.g. {"type":"image","source":{"type":"base64","media_type": +// "image/png","data":"…"}} or a "document" block with media_type application/pdf. +type anthropicMediaSource struct { Type string `json:"type"` // "base64" MediaType string `json:"media_type"` // e.g. image/png Data string `json:"data"` // base64-encoded bytes @@ -378,7 +379,21 @@ func anthropicBlocksFromParts(parts []llm.ContentPart) []anthropicContentBlock { } blocks = append(blocks, anthropicContentBlock{ Type: "image", - Source: &anthropicImageSource{ + Source: &anthropicMediaSource{ + Type: "base64", + MediaType: p.Media.MimeType, + Data: base64.StdEncoding.EncodeToString(p.Media.Bytes), + }, + }) + case llm.ContentPartDocument: + if p.Media == nil || len(p.Media.Bytes) == 0 { + continue + } + // Anthropic document block: {"type":"document","source":{"type": + // "base64","media_type":"application/pdf","data":…}} (#255 Phase 3). + blocks = append(blocks, anthropicContentBlock{ + Type: "document", + Source: &anthropicMediaSource{ Type: "base64", MediaType: p.Media.MimeType, Data: base64.StdEncoding.EncodeToString(p.Media.Bytes), diff --git a/forge-core/llm/providers/multimodal_test.go b/forge-core/llm/providers/multimodal_test.go index fa3a762b..4010bba6 100644 --- a/forge-core/llm/providers/multimodal_test.go +++ b/forge-core/llm/providers/multimodal_test.go @@ -48,6 +48,37 @@ func TestAnthropic_ImagePartBecomesSourceBlock(t *testing.T) { } } +// TestAnthropic_DocumentPartBecomesDocumentBlock verifies a PDF document part +// serializes to an Anthropic document source block (#255 Phase 3). +func TestAnthropic_DocumentPartBecomesDocumentBlock(t *testing.T) { + pdf := []byte("%PDF-1.7 body") + msg := llm.ChatMessage{ + Role: llm.RoleUser, + Content: "summarize", + Parts: []llm.ContentPart{ + llm.NewTextContentPart("summarize"), + llm.NewMediaContentPart(llm.ContentPartDocument, llm.MediaRef{MimeType: "application/pdf", Bytes: pdf}), + }, + } + c := NewAnthropicClient(llm.ClientConfig{Model: "claude-sonnet-5"}) + body := c.toAnthropicRequest(&llm.ChatRequest{Messages: []llm.ChatMessage{msg}}, false) + + var blocks []anthropicContentBlock + if err := json.Unmarshal(body.Messages[0].Content, &blocks); err != nil { + t.Fatalf("content should be a block array: %v", err) + } + if len(blocks) != 2 || blocks[0].Type != "text" || blocks[1].Type != "document" { + t.Fatalf("blocks = %+v, want [text, document]", blocks) + } + src := blocks[1].Source + if src == nil || src.Type != "base64" || src.MediaType != "application/pdf" { + t.Fatalf("document source malformed: %+v", src) + } + if src.Data != base64.StdEncoding.EncodeToString(pdf) { + t.Errorf("document data not base64 of the bytes") + } +} + // TestAnthropic_TextOnlyWireUnchanged pins that a text-only message still // marshals its content as a bare JSON string (no block array) — byte-identical // to the pre-#255 wire and prompt-cache prefix. diff --git a/forge-core/runtime/loop.go b/forge-core/runtime/loop.go index e5d51fc2..59235bbd 100644 --- a/forge-core/runtime/loop.go +++ b/forge-core/runtime/loop.go @@ -1090,32 +1090,37 @@ func a2aMessageToLLM(msg a2a.Message) llm.ChatMessage { Content: msg.PromptText(), } - // Project image file parts into multimodal content parts (#255 Phase 2). - // Content (PromptText) stays the flattened text-of-record — authoritative - // for the guardrail/intent scanners and back-compat. Parts is set only when - // the message actually carries a supported image, and then it carries the - // projected text as one block plus each image, so the provider serializes - // the full turn without dropping the text. Non-image file parts never reach - // here: the ingest gate (checkInboundMedia) already rejected them. - var images []llm.ContentPart + // Project image and document file parts into multimodal content parts + // (#255 Phase 2/3). Content (PromptText) stays the flattened text-of-record + // — authoritative for the guardrail/intent scanners and back-compat. Parts + // is set only when the message carries supported media, and then it carries + // the projected text as one block plus each media part, so the provider + // serializes the full turn without dropping the text. Unsupported file parts + // never reach here: the ingest gate (checkInboundMedia) already rejected them. + var media []llm.ContentPart for _, p := range msg.Parts { - if p.Kind != a2a.PartKindFile || p.File == nil { + if p.Kind != a2a.PartKindFile || p.File == nil || len(p.File.Bytes) == 0 { continue } - if !IsImageMIME(p.File.MimeType) || len(p.File.Bytes) == 0 { - continue + switch { + case IsImageMIME(p.File.MimeType): + media = append(media, llm.NewMediaContentPart(llm.ContentPartImage, llm.MediaRef{ + MimeType: NormalizeImageMIME(p.File.MimeType), + Bytes: p.File.Bytes, + })) + case IsDocumentMIME(p.File.MimeType): + media = append(media, llm.NewMediaContentPart(llm.ContentPartDocument, llm.MediaRef{ + MimeType: p.File.MimeType, + Bytes: p.File.Bytes, + })) } - images = append(images, llm.NewMediaContentPart(llm.ContentPartImage, llm.MediaRef{ - MimeType: NormalizeImageMIME(p.File.MimeType), - Bytes: p.File.Bytes, - })) } - if len(images) > 0 { - parts := make([]llm.ContentPart, 0, len(images)+1) + if len(media) > 0 { + parts := make([]llm.ContentPart, 0, len(media)+1) if out.Content != "" { parts = append(parts, llm.NewTextContentPart(out.Content)) } - out.Parts = append(parts, images...) + out.Parts = append(parts, media...) } return out } diff --git a/forge-core/runtime/loop_projection_test.go b/forge-core/runtime/loop_projection_test.go index 806c2ea7..314d7561 100644 --- a/forge-core/runtime/loop_projection_test.go +++ b/forge-core/runtime/loop_projection_test.go @@ -57,18 +57,42 @@ func TestA2AMessageToLLM_TextOnlyLeavesPartsNil(t *testing.T) { } } -// TestA2AMessageToLLM_NonImageFileNotProjected: a non-image file part (the gate -// would reject it) must not be projected into Parts if it somehow reaches here. -func TestA2AMessageToLLM_NonImageFileNotProjected(t *testing.T) { +// TestA2AMessageToLLM_ProjectsPDFDocument: a PDF file part becomes a document +// content part (#255 Phase 3), alongside the projected text. +func TestA2AMessageToLLM_ProjectsPDFDocument(t *testing.T) { + pdf := []byte("%PDF-1.7\n...") msg := a2a.Message{ Role: a2a.MessageRoleUser, Parts: []a2a.Part{ - a2a.NewTextPart("read this"), - a2a.NewFilePart(a2a.FileContent{Name: "d.pdf", MimeType: "application/pdf", Bytes: []byte{1}}), + a2a.NewTextPart("summarize this"), + a2a.NewFilePart(a2a.FileContent{Name: "d.pdf", MimeType: "application/pdf", Bytes: pdf}), + }, + } + got := a2aMessageToLLM(msg) + if len(got.Parts) != 2 || got.Parts[0].Type != llm.ContentPartText { + t.Fatalf("Parts = %+v, want [text, document]", got.Parts) + } + doc := got.Parts[1] + if doc.Type != llm.ContentPartDocument || doc.Media == nil || doc.Media.MimeType != "application/pdf" { + t.Fatalf("second part should be a pdf document; got %+v", doc) + } + if string(doc.Media.Bytes) != string(pdf) { + t.Errorf("document bytes not carried through") + } +} + +// TestA2AMessageToLLM_UnsupportedFileNotProjected: an unsupported file part (the +// gate would reject it) must not be projected into Parts if it reaches here. +func TestA2AMessageToLLM_UnsupportedFileNotProjected(t *testing.T) { + msg := a2a.Message{ + Role: a2a.MessageRoleUser, + Parts: []a2a.Part{ + a2a.NewTextPart("play this"), + a2a.NewFilePart(a2a.FileContent{Name: "v.mp4", MimeType: "video/mp4", Bytes: []byte{1}}), }, } got := a2aMessageToLLM(msg) if got.Parts != nil { - t.Errorf("non-image file part must not be projected; got %+v", got.Parts) + t.Errorf("unsupported file part must not be projected; got %+v", got.Parts) } } diff --git a/forge-core/runtime/media_limits.go b/forge-core/runtime/media_limits.go index 0740646a..6c16541c 100644 --- a/forge-core/runtime/media_limits.go +++ b/forge-core/runtime/media_limits.go @@ -37,8 +37,35 @@ const ( // 100_000 keeps the product ≤ 1e10 (safe) and no legitimate image is that // large on any single side. maxImageDim = 100_000 + + // MaxDocumentPartBytes caps a single document (PDF) part's raw bytes, + // matching Anthropic's ~32 MiB per-document limit. The request-body cap + // bounds the total; this bounds any one document. + MaxDocumentPartBytes = 32 << 20 + + // MaxDocumentPartsPerMessage caps how many document parts one message may + // carry. + MaxDocumentPartsPerMessage = 5 ) +// CheckDocumentLimits validates a single document part's bytes against the +// per-document size bound and a format sniff. Returns a non-empty reason when +// the part must be rejected, or "" when it passes. PDF is the only supported +// document format today; the magic-byte check rejects a mislabeled or corrupt +// payload before it reaches the provider. +func CheckDocumentLimits(mime string, data []byte) string { + if len(data) == 0 { + return "document part has no bytes" + } + if len(data) > MaxDocumentPartBytes { + return fmt.Sprintf("document exceeds the %d-byte per-document limit (%d bytes)", MaxDocumentPartBytes, len(data)) + } + if IsDocumentMIME(mime) && !bytes.HasPrefix(data, []byte("%PDF-")) { + return "document is not a valid PDF (missing %PDF- header)" + } + return "" +} + // CheckImageLimits validates a single image part's bytes against the per-image // size and decode-dimension bounds. It returns a non-empty reason string when // the part must be rejected, or "" when it passes. diff --git a/forge-core/runtime/media_limits_test.go b/forge-core/runtime/media_limits_test.go index e5a1ffb6..2b0ff60c 100644 --- a/forge-core/runtime/media_limits_test.go +++ b/forge-core/runtime/media_limits_test.go @@ -111,3 +111,28 @@ func TestCheckImageLimits(t *testing.T) { } }) } + +func TestCheckDocumentLimits(t *testing.T) { + t.Run("valid pdf passes", func(t *testing.T) { + if got := CheckDocumentLimits("application/pdf", []byte("%PDF-1.7\n...")); got != "" { + t.Errorf("valid pdf rejected: %q", got) + } + }) + t.Run("empty bytes rejected", func(t *testing.T) { + if CheckDocumentLimits("application/pdf", nil) == "" { + t.Error("empty document bytes must be rejected") + } + }) + t.Run("oversized rejected", func(t *testing.T) { + big := make([]byte, MaxDocumentPartBytes+1) + copy(big, []byte("%PDF-1.7")) + if CheckDocumentLimits("application/pdf", big) == "" { + t.Error("document over the byte cap must be rejected") + } + }) + t.Run("mislabeled non-pdf rejected", func(t *testing.T) { + if CheckDocumentLimits("application/pdf", []byte("PK\x03\x04 this is a zip")) == "" { + t.Error("a payload without the %PDF- header must be rejected") + } + }) +} diff --git a/forge-core/runtime/model_capabilities.go b/forge-core/runtime/model_capabilities.go index aef3d790..fee0b44f 100644 --- a/forge-core/runtime/model_capabilities.go +++ b/forge-core/runtime/model_capabilities.go @@ -24,14 +24,37 @@ var visionCapablePrefixes = []string{ "gemini-1.5", "gemini-2", } +// pdfCapablePrefixes lists model-name prefixes whose models accept a PDF +// document natively via their provider's document block (#255 Phase 3). Today +// that is Anthropic Claude 3.5 and later — the 3.5/3.7 releases plus the 4.x/5 +// families (claude-opus-4…, claude-sonnet-5…, claude-haiku-4…). NOTE the bare +// "claude-3" (Claude 3.0) is intentionally EXCLUDED: 3.0 predates PDF support, +// and 3.0 models are named claude-3-opus/sonnet/haiku (which don't match the +// claude-opus/sonnet/haiku family prefixes), so they correctly fall through to +// a loud reject. OpenAI Responses input_file and Gemini document support are +// deferred follow-ups. +var pdfCapablePrefixes = []string{ + "claude-3-5", "claude-3-7", "claude-opus", "claude-sonnet", "claude-haiku", +} + // ModelSupportsVision reports whether the named model accepts image input. // Case-insensitive prefix match; unknown/empty models return false. func ModelSupportsVision(model string) bool { + return matchesPrefix(model, visionCapablePrefixes) +} + +// ModelSupportsPDF reports whether the named model accepts a PDF document +// natively. Case-insensitive prefix match; unknown/empty models return false. +func ModelSupportsPDF(model string) bool { + return matchesPrefix(model, pdfCapablePrefixes) +} + +func matchesPrefix(model string, prefixes []string) bool { m := strings.ToLower(strings.TrimSpace(model)) if m == "" { return false } - for _, p := range visionCapablePrefixes { + for _, p := range prefixes { if strings.HasPrefix(m, p) { return true } @@ -50,6 +73,13 @@ func IsImageMIME(mime string) bool { return false } +// IsDocumentMIME reports whether a MIME type denotes a document forge can +// forward as a native document block. Restricted to PDF today (the only format +// with native provider support without extraction). +func IsDocumentMIME(mime string) bool { + return strings.ToLower(strings.TrimSpace(mime)) == "application/pdf" +} + // NormalizeImageMIME lowercases and canonicalizes an image MIME type — notably // mapping the common non-standard "image/jpg" to "image/jpeg", which is the // form Anthropic's image source block requires. diff --git a/forge-core/runtime/model_capabilities_test.go b/forge-core/runtime/model_capabilities_test.go index 099b7adf..a557ca36 100644 --- a/forge-core/runtime/model_capabilities_test.go +++ b/forge-core/runtime/model_capabilities_test.go @@ -26,6 +26,32 @@ func TestModelSupportsVision(t *testing.T) { } } +func TestModelSupportsPDF(t *testing.T) { + pdf := []string{"claude-3-5-sonnet", "claude-3-7-sonnet", "claude-opus-4-8", "claude-sonnet-5", "claude-haiku-4-5"} + for _, m := range pdf { + if !ModelSupportsPDF(m) { + t.Errorf("ModelSupportsPDF(%q) = false, want true", m) + } + } + noPDF := []string{"", "claude-3-opus", "claude-3-sonnet", "gpt-4o", "gpt-5", "gemini-2.5-flash", "o3-mini"} + for _, m := range noPDF { + if ModelSupportsPDF(m) { + t.Errorf("ModelSupportsPDF(%q) = true, want false", m) + } + } +} + +func TestIsDocumentMIME(t *testing.T) { + if !IsDocumentMIME("application/pdf") || !IsDocumentMIME("APPLICATION/PDF ") { + t.Error("application/pdf must be a document MIME") + } + for _, mt := range []string{"", "image/png", "text/plain", "application/zip", "video/mp4"} { + if IsDocumentMIME(mt) { + t.Errorf("IsDocumentMIME(%q) = true, want false", mt) + } + } +} + func TestIsImageMIME(t *testing.T) { for _, mt := range []string{"image/png", "image/jpeg", "image/jpg", "IMAGE/PNG", "image/gif", "image/webp"} { if !IsImageMIME(mt) { From 461d10acccc832dcd0aa967d3128abee13acd452 Mon Sep 17 00:00:00 2001 From: MK Date: Mon, 28 Sep 2026 00:57:21 -0400 Subject: [PATCH 2/4] fix(runtime): narrow PDF capability to Sonnet/Opus; clarify per-doc cap (#534 review) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit - ModelSupportsPDF now scopes to the Sonnet & Opus families only (claude-3-5-sonnet, claude-3-7-sonnet, claude-opus, claude-sonnet). Haiku is excluded — Anthropic's PDF support was Sonnet-first and Haiku's native-PDF coverage is unconfirmed, so a Haiku PDF now gets forge's clean model_not_document_capable reject instead of risking an opaque provider error. Fail closed; widen once confirmed. The 3.5/3.7 prefixes name "sonnet" explicitly so they no longer catch claude-3-5-haiku. - Clarifying comment on MaxDocumentPartBytes: the 32 MiB request-body cap is the binding constraint (base64 inflation ~33% → ~24 MiB raw max), so the per-doc number is provider-aligned but not independently reachable today. - Tests + docs + skill updated to the narrowed set. --- .claude/skills/forge.md | 2 +- docs/core-concepts/runtime-engine.md | 2 +- forge-cli/internal/surface/knowledge/forge.md | 2 +- forge-core/runtime/media_limits.go | 8 ++++-- forge-core/runtime/model_capabilities.go | 27 ++++++++++++------- forge-core/runtime/model_capabilities_test.go | 6 +++-- 6 files changed, 31 insertions(+), 16 deletions(-) diff --git a/.claude/skills/forge.md b/.claude/skills/forge.md index acf38d7c..3acfa5cc 100644 --- a/.claude/skills/forge.md +++ b/.claude/skills/forge.md @@ -1200,7 +1200,7 @@ when OTel tracing is enabled (OTel v1 / Phase 4 / #105). Both use | `AuditScheduleModify` | `schedule_modify` | Schedule mutated at runtime | | `EventAuthVerify` | `auth_verify` | Inbound request authenticated (`provider`, `user_id`, `org_id`, `token_kind`; `email` when the identity carries one). **Channel invoker:** for a channel-originated request the transport credential is the loopback token (`provider:internal`/`user_id:forge-internal`, recorded truthfully) and the human sender is stamped as `channel`/`channel_user`/`channel_email` from the `X-Forge-Channel*` headers — honored only for the runtime-internal identity (same trust gate as `applyChannelOnBehalfOf`). Slack/Teams resolve `channel_email`; Telegram (numeric id) & WhatsApp (msisdn) carry `channel_user` only | | `EventAuthFail` | `auth_fail` | Inbound request rejected (`reason`, `token_kind`) | -| `AuditInputMediaRejected` | `input_media_rejected` | Inbound `file` parts the runtime can't forward to the model → rejected 4xx, not silently dropped (#255). Fields: `dropped` (`["file:"]`), `count`, `reason` (`model_not_vision_capable` \| `model_not_document_capable` \| `unsupported_media_type` \| `too_many_image_parts` \| `too_many_document_parts` \| `image_limit_exceeded` \| `document_limit_exceeded`). Gate: `Runner.checkInboundMedia`. **Images** (png/jpeg/gif/webp) on a **vision model** (`coreruntime.ModelSupportsVision`) and **PDFs** on a **doc model** (`ModelSupportsPDF` = Anthropic Claude 3.5+), within the DoS bounds, are NOT rejected — `a2aMessageToLLM` projects them into `llm.ChatMessage.Parts` → Anthropic `image`/`document` source blocks, OpenAI `image_url` data URLs. Non-PDF docs/video still rejected; OpenAI-Responses PDF + extraction fallback are follow-ups. **DoS controls** (`media_limits.go` + `mediaSem`): body cap 32 MiB both transports; per-image ≤5 MiB & ≤50 MP/≤100k-per-side (`CheckImageLimits`); per-PDF ≤32 MiB + `%PDF-` sniff (`CheckDocumentLimits`); ≤20 images & ≤5 docs/msg; ≤4 concurrent media requests (excess shed with 429/unavailable) | +| `AuditInputMediaRejected` | `input_media_rejected` | Inbound `file` parts the runtime can't forward to the model → rejected 4xx, not silently dropped (#255). Fields: `dropped` (`["file:"]`), `count`, `reason` (`model_not_vision_capable` \| `model_not_document_capable` \| `unsupported_media_type` \| `too_many_image_parts` \| `too_many_document_parts` \| `image_limit_exceeded` \| `document_limit_exceeded`). Gate: `Runner.checkInboundMedia`. **Images** (png/jpeg/gif/webp) on a **vision model** (`coreruntime.ModelSupportsVision`) and **PDFs** on a **doc model** (`ModelSupportsPDF` = Anthropic Sonnet 3.5+/Opus 4+; Haiku excluded, fail-closed), within the DoS bounds, are NOT rejected — `a2aMessageToLLM` projects them into `llm.ChatMessage.Parts` → Anthropic `image`/`document` source blocks, OpenAI `image_url` data URLs. Non-PDF docs/video still rejected; OpenAI-Responses PDF + extraction fallback are follow-ups. **DoS controls** (`media_limits.go` + `mediaSem`): body cap 32 MiB both transports; per-image ≤5 MiB & ≤50 MP/≤100k-per-side (`CheckImageLimits`); per-PDF ≤32 MiB + `%PDF-` sniff (`CheckDocumentLimits`); ≤20 images & ≤5 docs/msg; ≤4 concurrent media requests (excess shed with 429/unavailable) | | `EventMCPServerStarted` | `mcp_server_started` | MCP server handshake succeeded | | `EventMCPServerFailed` | `mcp_server_failed` | MCP server dial / handshake failed | | `EventMCPServerDegraded` | `mcp_server_degraded` | MCP server in soft-fail | diff --git a/docs/core-concepts/runtime-engine.md b/docs/core-concepts/runtime-engine.md index b862681a..19597761 100644 --- a/docs/core-concepts/runtime-engine.md +++ b/docs/core-concepts/runtime-engine.md @@ -38,7 +38,7 @@ The same projection feeds the inbound guardrail and intent-alignment scanners, s A media `file` part is forwarded to the model as native input when the resolved model supports that modality: - **Images** (`image/png`, `image/jpeg`, `image/gif`, `image/webp`) → **vision-capable** models (`runtime.ModelSupportsVision` — OpenAI `gpt-4o`/`gpt-4.1`/`gpt-5`/`o1`/`o3`/`o4`, Anthropic Claude 3+, Gemini 1.5/2). Serialized as Anthropic `image` source blocks / OpenAI/Gemini `image_url` data URLs. -- **PDFs** (`application/pdf`) → **document-capable** models (`runtime.ModelSupportsPDF` — Anthropic Claude 3.5+). Serialized as Anthropic `document` source blocks. OpenAI Responses `input_file`, Gemini documents, and text-extraction fallback for non-native models are deferred follow-ups. +- **PDFs** (`application/pdf`) → **document-capable** models (`runtime.ModelSupportsPDF` — Anthropic Sonnet 3.5+ and Opus 4+; Haiku is excluded pending confirmation, so a Haiku PDF gets a clean reject rather than a provider error). Serialized as Anthropic `document` source blocks. OpenAI Responses `input_file`, Gemini documents, and text-extraction fallback for non-native models are deferred follow-ups. `a2aMessageToLLM` projects supported parts into `llm.ChatMessage.Parts` (the flattened text stays in `Content` as the text-of-record for the scanners). A text-only message keeps `Parts` empty and marshals byte-identically to before. diff --git a/forge-cli/internal/surface/knowledge/forge.md b/forge-cli/internal/surface/knowledge/forge.md index acf38d7c..3acfa5cc 100644 --- a/forge-cli/internal/surface/knowledge/forge.md +++ b/forge-cli/internal/surface/knowledge/forge.md @@ -1200,7 +1200,7 @@ when OTel tracing is enabled (OTel v1 / Phase 4 / #105). Both use | `AuditScheduleModify` | `schedule_modify` | Schedule mutated at runtime | | `EventAuthVerify` | `auth_verify` | Inbound request authenticated (`provider`, `user_id`, `org_id`, `token_kind`; `email` when the identity carries one). **Channel invoker:** for a channel-originated request the transport credential is the loopback token (`provider:internal`/`user_id:forge-internal`, recorded truthfully) and the human sender is stamped as `channel`/`channel_user`/`channel_email` from the `X-Forge-Channel*` headers — honored only for the runtime-internal identity (same trust gate as `applyChannelOnBehalfOf`). Slack/Teams resolve `channel_email`; Telegram (numeric id) & WhatsApp (msisdn) carry `channel_user` only | | `EventAuthFail` | `auth_fail` | Inbound request rejected (`reason`, `token_kind`) | -| `AuditInputMediaRejected` | `input_media_rejected` | Inbound `file` parts the runtime can't forward to the model → rejected 4xx, not silently dropped (#255). Fields: `dropped` (`["file:"]`), `count`, `reason` (`model_not_vision_capable` \| `model_not_document_capable` \| `unsupported_media_type` \| `too_many_image_parts` \| `too_many_document_parts` \| `image_limit_exceeded` \| `document_limit_exceeded`). Gate: `Runner.checkInboundMedia`. **Images** (png/jpeg/gif/webp) on a **vision model** (`coreruntime.ModelSupportsVision`) and **PDFs** on a **doc model** (`ModelSupportsPDF` = Anthropic Claude 3.5+), within the DoS bounds, are NOT rejected — `a2aMessageToLLM` projects them into `llm.ChatMessage.Parts` → Anthropic `image`/`document` source blocks, OpenAI `image_url` data URLs. Non-PDF docs/video still rejected; OpenAI-Responses PDF + extraction fallback are follow-ups. **DoS controls** (`media_limits.go` + `mediaSem`): body cap 32 MiB both transports; per-image ≤5 MiB & ≤50 MP/≤100k-per-side (`CheckImageLimits`); per-PDF ≤32 MiB + `%PDF-` sniff (`CheckDocumentLimits`); ≤20 images & ≤5 docs/msg; ≤4 concurrent media requests (excess shed with 429/unavailable) | +| `AuditInputMediaRejected` | `input_media_rejected` | Inbound `file` parts the runtime can't forward to the model → rejected 4xx, not silently dropped (#255). Fields: `dropped` (`["file:"]`), `count`, `reason` (`model_not_vision_capable` \| `model_not_document_capable` \| `unsupported_media_type` \| `too_many_image_parts` \| `too_many_document_parts` \| `image_limit_exceeded` \| `document_limit_exceeded`). Gate: `Runner.checkInboundMedia`. **Images** (png/jpeg/gif/webp) on a **vision model** (`coreruntime.ModelSupportsVision`) and **PDFs** on a **doc model** (`ModelSupportsPDF` = Anthropic Sonnet 3.5+/Opus 4+; Haiku excluded, fail-closed), within the DoS bounds, are NOT rejected — `a2aMessageToLLM` projects them into `llm.ChatMessage.Parts` → Anthropic `image`/`document` source blocks, OpenAI `image_url` data URLs. Non-PDF docs/video still rejected; OpenAI-Responses PDF + extraction fallback are follow-ups. **DoS controls** (`media_limits.go` + `mediaSem`): body cap 32 MiB both transports; per-image ≤5 MiB & ≤50 MP/≤100k-per-side (`CheckImageLimits`); per-PDF ≤32 MiB + `%PDF-` sniff (`CheckDocumentLimits`); ≤20 images & ≤5 docs/msg; ≤4 concurrent media requests (excess shed with 429/unavailable) | | `EventMCPServerStarted` | `mcp_server_started` | MCP server handshake succeeded | | `EventMCPServerFailed` | `mcp_server_failed` | MCP server dial / handshake failed | | `EventMCPServerDegraded` | `mcp_server_degraded` | MCP server in soft-fail | diff --git a/forge-core/runtime/media_limits.go b/forge-core/runtime/media_limits.go index 6c16541c..6e7c0db4 100644 --- a/forge-core/runtime/media_limits.go +++ b/forge-core/runtime/media_limits.go @@ -39,8 +39,12 @@ const ( maxImageDim = 100_000 // MaxDocumentPartBytes caps a single document (PDF) part's raw bytes, - // matching Anthropic's ~32 MiB per-document limit. The request-body cap - // bounds the total; this bounds any one document. + // matching Anthropic's ~32 MiB per-document limit. NOTE the 32 MiB + // request-body cap is the binding constraint in practice: a PDF base64- + // inflates ~33% in the JSON envelope, so MaxBytesReader rejects anything + // over ~24 MiB raw before this per-part check runs (#534 review). This + // value stays aligned with the provider limit for when a future + // envelope/media-split change lifts the base64 overhead. MaxDocumentPartBytes = 32 << 20 // MaxDocumentPartsPerMessage caps how many document parts one message may diff --git a/forge-core/runtime/model_capabilities.go b/forge-core/runtime/model_capabilities.go index fee0b44f..ea8c5e4b 100644 --- a/forge-core/runtime/model_capabilities.go +++ b/forge-core/runtime/model_capabilities.go @@ -25,16 +25,25 @@ var visionCapablePrefixes = []string{ } // pdfCapablePrefixes lists model-name prefixes whose models accept a PDF -// document natively via their provider's document block (#255 Phase 3). Today -// that is Anthropic Claude 3.5 and later — the 3.5/3.7 releases plus the 4.x/5 -// families (claude-opus-4…, claude-sonnet-5…, claude-haiku-4…). NOTE the bare -// "claude-3" (Claude 3.0) is intentionally EXCLUDED: 3.0 predates PDF support, -// and 3.0 models are named claude-3-opus/sonnet/haiku (which don't match the -// claude-opus/sonnet/haiku family prefixes), so they correctly fall through to -// a loud reject. OpenAI Responses input_file and Gemini document support are -// deferred follow-ups. +// document natively via Anthropic's document block (#255 Phase 3). Scoped +// conservatively to the SONNET and OPUS families, which are confirmed on +// Anthropic's PDF-support matrix: the 3.5/3.7 Sonnet releases plus the 4.x/5 +// Sonnet & Opus families (claude-sonnet-4…, claude-sonnet-5, claude-opus-4…). +// +// Deliberately NARROW (#534 review): +// - HAIKU is excluded — Anthropic's PDF support was Sonnet-first and Haiku's +// native-PDF coverage is unconfirmed; rejecting a Haiku PDF cleanly at the +// gate is preferable to sending it and getting an opaque provider error. +// (Fail closed; widen here once confirmed.) +// - Bare "claude-3" (Claude 3.0) is excluded — it predates PDF support. The +// 3.0 models are claude-3-{opus,sonnet,haiku}, which do NOT match the +// "claude-sonnet"/"claude-opus" family prefixes, so they fall through to a +// loud reject. The 3.5/3.7 prefixes below name "sonnet" explicitly so they +// don't catch claude-3-5-haiku. +// +// OpenAI Responses input_file and Gemini document support are deferred follow-ups. var pdfCapablePrefixes = []string{ - "claude-3-5", "claude-3-7", "claude-opus", "claude-sonnet", "claude-haiku", + "claude-3-5-sonnet", "claude-3-7-sonnet", "claude-opus", "claude-sonnet", } // ModelSupportsVision reports whether the named model accepts image input. diff --git a/forge-core/runtime/model_capabilities_test.go b/forge-core/runtime/model_capabilities_test.go index a557ca36..e436807d 100644 --- a/forge-core/runtime/model_capabilities_test.go +++ b/forge-core/runtime/model_capabilities_test.go @@ -27,13 +27,15 @@ func TestModelSupportsVision(t *testing.T) { } func TestModelSupportsPDF(t *testing.T) { - pdf := []string{"claude-3-5-sonnet", "claude-3-7-sonnet", "claude-opus-4-8", "claude-sonnet-5", "claude-haiku-4-5"} + pdf := []string{"claude-3-5-sonnet", "claude-3-7-sonnet", "claude-opus-4-8", "claude-sonnet-5", "claude-sonnet-4-5"} for _, m := range pdf { if !ModelSupportsPDF(m) { t.Errorf("ModelSupportsPDF(%q) = false, want true", m) } } - noPDF := []string{"", "claude-3-opus", "claude-3-sonnet", "gpt-4o", "gpt-5", "gemini-2.5-flash", "o3-mini"} + // Haiku is deliberately excluded (Sonnet-first PDF support, Haiku unconfirmed + // → fail closed), as is bare Claude 3.0 and the 3.5 Haiku release. + noPDF := []string{"", "claude-3-opus", "claude-3-sonnet", "claude-3-5-haiku", "claude-haiku-4-5", "gpt-4o", "gpt-5", "gemini-2.5-flash", "o3-mini"} for _, m := range noPDF { if ModelSupportsPDF(m) { t.Errorf("ModelSupportsPDF(%q) = true, want false", m) From 82c2ae5225ca42fa72844d48225951b92e73eca4 Mon Sep 17 00:00:00 2001 From: MK Date: Mon, 28 Sep 2026 01:02:48 -0400 Subject: [PATCH 3/4] fix(runtime): PDF-capable set includes Haiku 4.5 and Fable 5 (#534 review) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Confirmed against Anthropic's PDF matrix: Claude Haiku 4.5 fully supports native PDF document input, and Fable 5 supports it too. Add "claude-haiku" (matches the 4.x/5 family naming claude-haiku-4-5…, not the older claude-3-5-haiku, so Haiku 3.5 stays excluded pending confirmation) and "claude-fable" to pdfCapablePrefixes. Tests + docs + skill updated. --- .claude/skills/forge.md | 2 +- docs/core-concepts/runtime-engine.md | 2 +- forge-cli/internal/surface/knowledge/forge.md | 2 +- forge-core/runtime/model_capabilities.go | 30 +++++++++---------- forge-core/runtime/model_capabilities_test.go | 7 ++--- 5 files changed, 21 insertions(+), 22 deletions(-) diff --git a/.claude/skills/forge.md b/.claude/skills/forge.md index 3acfa5cc..44dce16b 100644 --- a/.claude/skills/forge.md +++ b/.claude/skills/forge.md @@ -1200,7 +1200,7 @@ when OTel tracing is enabled (OTel v1 / Phase 4 / #105). Both use | `AuditScheduleModify` | `schedule_modify` | Schedule mutated at runtime | | `EventAuthVerify` | `auth_verify` | Inbound request authenticated (`provider`, `user_id`, `org_id`, `token_kind`; `email` when the identity carries one). **Channel invoker:** for a channel-originated request the transport credential is the loopback token (`provider:internal`/`user_id:forge-internal`, recorded truthfully) and the human sender is stamped as `channel`/`channel_user`/`channel_email` from the `X-Forge-Channel*` headers — honored only for the runtime-internal identity (same trust gate as `applyChannelOnBehalfOf`). Slack/Teams resolve `channel_email`; Telegram (numeric id) & WhatsApp (msisdn) carry `channel_user` only | | `EventAuthFail` | `auth_fail` | Inbound request rejected (`reason`, `token_kind`) | -| `AuditInputMediaRejected` | `input_media_rejected` | Inbound `file` parts the runtime can't forward to the model → rejected 4xx, not silently dropped (#255). Fields: `dropped` (`["file:"]`), `count`, `reason` (`model_not_vision_capable` \| `model_not_document_capable` \| `unsupported_media_type` \| `too_many_image_parts` \| `too_many_document_parts` \| `image_limit_exceeded` \| `document_limit_exceeded`). Gate: `Runner.checkInboundMedia`. **Images** (png/jpeg/gif/webp) on a **vision model** (`coreruntime.ModelSupportsVision`) and **PDFs** on a **doc model** (`ModelSupportsPDF` = Anthropic Sonnet 3.5+/Opus 4+; Haiku excluded, fail-closed), within the DoS bounds, are NOT rejected — `a2aMessageToLLM` projects them into `llm.ChatMessage.Parts` → Anthropic `image`/`document` source blocks, OpenAI `image_url` data URLs. Non-PDF docs/video still rejected; OpenAI-Responses PDF + extraction fallback are follow-ups. **DoS controls** (`media_limits.go` + `mediaSem`): body cap 32 MiB both transports; per-image ≤5 MiB & ≤50 MP/≤100k-per-side (`CheckImageLimits`); per-PDF ≤32 MiB + `%PDF-` sniff (`CheckDocumentLimits`); ≤20 images & ≤5 docs/msg; ≤4 concurrent media requests (excess shed with 429/unavailable) | +| `AuditInputMediaRejected` | `input_media_rejected` | Inbound `file` parts the runtime can't forward to the model → rejected 4xx, not silently dropped (#255). Fields: `dropped` (`["file:"]`), `count`, `reason` (`model_not_vision_capable` \| `model_not_document_capable` \| `unsupported_media_type` \| `too_many_image_parts` \| `too_many_document_parts` \| `image_limit_exceeded` \| `document_limit_exceeded`). Gate: `Runner.checkInboundMedia`. **Images** (png/jpeg/gif/webp) on a **vision model** (`coreruntime.ModelSupportsVision`) and **PDFs** on a **doc model** (`ModelSupportsPDF` = Anthropic Sonnet 3.5+/Opus 4+/Haiku 4.5+/Fable 5; Claude 3.0 & 3.5-Haiku excluded, fail-closed), within the DoS bounds, are NOT rejected — `a2aMessageToLLM` projects them into `llm.ChatMessage.Parts` → Anthropic `image`/`document` source blocks, OpenAI `image_url` data URLs. Non-PDF docs/video still rejected; OpenAI-Responses PDF + extraction fallback are follow-ups. **DoS controls** (`media_limits.go` + `mediaSem`): body cap 32 MiB both transports; per-image ≤5 MiB & ≤50 MP/≤100k-per-side (`CheckImageLimits`); per-PDF ≤32 MiB + `%PDF-` sniff (`CheckDocumentLimits`); ≤20 images & ≤5 docs/msg; ≤4 concurrent media requests (excess shed with 429/unavailable) | | `EventMCPServerStarted` | `mcp_server_started` | MCP server handshake succeeded | | `EventMCPServerFailed` | `mcp_server_failed` | MCP server dial / handshake failed | | `EventMCPServerDegraded` | `mcp_server_degraded` | MCP server in soft-fail | diff --git a/docs/core-concepts/runtime-engine.md b/docs/core-concepts/runtime-engine.md index 19597761..c2d48ea6 100644 --- a/docs/core-concepts/runtime-engine.md +++ b/docs/core-concepts/runtime-engine.md @@ -38,7 +38,7 @@ The same projection feeds the inbound guardrail and intent-alignment scanners, s A media `file` part is forwarded to the model as native input when the resolved model supports that modality: - **Images** (`image/png`, `image/jpeg`, `image/gif`, `image/webp`) → **vision-capable** models (`runtime.ModelSupportsVision` — OpenAI `gpt-4o`/`gpt-4.1`/`gpt-5`/`o1`/`o3`/`o4`, Anthropic Claude 3+, Gemini 1.5/2). Serialized as Anthropic `image` source blocks / OpenAI/Gemini `image_url` data URLs. -- **PDFs** (`application/pdf`) → **document-capable** models (`runtime.ModelSupportsPDF` — Anthropic Sonnet 3.5+ and Opus 4+; Haiku is excluded pending confirmation, so a Haiku PDF gets a clean reject rather than a provider error). Serialized as Anthropic `document` source blocks. OpenAI Responses `input_file`, Gemini documents, and text-extraction fallback for non-native models are deferred follow-ups. +- **PDFs** (`application/pdf`) → **document-capable** models (`runtime.ModelSupportsPDF` — Anthropic Sonnet 3.5+, Opus 4+, Haiku 4.5+, and Fable 5; the older Claude 3.0 and 3.5 Haiku are excluded, so a PDF sent to those gets a clean reject rather than a provider error). Serialized as Anthropic `document` source blocks. OpenAI Responses `input_file`, Gemini documents, and text-extraction fallback for non-native models are deferred follow-ups. `a2aMessageToLLM` projects supported parts into `llm.ChatMessage.Parts` (the flattened text stays in `Content` as the text-of-record for the scanners). A text-only message keeps `Parts` empty and marshals byte-identically to before. diff --git a/forge-cli/internal/surface/knowledge/forge.md b/forge-cli/internal/surface/knowledge/forge.md index 3acfa5cc..44dce16b 100644 --- a/forge-cli/internal/surface/knowledge/forge.md +++ b/forge-cli/internal/surface/knowledge/forge.md @@ -1200,7 +1200,7 @@ when OTel tracing is enabled (OTel v1 / Phase 4 / #105). Both use | `AuditScheduleModify` | `schedule_modify` | Schedule mutated at runtime | | `EventAuthVerify` | `auth_verify` | Inbound request authenticated (`provider`, `user_id`, `org_id`, `token_kind`; `email` when the identity carries one). **Channel invoker:** for a channel-originated request the transport credential is the loopback token (`provider:internal`/`user_id:forge-internal`, recorded truthfully) and the human sender is stamped as `channel`/`channel_user`/`channel_email` from the `X-Forge-Channel*` headers — honored only for the runtime-internal identity (same trust gate as `applyChannelOnBehalfOf`). Slack/Teams resolve `channel_email`; Telegram (numeric id) & WhatsApp (msisdn) carry `channel_user` only | | `EventAuthFail` | `auth_fail` | Inbound request rejected (`reason`, `token_kind`) | -| `AuditInputMediaRejected` | `input_media_rejected` | Inbound `file` parts the runtime can't forward to the model → rejected 4xx, not silently dropped (#255). Fields: `dropped` (`["file:"]`), `count`, `reason` (`model_not_vision_capable` \| `model_not_document_capable` \| `unsupported_media_type` \| `too_many_image_parts` \| `too_many_document_parts` \| `image_limit_exceeded` \| `document_limit_exceeded`). Gate: `Runner.checkInboundMedia`. **Images** (png/jpeg/gif/webp) on a **vision model** (`coreruntime.ModelSupportsVision`) and **PDFs** on a **doc model** (`ModelSupportsPDF` = Anthropic Sonnet 3.5+/Opus 4+; Haiku excluded, fail-closed), within the DoS bounds, are NOT rejected — `a2aMessageToLLM` projects them into `llm.ChatMessage.Parts` → Anthropic `image`/`document` source blocks, OpenAI `image_url` data URLs. Non-PDF docs/video still rejected; OpenAI-Responses PDF + extraction fallback are follow-ups. **DoS controls** (`media_limits.go` + `mediaSem`): body cap 32 MiB both transports; per-image ≤5 MiB & ≤50 MP/≤100k-per-side (`CheckImageLimits`); per-PDF ≤32 MiB + `%PDF-` sniff (`CheckDocumentLimits`); ≤20 images & ≤5 docs/msg; ≤4 concurrent media requests (excess shed with 429/unavailable) | +| `AuditInputMediaRejected` | `input_media_rejected` | Inbound `file` parts the runtime can't forward to the model → rejected 4xx, not silently dropped (#255). Fields: `dropped` (`["file:"]`), `count`, `reason` (`model_not_vision_capable` \| `model_not_document_capable` \| `unsupported_media_type` \| `too_many_image_parts` \| `too_many_document_parts` \| `image_limit_exceeded` \| `document_limit_exceeded`). Gate: `Runner.checkInboundMedia`. **Images** (png/jpeg/gif/webp) on a **vision model** (`coreruntime.ModelSupportsVision`) and **PDFs** on a **doc model** (`ModelSupportsPDF` = Anthropic Sonnet 3.5+/Opus 4+/Haiku 4.5+/Fable 5; Claude 3.0 & 3.5-Haiku excluded, fail-closed), within the DoS bounds, are NOT rejected — `a2aMessageToLLM` projects them into `llm.ChatMessage.Parts` → Anthropic `image`/`document` source blocks, OpenAI `image_url` data URLs. Non-PDF docs/video still rejected; OpenAI-Responses PDF + extraction fallback are follow-ups. **DoS controls** (`media_limits.go` + `mediaSem`): body cap 32 MiB both transports; per-image ≤5 MiB & ≤50 MP/≤100k-per-side (`CheckImageLimits`); per-PDF ≤32 MiB + `%PDF-` sniff (`CheckDocumentLimits`); ≤20 images & ≤5 docs/msg; ≤4 concurrent media requests (excess shed with 429/unavailable) | | `EventMCPServerStarted` | `mcp_server_started` | MCP server handshake succeeded | | `EventMCPServerFailed` | `mcp_server_failed` | MCP server dial / handshake failed | | `EventMCPServerDegraded` | `mcp_server_degraded` | MCP server in soft-fail | diff --git a/forge-core/runtime/model_capabilities.go b/forge-core/runtime/model_capabilities.go index ea8c5e4b..ee8a6f09 100644 --- a/forge-core/runtime/model_capabilities.go +++ b/forge-core/runtime/model_capabilities.go @@ -25,25 +25,25 @@ var visionCapablePrefixes = []string{ } // pdfCapablePrefixes lists model-name prefixes whose models accept a PDF -// document natively via Anthropic's document block (#255 Phase 3). Scoped -// conservatively to the SONNET and OPUS families, which are confirmed on -// Anthropic's PDF-support matrix: the 3.5/3.7 Sonnet releases plus the 4.x/5 -// Sonnet & Opus families (claude-sonnet-4…, claude-sonnet-5, claude-opus-4…). +// document natively via Anthropic's document block (#255 Phase 3). Covers the +// confirmed families: Sonnet (3.5/3.7 + 4.x/5), Opus (4.x), Haiku 4.5+, and +// Fable 5. // -// Deliberately NARROW (#534 review): -// - HAIKU is excluded — Anthropic's PDF support was Sonnet-first and Haiku's -// native-PDF coverage is unconfirmed; rejecting a Haiku PDF cleanly at the -// gate is preferable to sending it and getting an opaque provider error. -// (Fail closed; widen here once confirmed.) -// - Bare "claude-3" (Claude 3.0) is excluded — it predates PDF support. The -// 3.0 models are claude-3-{opus,sonnet,haiku}, which do NOT match the -// "claude-sonnet"/"claude-opus" family prefixes, so they fall through to a -// loud reject. The 3.5/3.7 prefixes below name "sonnet" explicitly so they -// don't catch claude-3-5-haiku. +// Prefix design (#534 review): +// - "claude-haiku" matches the 4.x/5 family naming (claude-haiku-4-5…), which +// Anthropic confirms supports native PDF. It does NOT match the older +// "claude-3-5-haiku" naming, so Haiku 3.5 (unconfirmed) stays excluded. +// - The 3.5/3.7 prefixes name "sonnet" explicitly so they don't catch +// claude-3-5-haiku. +// - Bare "claude-3" (Claude 3.0) is excluded — it predates PDF support. Its +// models are claude-3-{opus,sonnet,haiku}, which don't match the +// "claude-opus"/"claude-sonnet"/"claude-haiku" family prefixes, so they +// fall through to a loud reject. // +// Unknown-model default is fail-closed (loud reject, not a provider error). // OpenAI Responses input_file and Gemini document support are deferred follow-ups. var pdfCapablePrefixes = []string{ - "claude-3-5-sonnet", "claude-3-7-sonnet", "claude-opus", "claude-sonnet", + "claude-3-5-sonnet", "claude-3-7-sonnet", "claude-opus", "claude-sonnet", "claude-haiku", "claude-fable", } // ModelSupportsVision reports whether the named model accepts image input. diff --git a/forge-core/runtime/model_capabilities_test.go b/forge-core/runtime/model_capabilities_test.go index e436807d..7367223a 100644 --- a/forge-core/runtime/model_capabilities_test.go +++ b/forge-core/runtime/model_capabilities_test.go @@ -27,15 +27,14 @@ func TestModelSupportsVision(t *testing.T) { } func TestModelSupportsPDF(t *testing.T) { - pdf := []string{"claude-3-5-sonnet", "claude-3-7-sonnet", "claude-opus-4-8", "claude-sonnet-5", "claude-sonnet-4-5"} + pdf := []string{"claude-3-5-sonnet", "claude-3-7-sonnet", "claude-opus-4-8", "claude-sonnet-5", "claude-sonnet-4-5", "claude-haiku-4-5", "claude-fable-5"} for _, m := range pdf { if !ModelSupportsPDF(m) { t.Errorf("ModelSupportsPDF(%q) = false, want true", m) } } - // Haiku is deliberately excluded (Sonnet-first PDF support, Haiku unconfirmed - // → fail closed), as is bare Claude 3.0 and the 3.5 Haiku release. - noPDF := []string{"", "claude-3-opus", "claude-3-sonnet", "claude-3-5-haiku", "claude-haiku-4-5", "gpt-4o", "gpt-5", "gemini-2.5-flash", "o3-mini"} + // Excluded: bare Claude 3.0, the 3.5 Haiku release (unconfirmed), and non-Anthropic. + noPDF := []string{"", "claude-3-opus", "claude-3-sonnet", "claude-3-5-haiku", "gpt-4o", "gpt-5", "gemini-2.5-flash", "o3-mini"} for _, m := range noPDF { if ModelSupportsPDF(m) { t.Errorf("ModelSupportsPDF(%q) = true, want false", m) From f8c6312e36ff329488d7e360be027cf44e99f6aa Mon Sep 17 00:00:00 2001 From: MK Date: Mon, 28 Sep 2026 01:13:35 -0400 Subject: [PATCH 4/4] fix(runtime): Fable 5 is vision-capable too, not just PDF-capable (#534 review) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit 82c2ae5 added claude-fable to pdfCapablePrefixes but not visionCapablePrefixes, so Fable 5 accepted a PDF while rejecting an image — the inverse of every other multimodal Claude. Claude document support is built on vision infra, so a PDF-capable model is necessarily vision-capable. Add claude-fable to visionCapablePrefixes and, to stop this class of mismatch recurring, add TestPDFCapableImpliesVisionCapable asserting every pdfCapablePrefixes family is also vision-capable. --- forge-core/runtime/model_capabilities.go | 8 +++++--- forge-core/runtime/model_capabilities_test.go | 16 +++++++++++++++- 2 files changed, 20 insertions(+), 4 deletions(-) diff --git a/forge-core/runtime/model_capabilities.go b/forge-core/runtime/model_capabilities.go index ee8a6f09..9cbdaca4 100644 --- a/forge-core/runtime/model_capabilities.go +++ b/forge-core/runtime/model_capabilities.go @@ -17,9 +17,11 @@ var visionCapablePrefixes = []string{ "gpt-4o", "gpt-4.1", "gpt-4-turbo", "gpt-4-vision", "gpt-5", "o1", "o3", "o4", // Anthropic: Claude 3 and later are all vision-capable. The 4.x/5 models - // carry the family name (claude-opus-4…, claude-sonnet-5…) so the family - // prefixes cover them. - "claude-3", "claude-opus", "claude-sonnet", "claude-haiku", + // carry the family name (claude-opus-4…, claude-sonnet-5…, claude-fable-5) + // so the family prefixes cover them. Every pdfCapablePrefixes family is + // listed here too — Claude document support is built on vision infra, so a + // PDF-capable model is necessarily vision-capable. + "claude-3", "claude-opus", "claude-sonnet", "claude-haiku", "claude-fable", // Google Gemini (served through the OpenAI-compat client). "gemini-1.5", "gemini-2", } diff --git a/forge-core/runtime/model_capabilities_test.go b/forge-core/runtime/model_capabilities_test.go index 7367223a..0f53a7f3 100644 --- a/forge-core/runtime/model_capabilities_test.go +++ b/forge-core/runtime/model_capabilities_test.go @@ -6,7 +6,7 @@ func TestModelSupportsVision(t *testing.T) { vision := []string{ "gpt-4o", "gpt-4o-mini", "gpt-4.1", "gpt-4-turbo", "gpt-5", "o1", "o3-mini", "o4-mini", - "claude-3-5-sonnet", "claude-opus-4-8", "claude-sonnet-5", "claude-haiku-4-5", + "claude-3-5-sonnet", "claude-opus-4-8", "claude-sonnet-5", "claude-haiku-4-5", "claude-fable-5", "gemini-1.5-pro", "gemini-2.5-flash", "GPT-4O", // case-insensitive } @@ -42,6 +42,20 @@ func TestModelSupportsPDF(t *testing.T) { } } +// TestPDFCapableImpliesVisionCapable pins the invariant that every PDF-capable +// model is also vision-capable — Claude document support is built on vision +// infra, so a model declared document-capable must accept images too (#534 +// review). This catches the class of bug where a family is added to +// pdfCapablePrefixes but forgotten in visionCapablePrefixes. +func TestPDFCapableImpliesVisionCapable(t *testing.T) { + for _, prefix := range pdfCapablePrefixes { + sample := prefix + "-x" // a concrete model name in that family + if !ModelSupportsVision(sample) { + t.Errorf("model %q is PDF-capable but not vision-capable — mirror %q into visionCapablePrefixes", sample, prefix) + } + } +} + func TestIsDocumentMIME(t *testing.T) { if !IsDocumentMIME("application/pdf") || !IsDocumentMIME("APPLICATION/PDF ") { t.Error("application/pdf must be a document MIME")