From 0c8fe185918ea7154361af8bc24ec245de657195 Mon Sep 17 00:00:00 2001 From: MK Date: Thu, 24 Sep 2026 21:46:32 -0400 Subject: [PATCH] feat(llm): image input to vision models (#255 Phase 2) Forge now forwards inbound image file parts to the model as native vision input, instead of rejecting them at the ingest gate. Builds on Phase 0 (reject gate) + Phase 1 (ChatMessage.Parts). - Capability table (forge-core/runtime/model_capabilities.go): ModelSupportsVision (prefix map, mirrors ModelContextWindows) + IsImageMIME / NormalizeImageMIME. - Anthropic provider: anthropicContentBlock gains a Source field; convertMessage serializes ChatMessage.Parts as a block array (text + {type:image, source:{base64,media_type,data}}). - OpenAI provider: openaiMessage.Content string -> any; multimodal messages serialize as an image_url content-parts array (data: URLs). Gemini inherits this via the OpenAI-compat client. Text-only messages stay byte-identical. - Projection (a2aMessageToLLM): image file parts -> ChatMessage.Parts (text-of- record stays in Content for the guardrail/intent scanners; PromptText still omits file bytes). - Ingest gate (checkInboundMedia): images pass on a vision-capable model; images on a text-only model reject with reason model_not_vision_capable; documents/video still reject with unsupported_media_type. Fails closed when no model is resolved. Tests: capability + MIME matrix; Anthropic image-source + OpenAI image_url (and text-only byte-identical wire); projection (image->Parts, text-only nil, non-image skipped); gate matrix (image/vision accept, image/non-vision reject, document reject, text-only pass, nil-model fail-closed). Docs: runtime-engine multimodal section, audit-logging reason codes, forge.md skill (synced). --- .claude/skills/forge.md | 2 +- docs/core-concepts/runtime-engine.md | 8 +- docs/security/audit-logging.md | 2 +- forge-cli/internal/surface/knowledge/forge.md | 2 +- forge-cli/runtime/media_gate_test.go | 75 +++++++++++++ forge-cli/runtime/runner.go | 36 +++++- forge-core/llm/providers/anthropic.go | 57 ++++++++-- forge-core/llm/providers/multimodal_test.go | 105 ++++++++++++++++++ forge-core/llm/providers/openai.go | 57 ++++++++-- forge-core/runtime/loop.go | 31 +++++- forge-core/runtime/loop_projection_test.go | 74 ++++++++++++ forge-core/runtime/model_capabilities.go | 62 +++++++++++ forge-core/runtime/model_capabilities_test.go | 49 ++++++++ 13 files changed, 535 insertions(+), 25 deletions(-) create mode 100644 forge-cli/runtime/media_gate_test.go create mode 100644 forge-core/llm/providers/multimodal_test.go create mode 100644 forge-core/runtime/loop_projection_test.go create mode 100644 forge-core/runtime/model_capabilities.go create mode 100644 forge-core/runtime/model_capabilities_test.go diff --git a/.claude/skills/forge.md b/.claude/skills/forge.md index 9cfe1bdf..1c6cf0b6 100644 --- a/.claude/skills/forge.md +++ b/.claude/skills/forge.md @@ -1200,7 +1200,7 @@ when OTel tracing is enabled (OTel v1 / Phase 4 / #105). Both use | `AuditScheduleModify` | `schedule_modify` | Schedule mutated at runtime | | `EventAuthVerify` | `auth_verify` | Inbound request authenticated (`provider`, `user_id`, `org_id`, `token_kind`; `email` when the identity carries one). **Channel invoker:** for a channel-originated request the transport credential is the loopback token (`provider:internal`/`user_id:forge-internal`, recorded truthfully) and the human sender is stamped as `channel`/`channel_user`/`channel_email` from the `X-Forge-Channel*` headers — honored only for the runtime-internal identity (same trust gate as `applyChannelOnBehalfOf`). Slack/Teams resolve `channel_email`; Telegram (numeric id) & WhatsApp (msisdn) carry `channel_user` only | | `EventAuthFail` | `auth_fail` | Inbound request rejected (`reason`, `token_kind`) | -| `AuditInputMediaRejected` | `input_media_rejected` | Inbound message carried `file` parts the runtime can't forward to the model → rejected 4xx instead of silently dropped (#255). Fields: `dropped` (`["file:"]`), `count`, `reason`. Gate: `Runner.checkInboundMedia` at the four send handlers; `a2a.Message.FileParts()` | +| `AuditInputMediaRejected` | `input_media_rejected` | Inbound `file` parts the runtime can't forward to the model → rejected 4xx, not silently dropped (#255). Fields: `dropped` (`["file:"]`), `count`, `reason` (`model_not_vision_capable` \| `unsupported_media_type`). Gate: `Runner.checkInboundMedia`. **Images** (png/jpeg/gif/webp) on a **vision model** (`coreruntime.ModelSupportsVision`) are NOT rejected — `a2aMessageToLLM` projects them into `llm.ChatMessage.Parts` → Anthropic `image` source blocks / OpenAI `image_url` data URLs. Docs/video still rejected (Phase 3+) | | `EventMCPServerStarted` | `mcp_server_started` | MCP server handshake succeeded | | `EventMCPServerFailed` | `mcp_server_failed` | MCP server dial / handshake failed | | `EventMCPServerDegraded` | `mcp_server_degraded` | MCP server in soft-fail | diff --git a/docs/core-concepts/runtime-engine.md b/docs/core-concepts/runtime-engine.md index a6346f82..648ffe83 100644 --- a/docs/core-concepts/runtime-engine.md +++ b/docs/core-concepts/runtime-engine.md @@ -29,10 +29,16 @@ An inbound A2A message is a list of typed parts. `a2a.Message.PromptText()` proj - **Text parts** are included verbatim (joined by newlines). - **Data parts** (`{"kind":"data","data":{…}}`) are appended as a fenced ` ```json ` block of the part's fields. This means structured input — e.g. a workflow step's output dispatched as a data part with no text part — still reaches the model instead of producing an empty prompt (a message with only a data part must never execute as empty). -- **File parts** are not projected into prompt text (their bytes/URIs aren't text). +- **File parts** are not projected into prompt *text* (their bytes/URIs aren't text). Image file parts are handled separately as multimodal input (below). The same projection feeds the inbound guardrail and intent-alignment scanners, so the security checks see exactly what the model sees — a payload carried in a data part can't reach the LLM while bypassing them. Each data part's projected block is capped (~16KiB, rune-safe) — the cap applies identically to the scanners and the prompt, so truncation can't open a divergence. +#### Image input (multimodal) + +An image `file` part (`image/png`, `image/jpeg`, `image/gif`, `image/webp`) is forwarded to the model as native vision input when the resolved model is **vision-capable** (`runtime.ModelSupportsVision` — OpenAI `gpt-4o`/`gpt-4.1`/`gpt-5`/`o1`/`o3`/`o4`, Anthropic Claude 3+, Gemini 1.5/2). `a2aMessageToLLM` projects such parts into `llm.ChatMessage.Parts` (the flattened text stays in `Content` as the text-of-record for the scanners), and each provider serializes them natively — Anthropic `image` source blocks, OpenAI/Gemini `image_url` data URLs. A text-only message keeps `Parts` empty and marshals byte-identically to before. + +Media the model can't consume is **rejected loudly, never silently dropped** (the `checkInboundMedia` ingest gate): an image on a text-only model, or a document/video part (not yet supported), returns a 4xx and emits the `input_media_rejected` audit event. Note the image **bytes** themselves are not text-scannable, so guardrail/intent scanning still applies only to the text/data projection; this is an accepted limitation. + **Note for guardrail pattern authors:** parts join with **newlines** (matching what the model sees). A pattern intended to match content that may span a part boundary should use `\s+` rather than a literal space — a payload split across two text parts joins as `…end\nstart…`. ### Session Recovery Deduplication diff --git a/docs/security/audit-logging.md b/docs/security/audit-logging.md index ac2cdfa6..26dfd4dc 100644 --- a/docs/security/audit-logging.md +++ b/docs/security/audit-logging.md @@ -32,7 +32,7 @@ All runtime security events are emitted as structured NDJSON to stderr with corr | `mcp_auth_resolved` | The parked call's consent arrived (or the wait was canceled) and it resumed (#330). Carries `server`, `subject`, `wait_ms`. Emitted **once**, attributed to the parked invocation (#366). | | `mcp_auth_timeout` | No consent within the window; the parked MCP call fails `no_token` (#330). Carries `server`, `subject`, `wait_ms`, `decision`. | | `auth_fail` | Inbound request rejected (with `reason`, `token_kind`). No `task_id` (none is ever created), but carries `workflow_execution_id` when the request had the execution header — so a rejected request is still attributable to its workflow run (#278). | -| `input_media_rejected` | An inbound `tasks/send`/`sendSubscribe` message carried one or more `file` parts (image/document/…) the runtime cannot forward to the model, so the request was rejected with a 4xx (JSON-RPC invalid-params / HTTP 400) instead of silently dropping the attachment and returning a plausible answer that ignored it (#255). Carries `fields.dropped` (`["file:", …]`), `fields.count`, and `fields.reason` (`file_input_unsupported`). Multimodal input arrives in later slices; until then this makes the accepted-but-dropped case loud. | +| `input_media_rejected` | An inbound `tasks/send`/`sendSubscribe` message carried `file` parts the runtime cannot forward to the model, so the request was rejected with a 4xx (JSON-RPC invalid-params / HTTP 400) instead of silently dropping the attachment and returning a plausible answer that ignored it (#255). Carries `fields.dropped` (`["file:", …]`), `fields.count`, and `fields.reason` — `model_not_vision_capable` (an image sent to a text-only model) or `unsupported_media_type` (documents/video — not yet accepted). **Images** (`png`/`jpeg`/`gif`/`webp`) sent to a **vision-capable model** are NOT rejected: they are projected into the model request as inline vision input and do not emit this event. | | `agent_card_published` | Agent Card finalized at startup or hot-reload (with `name`, `version`, `protocol_version`, `url`, `skill_count`, `capabilities`, `security_schemes`, `card_size_bytes`, `card_sha256`). See [Agent Card reference](../reference/a2a-agent-card.md). | | `policy_loaded` | One per non-empty policy layer at startup (system / user / workspace). Carries `fields.layer`, `source` (file path), deny-list size counts, and max bounds. See [Platform Policy](platform-policy.md). | | `policy_violation_at_build_time` | One per violation when `forge.yaml` conflicts with any policy layer. Agent refuses to start. Carries `fields.violation_kind` / `offending_value` / `forge_yaml_field` plus `layer` + `source` identifying the enforcing file. See [Platform Policy](platform-policy.md). | diff --git a/forge-cli/internal/surface/knowledge/forge.md b/forge-cli/internal/surface/knowledge/forge.md index 9cfe1bdf..1c6cf0b6 100644 --- a/forge-cli/internal/surface/knowledge/forge.md +++ b/forge-cli/internal/surface/knowledge/forge.md @@ -1200,7 +1200,7 @@ when OTel tracing is enabled (OTel v1 / Phase 4 / #105). Both use | `AuditScheduleModify` | `schedule_modify` | Schedule mutated at runtime | | `EventAuthVerify` | `auth_verify` | Inbound request authenticated (`provider`, `user_id`, `org_id`, `token_kind`; `email` when the identity carries one). **Channel invoker:** for a channel-originated request the transport credential is the loopback token (`provider:internal`/`user_id:forge-internal`, recorded truthfully) and the human sender is stamped as `channel`/`channel_user`/`channel_email` from the `X-Forge-Channel*` headers — honored only for the runtime-internal identity (same trust gate as `applyChannelOnBehalfOf`). Slack/Teams resolve `channel_email`; Telegram (numeric id) & WhatsApp (msisdn) carry `channel_user` only | | `EventAuthFail` | `auth_fail` | Inbound request rejected (`reason`, `token_kind`) | -| `AuditInputMediaRejected` | `input_media_rejected` | Inbound message carried `file` parts the runtime can't forward to the model → rejected 4xx instead of silently dropped (#255). Fields: `dropped` (`["file:"]`), `count`, `reason`. Gate: `Runner.checkInboundMedia` at the four send handlers; `a2a.Message.FileParts()` | +| `AuditInputMediaRejected` | `input_media_rejected` | Inbound `file` parts the runtime can't forward to the model → rejected 4xx, not silently dropped (#255). Fields: `dropped` (`["file:"]`), `count`, `reason` (`model_not_vision_capable` \| `unsupported_media_type`). Gate: `Runner.checkInboundMedia`. **Images** (png/jpeg/gif/webp) on a **vision model** (`coreruntime.ModelSupportsVision`) are NOT rejected — `a2aMessageToLLM` projects them into `llm.ChatMessage.Parts` → Anthropic `image` source blocks / OpenAI `image_url` data URLs. Docs/video still rejected (Phase 3+) | | `EventMCPServerStarted` | `mcp_server_started` | MCP server handshake succeeded | | `EventMCPServerFailed` | `mcp_server_failed` | MCP server dial / handshake failed | | `EventMCPServerDegraded` | `mcp_server_degraded` | MCP server in soft-fail | diff --git a/forge-cli/runtime/media_gate_test.go b/forge-cli/runtime/media_gate_test.go new file mode 100644 index 00000000..4170c7fa --- /dev/null +++ b/forge-cli/runtime/media_gate_test.go @@ -0,0 +1,75 @@ +package runtime + +import ( + "context" + "strings" + "testing" + + "github.com/initializ/forge/forge-core/a2a" + "github.com/initializ/forge/forge-core/llm" + coreruntime "github.com/initializ/forge/forge-core/runtime" +) + +func runnerWithModel(model string) *Runner { + return &Runner{modelConfig: &coreruntime.ModelConfig{Client: llm.ClientConfig{Model: model}}} +} + +func imagePartMsg(mime string) a2a.Message { + return a2a.Message{ + Role: a2a.MessageRoleUser, + Parts: []a2a.Part{ + a2a.NewTextPart("look"), + a2a.NewFilePart(a2a.FileContent{MimeType: mime, Bytes: []byte{1, 2, 3}}), + }, + } +} + +// TestCheckInboundMedia_Phase2 covers the flipped gate: images pass on a +// vision-capable model, and everything else is still rejected loudly (#255). +func TestCheckInboundMedia_Phase2(t *testing.T) { + ctx := context.Background() + + t.Run("image accepted on vision model", func(t *testing.T) { + r := runnerWithModel("gpt-4o") + if got := r.checkInboundMedia(ctx, imagePartMsg("image/png"), nil); got != "" { + t.Errorf("image on a vision model must be accepted, got reject: %q", got) + } + }) + + t.Run("image rejected on non-vision model", func(t *testing.T) { + r := runnerWithModel("gpt-3.5-turbo") + got := r.checkInboundMedia(ctx, imagePartMsg("image/png"), nil) + if got == "" { + t.Fatal("image on a non-vision model must be rejected") + } + if !strings.Contains(got, "image/png") || !strings.Contains(got, "does not support image") { + t.Errorf("reject message should name the mime and the non-vision reason; got %q", got) + } + }) + + t.Run("document rejected even on vision model", func(t *testing.T) { + r := runnerWithModel("gpt-4o") + got := r.checkInboundMedia(ctx, imagePartMsg("application/pdf"), nil) + if got == "" { + t.Fatal("a document part must still be rejected in Phase 2") + } + if !strings.Contains(got, "application/pdf") { + t.Errorf("reject message should name the pdf mime; got %q", got) + } + }) + + t.Run("text-only accepted", func(t *testing.T) { + r := runnerWithModel("gpt-3.5-turbo") + msg := a2a.Message{Role: a2a.MessageRoleUser, Parts: []a2a.Part{a2a.NewTextPart("hi")}} + if got := r.checkInboundMedia(ctx, msg, nil); got != "" { + t.Errorf("text-only must never be gated, got %q", got) + } + }) + + t.Run("nil modelConfig treats as non-vision", func(t *testing.T) { + r := &Runner{} + if got := r.checkInboundMedia(ctx, imagePartMsg("image/png"), nil); got == "" { + t.Error("with no resolved model, image input must be rejected (fail closed)") + } + }) +} diff --git a/forge-cli/runtime/runner.go b/forge-cli/runtime/runner.go index 8e79f3d5..a3feb969 100644 --- a/forge-cli/runtime/runner.go +++ b/forge-cli/runtime/runner.go @@ -2317,27 +2317,53 @@ func (r *Runner) checkInboundMedia(ctx context.Context, msg a2a.Message, auditLo if len(files) == 0 { return "" } - dropped := make([]string, 0, len(files)) + // Image parts are forwarded to the model as inline vision input when the + // resolved model is vision-capable (#255 Phase 2). Everything else — images + // on a non-vision model, and documents/video (later phases) — is still + // rejected loudly rather than silently dropped. + visionCapable := r.modelConfig != nil && coreruntime.ModelSupportsVision(r.modelConfig.Client.Model) + + var dropped []string + reason := "unsupported_media_type" for _, f := range files { mt := f.MimeType if mt == "" { mt = "application/octet-stream" } + if coreruntime.IsImageMIME(mt) { + if visionCapable { + continue // accepted → projected to the model as vision input + } + reason = "model_not_vision_capable" + } dropped = append(dropped, "file:"+mt) } + if len(dropped) == 0 { + return "" // all file parts are images the model can consume + } if auditLogger != nil { auditLogger.EmitFromContext(ctx, coreruntime.AuditEvent{ Event: coreruntime.AuditInputMediaRejected, Fields: map[string]any{ "dropped": dropped, - "count": len(files), - "reason": "file_input_unsupported", + "count": len(dropped), + "reason": reason, }, }) } + var detail string + if reason == "model_not_vision_capable" { + model := "" + if r.modelConfig != nil { + model = r.modelConfig.Client.Model + } + detail = fmt.Sprintf("the configured model %q does not support image input", model) + } else { + detail = "only image input (png/jpeg/gif/webp) is accepted; documents and video are not yet supported" + } return fmt.Sprintf( - "this agent does not accept file/image input: %d file part(s) rejected (%s). Send text or data parts instead.", - len(files), strings.Join(dropped, ", "), + "%d media part(s) rejected (%s): %s.", + len(dropped), strings.Join(dropped, ", "), detail, ) } diff --git a/forge-core/llm/providers/anthropic.go b/forge-core/llm/providers/anthropic.go index f990975a..3add08b2 100644 --- a/forge-core/llm/providers/anthropic.go +++ b/forge-core/llm/providers/anthropic.go @@ -4,6 +4,7 @@ import ( "bufio" "bytes" "context" + "encoding/base64" "encoding/json" "fmt" "io" @@ -230,13 +231,22 @@ type anthropicMessage struct { } type anthropicContentBlock struct { - Type string `json:"type"` - Text string `json:"text,omitempty"` - ID string `json:"id,omitempty"` - Name string `json:"name,omitempty"` - Input json.RawMessage `json:"input,omitempty"` - ToolUseID string `json:"tool_use_id,omitempty"` - Content string `json:"content,omitempty"` + Type string `json:"type"` + Text string `json:"text,omitempty"` + ID string `json:"id,omitempty"` + Name string `json:"name,omitempty"` + Input json.RawMessage `json:"input,omitempty"` + ToolUseID string `json:"tool_use_id,omitempty"` + Content string `json:"content,omitempty"` + Source *anthropicImageSource `json:"source,omitempty"` // type=="image" (#255) +} + +// anthropicImageSource is the source of an image content block: +// {"type":"image","source":{"type":"base64","media_type":"image/png","data":"…"}}. +type anthropicImageSource struct { + Type string `json:"type"` // "base64" + MediaType string `json:"media_type"` // e.g. image/png + Data string `json:"data"` // base64-encoded bytes } type anthropicTool struct { @@ -343,11 +353,44 @@ func (c *AnthropicClient) convertMessage(m llm.ChatMessage) anthropicMessage { return anthropicMessage{Role: "assistant", Content: data} } + // Multimodal message: serialize content-parts as a block array (#255). + if len(m.Parts) > 0 { + data, _ := json.Marshal(anthropicBlocksFromParts(m.Parts)) + return anthropicMessage{Role: role, Content: data} + } + // Simple text message data, _ := json.Marshal(m.Content) return anthropicMessage{Role: role, Content: data} } +// anthropicBlocksFromParts maps provider-agnostic content parts to Anthropic +// content blocks: text → {type:text}, image → {type:image, source:{base64}}. +// A media part with no inline bytes is skipped (rehydration is the caller's +// responsibility before the request is built). +func anthropicBlocksFromParts(parts []llm.ContentPart) []anthropicContentBlock { + blocks := make([]anthropicContentBlock, 0, len(parts)) + for _, p := range parts { + switch p.Type { + case llm.ContentPartImage: + if p.Media == nil || len(p.Media.Bytes) == 0 { + continue + } + blocks = append(blocks, anthropicContentBlock{ + Type: "image", + Source: &anthropicImageSource{ + Type: "base64", + MediaType: p.Media.MimeType, + Data: base64.StdEncoding.EncodeToString(p.Media.Bytes), + }, + }) + default: // text + blocks = append(blocks, anthropicContentBlock{Type: "text", Text: p.Text}) + } + } + return blocks +} + // Anthropic-specific response types. type anthropicResponse struct { ID string `json:"id"` diff --git a/forge-core/llm/providers/multimodal_test.go b/forge-core/llm/providers/multimodal_test.go new file mode 100644 index 00000000..fa3a762b --- /dev/null +++ b/forge-core/llm/providers/multimodal_test.go @@ -0,0 +1,105 @@ +package providers + +import ( + "encoding/base64" + "encoding/json" + "strings" + "testing" + + "github.com/initializ/forge/forge-core/llm" +) + +var testImageBytes = []byte{0x89, 0x50, 0x4e, 0x47, 0x0d, 0x0a} // PNG magic-ish + +func imageMessage() llm.ChatMessage { + return llm.ChatMessage{ + Role: llm.RoleUser, + Content: "what is in this image?", + Parts: []llm.ContentPart{ + llm.NewTextContentPart("what is in this image?"), + llm.NewMediaContentPart(llm.ContentPartImage, llm.MediaRef{MimeType: "image/png", Bytes: testImageBytes}), + }, + } +} + +// TestAnthropic_ImagePartBecomesSourceBlock verifies a multimodal user message +// serializes to an Anthropic content-block array with a base64 image source +// (#255 Phase 2). +func TestAnthropic_ImagePartBecomesSourceBlock(t *testing.T) { + c := NewAnthropicClient(llm.ClientConfig{Model: "claude-sonnet-5"}) + body := c.toAnthropicRequest(&llm.ChatRequest{Messages: []llm.ChatMessage{imageMessage()}}, false) + + if len(body.Messages) != 1 { + t.Fatalf("got %d messages", len(body.Messages)) + } + var blocks []anthropicContentBlock + if err := json.Unmarshal(body.Messages[0].Content, &blocks); err != nil { + t.Fatalf("content should be a block array: %v", err) + } + if len(blocks) != 2 || blocks[0].Type != "text" || blocks[1].Type != "image" { + t.Fatalf("blocks = %+v, want [text, image]", blocks) + } + src := blocks[1].Source + if src == nil || src.Type != "base64" || src.MediaType != "image/png" { + t.Fatalf("image source malformed: %+v", src) + } + if src.Data != base64.StdEncoding.EncodeToString(testImageBytes) { + t.Errorf("image data not base64 of the bytes") + } +} + +// TestAnthropic_TextOnlyWireUnchanged pins that a text-only message still +// marshals its content as a bare JSON string (no block array) — byte-identical +// to the pre-#255 wire and prompt-cache prefix. +func TestAnthropic_TextOnlyWireUnchanged(t *testing.T) { + c := NewAnthropicClient(llm.ClientConfig{Model: "claude-sonnet-5"}) + body := c.toAnthropicRequest(&llm.ChatRequest{Messages: []llm.ChatMessage{{Role: llm.RoleUser, Content: "hello"}}}, false) + if got := string(body.Messages[0].Content); got != `"hello"` { + t.Errorf("text-only content = %s, want \"hello\"", got) + } +} + +// TestOpenAI_ImagePartBecomesImageURL verifies a multimodal message serializes +// to an OpenAI content-parts array with an inline data: image_url. +func TestOpenAI_ImagePartBecomesImageURL(t *testing.T) { + c := NewOpenAIClient(llm.ClientConfig{Model: "gpt-4o"}) + body := c.toOpenAIRequest(&llm.ChatRequest{Messages: []llm.ChatMessage{imageMessage()}}, false) + + parts, ok := body.Messages[0].Content.([]openaiContentPart) + if !ok { + t.Fatalf("content should be []openaiContentPart, got %T", body.Messages[0].Content) + } + if len(parts) != 2 || parts[0].Type != "text" || parts[1].Type != "image_url" { + t.Fatalf("parts = %+v, want [text, image_url]", parts) + } + wantURL := "data:image/png;base64," + base64.StdEncoding.EncodeToString(testImageBytes) + if parts[1].ImageURL == nil || parts[1].ImageURL.URL != wantURL { + t.Errorf("image_url = %+v, want %s", parts[1].ImageURL, wantURL) + } +} + +// TestOpenAI_TextOnlyWireUnchanged pins the byte-identical text-only wire: a +// plain string content, no parts array. +func TestOpenAI_TextOnlyWireUnchanged(t *testing.T) { + c := NewOpenAIClient(llm.ClientConfig{Model: "gpt-4o"}) + body := c.toOpenAIRequest(&llm.ChatRequest{Messages: []llm.ChatMessage{{Role: llm.RoleUser, Content: "hello"}}}, false) + data, err := json.Marshal(body.Messages[0]) + if err != nil { + t.Fatal(err) + } + if !strings.Contains(string(data), `"content":"hello"`) { + t.Errorf("text-only message must marshal content as a bare string; got %s", data) + } +} + +// TestOpenAI_AssistantToolCallOmitsContent pins the existing invariant that an +// assistant tool-call-only turn still omits content entirely (not ""). +func TestOpenAI_AssistantToolCallOmitsContent(t *testing.T) { + c := NewOpenAIClient(llm.ClientConfig{Model: "gpt-4o"}) + msg := llm.ChatMessage{Role: "assistant", ToolCalls: []llm.ToolCall{{ID: "c1", Type: "function", Function: llm.FunctionCall{Name: "x", Arguments: "{}"}}}} + body := c.toOpenAIRequest(&llm.ChatRequest{Messages: []llm.ChatMessage{msg}}, false) + data, _ := json.Marshal(body.Messages[0]) + if strings.Contains(string(data), `"content"`) { + t.Errorf("assistant tool-call-only turn must omit content; got %s", data) + } +} diff --git a/forge-core/llm/providers/openai.go b/forge-core/llm/providers/openai.go index bba1bb28..25aff126 100644 --- a/forge-core/llm/providers/openai.go +++ b/forge-core/llm/providers/openai.go @@ -6,6 +6,7 @@ import ( "bytes" "context" "crypto/sha256" + "encoding/base64" "encoding/hex" "encoding/json" "fmt" @@ -180,13 +181,28 @@ type streamOptions struct { } type openaiMessage struct { + // Content is either a plain string (text-only, the common/back-compat + // case) or a []openaiContentPart for multimodal messages (#255). Kept as + // `any` so a text-only message marshals byte-identically to before. Role string `json:"role"` - Content *string `json:"content,omitempty"` + Content any `json:"content,omitempty"` ToolCalls []llm.ToolCall `json:"tool_calls,omitempty"` ToolCallID string `json:"tool_call_id,omitempty"` Name string `json:"name,omitempty"` } +// openaiContentPart is one element of a multimodal message content array: +// {"type":"text","text":…} or {"type":"image_url","image_url":{"url":…}}. +type openaiContentPart struct { + Type string `json:"type"` + Text string `json:"text,omitempty"` + ImageURL *openaiImageURL `json:"image_url,omitempty"` +} + +type openaiImageURL struct { + URL string `json:"url"` // a data: URL (data:;base64,) or http(s) URL +} + func (c *OpenAIClient) toOpenAIRequest(req *llm.ChatRequest, stream bool) openaiRequest { model := req.Model if model == "" { @@ -201,13 +217,17 @@ func (c *OpenAIClient) toOpenAIRequest(req *llm.ChatRequest, stream bool) openai ToolCallID: m.ToolCallID, Name: m.Name, } - // Assistant messages with tool_calls may omit content (valid per OpenAI spec). - // All other roles must always include content as a string. - if m.Role == "assistant" && len(m.ToolCalls) > 0 && m.Content == "" { - // Leave Content nil — omitempty will omit the field entirely. - } else { - content := m.Content - msg.Content = &content + switch { + case len(m.Parts) > 0: + // Multimodal message: content is an array of text/image parts (#255). + msg.Content = openaiPartsFromContent(m.Parts) + case m.Role == "assistant" && len(m.ToolCalls) > 0 && m.Content == "": + // Leave Content nil — omitempty omits the field entirely (valid + // per OpenAI spec for a tool-call-only assistant turn). + default: + // All other roles include content as a plain string (byte-identical + // to the pre-#255 wire). + msg.Content = m.Content } msgs[i] = msg } @@ -232,6 +252,27 @@ func (c *OpenAIClient) toOpenAIRequest(req *llm.ChatRequest, stream bool) openai return r } +// openaiPartsFromContent maps provider-agnostic content parts to OpenAI +// message-content parts: text → {type:text}, image → {type:image_url} with the +// bytes inlined as a data: URL. A media part with no inline bytes is skipped +// (rehydration is the caller's responsibility before the request is built). +func openaiPartsFromContent(parts []llm.ContentPart) []openaiContentPart { + out := make([]openaiContentPart, 0, len(parts)) + for _, p := range parts { + switch p.Type { + case llm.ContentPartImage: + if p.Media == nil || len(p.Media.Bytes) == 0 { + continue + } + dataURL := fmt.Sprintf("data:%s;base64,%s", p.Media.MimeType, base64.StdEncoding.EncodeToString(p.Media.Bytes)) + out = append(out, openaiContentPart{Type: "image_url", ImageURL: &openaiImageURL{URL: dataURL}}) + default: // text + out = append(out, openaiContentPart{Type: "text", Text: p.Text}) + } + } + return out +} + // derivePromptCacheKey builds a stable cache-routing key from the parts of // the request that define the cacheable prefix: model, system prompt, and // tool names. Identical (model, system, tools) across turns → identical key diff --git a/forge-core/runtime/loop.go b/forge-core/runtime/loop.go index 42d0a893..e5d51fc2 100644 --- a/forge-core/runtime/loop.go +++ b/forge-core/runtime/loop.go @@ -1085,10 +1085,39 @@ func a2aMessageToLLM(msg a2a.Message) llm.ChatMessage { role = llm.RoleAssistant } - return llm.ChatMessage{ + out := llm.ChatMessage{ Role: role, Content: msg.PromptText(), } + + // Project image file parts into multimodal content parts (#255 Phase 2). + // Content (PromptText) stays the flattened text-of-record — authoritative + // for the guardrail/intent scanners and back-compat. Parts is set only when + // the message actually carries a supported image, and then it carries the + // projected text as one block plus each image, so the provider serializes + // the full turn without dropping the text. Non-image file parts never reach + // here: the ingest gate (checkInboundMedia) already rejected them. + var images []llm.ContentPart + for _, p := range msg.Parts { + if p.Kind != a2a.PartKindFile || p.File == nil { + continue + } + if !IsImageMIME(p.File.MimeType) || len(p.File.Bytes) == 0 { + continue + } + images = append(images, llm.NewMediaContentPart(llm.ContentPartImage, llm.MediaRef{ + MimeType: NormalizeImageMIME(p.File.MimeType), + Bytes: p.File.Bytes, + })) + } + if len(images) > 0 { + parts := make([]llm.ContentPart, 0, len(images)+1) + if out.Content != "" { + parts = append(parts, llm.NewTextContentPart(out.Content)) + } + out.Parts = append(parts, images...) + } + return out } // a2aMessagesEqual reports whether two A2A messages have the same role diff --git a/forge-core/runtime/loop_projection_test.go b/forge-core/runtime/loop_projection_test.go new file mode 100644 index 00000000..806c2ea7 --- /dev/null +++ b/forge-core/runtime/loop_projection_test.go @@ -0,0 +1,74 @@ +package runtime + +import ( + "testing" + + "github.com/initializ/forge/forge-core/a2a" + "github.com/initializ/forge/forge-core/llm" +) + +// TestA2AMessageToLLM_ProjectsImageParts verifies image file parts become +// multimodal ContentParts (text-of-record + image), while Content stays the +// flattened PromptText for the scanners (#255 Phase 2). +func TestA2AMessageToLLM_ProjectsImageParts(t *testing.T) { + imgBytes := []byte{1, 2, 3, 4} + msg := a2a.Message{ + Role: a2a.MessageRoleUser, + Parts: []a2a.Part{ + a2a.NewTextPart("describe this"), + a2a.NewFilePart(a2a.FileContent{Name: "p.jpg", MimeType: "image/jpg", Bytes: imgBytes}), + }, + } + got := a2aMessageToLLM(msg) + + if got.Role != llm.RoleUser { + t.Errorf("role = %q", got.Role) + } + if got.Content != "describe this" { + t.Errorf("Content (text-of-record) = %q, want %q", got.Content, "describe this") + } + if len(got.Parts) != 2 { + t.Fatalf("Parts = %+v, want [text, image]", got.Parts) + } + if got.Parts[0].Type != llm.ContentPartText || got.Parts[0].Text != "describe this" { + t.Errorf("first part should be the projected text; got %+v", got.Parts[0]) + } + img := got.Parts[1] + if img.Type != llm.ContentPartImage || img.Media == nil { + t.Fatalf("second part should be an image media part; got %+v", img) + } + if img.Media.MimeType != "image/jpeg" { // image/jpg normalized + t.Errorf("image mime = %q, want image/jpeg (normalized)", img.Media.MimeType) + } + if string(img.Media.Bytes) != string(imgBytes) { + t.Errorf("image bytes not carried through") + } +} + +// TestA2AMessageToLLM_TextOnlyLeavesPartsNil pins that a text-only message +// keeps Parts nil (providers use the plain Content string — back-compat). +func TestA2AMessageToLLM_TextOnlyLeavesPartsNil(t *testing.T) { + got := a2aMessageToLLM(a2a.Message{Role: a2a.MessageRoleUser, Parts: []a2a.Part{a2a.NewTextPart("hi")}}) + if got.Parts != nil { + t.Errorf("text-only message must leave Parts nil, got %+v", got.Parts) + } + if got.Content != "hi" { + t.Errorf("Content = %q, want hi", got.Content) + } +} + +// TestA2AMessageToLLM_NonImageFileNotProjected: a non-image file part (the gate +// would reject it) must not be projected into Parts if it somehow reaches here. +func TestA2AMessageToLLM_NonImageFileNotProjected(t *testing.T) { + msg := a2a.Message{ + Role: a2a.MessageRoleUser, + Parts: []a2a.Part{ + a2a.NewTextPart("read this"), + a2a.NewFilePart(a2a.FileContent{Name: "d.pdf", MimeType: "application/pdf", Bytes: []byte{1}}), + }, + } + got := a2aMessageToLLM(msg) + if got.Parts != nil { + t.Errorf("non-image file part must not be projected; got %+v", got.Parts) + } +} diff --git a/forge-core/runtime/model_capabilities.go b/forge-core/runtime/model_capabilities.go new file mode 100644 index 00000000..aef3d790 --- /dev/null +++ b/forge-core/runtime/model_capabilities.go @@ -0,0 +1,62 @@ +package runtime + +import "strings" + +// visionCapablePrefixes lists model-name prefixes whose models accept image +// input via their provider's native multimodal API (#255). It mirrors the +// prefix-map style of ModelContextWindows and is the single runtime source of +// truth for image capability — the catalog is display-only and can't describe +// Claude (its Models list is empty). +// +// Conservative by design: a model NOT matched here is treated as text-only, so +// an image sent to it is rejected loudly at ingest rather than silently dropped +// or sent as a malformed request. Base "gpt-4"/"gpt-3.5" are intentionally +// absent (only the vision variants are listed). +var visionCapablePrefixes = []string{ + // OpenAI (chat + reasoning models with vision). + "gpt-4o", "gpt-4.1", "gpt-4-turbo", "gpt-4-vision", "gpt-5", + "o1", "o3", "o4", + // Anthropic: Claude 3 and later are all vision-capable. The 4.x/5 models + // carry the family name (claude-opus-4…, claude-sonnet-5…) so the family + // prefixes cover them. + "claude-3", "claude-opus", "claude-sonnet", "claude-haiku", + // Google Gemini (served through the OpenAI-compat client). + "gemini-1.5", "gemini-2", +} + +// ModelSupportsVision reports whether the named model accepts image input. +// Case-insensitive prefix match; unknown/empty models return false. +func ModelSupportsVision(model string) bool { + m := strings.ToLower(strings.TrimSpace(model)) + if m == "" { + return false + } + for _, p := range visionCapablePrefixes { + if strings.HasPrefix(m, p) { + return true + } + } + return false +} + +// IsImageMIME reports whether a MIME type denotes an image forge can forward as +// inline vision input. Restricted to the formats the vision providers accept +// (Anthropic + OpenAI both support png/jpeg/gif/webp). +func IsImageMIME(mime string) bool { + switch NormalizeImageMIME(mime) { + case "image/png", "image/jpeg", "image/gif", "image/webp": + return true + } + return false +} + +// NormalizeImageMIME lowercases and canonicalizes an image MIME type — notably +// mapping the common non-standard "image/jpg" to "image/jpeg", which is the +// form Anthropic's image source block requires. +func NormalizeImageMIME(mime string) string { + m := strings.ToLower(strings.TrimSpace(mime)) + if m == "image/jpg" { + return "image/jpeg" + } + return m +} diff --git a/forge-core/runtime/model_capabilities_test.go b/forge-core/runtime/model_capabilities_test.go new file mode 100644 index 00000000..099b7adf --- /dev/null +++ b/forge-core/runtime/model_capabilities_test.go @@ -0,0 +1,49 @@ +package runtime + +import "testing" + +func TestModelSupportsVision(t *testing.T) { + vision := []string{ + "gpt-4o", "gpt-4o-mini", "gpt-4.1", "gpt-4-turbo", "gpt-5", + "o1", "o3-mini", "o4-mini", + "claude-3-5-sonnet", "claude-opus-4-8", "claude-sonnet-5", "claude-haiku-4-5", + "gemini-1.5-pro", "gemini-2.5-flash", + "GPT-4O", // case-insensitive + } + for _, m := range vision { + if !ModelSupportsVision(m) { + t.Errorf("ModelSupportsVision(%q) = false, want true", m) + } + } + textOnly := []string{ + "", "gpt-3.5-turbo", "gpt-4", "gpt-4-0613", "text-embedding-3-large", + "claude-2.1", "claude-instant-1.2", "llama3.1", "mistral-large", "some-random-model", + } + for _, m := range textOnly { + if ModelSupportsVision(m) { + t.Errorf("ModelSupportsVision(%q) = true, want false", m) + } + } +} + +func TestIsImageMIME(t *testing.T) { + for _, mt := range []string{"image/png", "image/jpeg", "image/jpg", "IMAGE/PNG", "image/gif", "image/webp"} { + if !IsImageMIME(mt) { + t.Errorf("IsImageMIME(%q) = false, want true", mt) + } + } + for _, mt := range []string{"", "application/pdf", "video/mp4", "text/plain", "image/tiff"} { + if IsImageMIME(mt) { + t.Errorf("IsImageMIME(%q) = true, want false", mt) + } + } +} + +func TestNormalizeImageMIME(t *testing.T) { + if got := NormalizeImageMIME("image/jpg"); got != "image/jpeg" { + t.Errorf("NormalizeImageMIME(image/jpg) = %q, want image/jpeg", got) + } + if got := NormalizeImageMIME("IMAGE/PNG "); got != "image/png" { + t.Errorf("NormalizeImageMIME normalized = %q, want image/png", got) + } +}