Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
2 changes: 1 addition & 1 deletion apps/ingest/src/ai_session.rs
Original file line number Diff line number Diff line change
Expand Up @@ -1707,7 +1707,7 @@ mod tests {
("gen_ai.operation.name", "chat"),
("gen_ai.usage.input_tokens", "2"),
("gen_ai.usage.cache_read.input_tokens", "114514"),
("gen_ai.usage.cache_creation.input_tokens", "3549"),
("gen_ai.usage.cache_write.input_tokens", "3549"),
("gen_ai.response.time_to_first_chunk", "0.934"),
] {
assert_eq!(attr_value(llm, key).as_deref(), Some(value), "{key}");
Expand Down
2 changes: 1 addition & 1 deletion apps/ingest/src/ai_session/claude_code.rs
Original file line number Diff line number Diff line change
Expand Up @@ -113,7 +113,7 @@ pub(super) fn normalize(span: &mut Span) {
text(attrs, "cache_read_tokens"),
);
add(
"gen_ai.usage.cache_creation.input_tokens",
"gen_ai.usage.cache_write.input_tokens",
text(attrs, "cache_creation_tokens"),
);
add(
Expand Down
30 changes: 30 additions & 0 deletions packages/agent-sessions/src/session-turns.test.ts
Original file line number Diff line number Diff line change
Expand Up @@ -255,6 +255,23 @@ describe("buildSessionTurns", () => {
])
})

it("does not open a turn at a root memory operation", () => {
Comment thread
maple-review-bot[bot] marked this conversation as resolved.
const turns = buildSessionTurns([
makeSpan({
spanId: "memory",
spanName: "search_memory chat-history",
startMs: 0,
durationMs: SECOND,
genAi: { operationName: "search_memory" },
}),
agentSpan({ spanId: "agent", startMs: 10 * SECOND, durationMs: 10 * SECOND }),
llmSpan({ spanId: "chat", parentSpanId: "agent", startMs: 11 * SECOND, durationMs: SECOND }),
])

expect(turns.map((turn) => turn.anchorKind)).toEqual(["agent-root"])
expect(turns[0]!.spans.map((span) => span.spanId)).toEqual(["memory", "agent", "chat"])
})

it("falls back to root agent invocations when no conversation id exists", () => {
const turns = buildSessionTurns([
agentSpan({ spanId: "agent-1", startMs: 0, durationMs: 10 * SECOND }),
Expand Down Expand Up @@ -608,6 +625,19 @@ describe("isLlmCall", () => {
expect(classifyAiSpan(embedding)).toBe("inference")
expect(isLlmCall(embedding)).toBe(false)
})

it("reads a memory operation as agent work, even when it names a model", () => {
const memory = makeSpan({
spanId: "a",
startMs: 0,
durationMs: 1,
spanName: "search_memory chat-history",
genAi: { operationName: "search_memory", requestModel: "text-embedding-3-small" },
})

expect(classifyAiSpan(memory)).toBe("agent")
expect(isLlmCall(memory)).toBe(false)
})
})

describe("the ingest gateway's verdicts", () => {
Expand Down
18 changes: 15 additions & 3 deletions packages/agent-sessions/src/session-turns.ts
Original file line number Diff line number Diff line change
Expand Up @@ -14,6 +14,7 @@
import {
AI_AGENT_OPERATIONS,
AI_INFERENCE_OPERATIONS,
AI_MEMORY_OPERATIONS,
AI_RETRIEVAL_OPERATIONS,
AI_TOOL_OPERATIONS,
} from "@maple/domain/gen-ai"
Expand All @@ -37,8 +38,11 @@ const INFERENCE_OPS: ReadonlySet<string> = new Set(AI_INFERENCE_OPERATIONS)
const RETRIEVAL_OPS: ReadonlySet<string> = new Set(AI_RETRIEVAL_OPERATIONS)
const TOOL_OPS: ReadonlySet<string> = new Set(AI_TOOL_OPERATIONS)
const AGENT_OPS: ReadonlySet<string> = new Set(AI_AGENT_OPERATIONS)
/** A memory-store operation is the agent's own bookkeeping, so it reads as agent
* work — not a model turn, not a tool call. */
const MEMORY_OPS: ReadonlySet<string> = new Set(AI_MEMORY_OPERATIONS)

/** Every operation name the four sets above recognise. A span name conventionally
/** Every operation name the five sets above recognise. A span name conventionally
* leads with one ("execute_tool read_file", "chat gpt-5"), so a view that wants
* to set the operation apart from its subject needs to know which words are
* operations — including for the reporters that skip `gen_ai.operation.name`. */
Expand All @@ -47,6 +51,7 @@ export const GEN_AI_OPERATIONS: ReadonlySet<string> = new Set([
...RETRIEVAL_OPS,
...TOOL_OPS,
...AGENT_OPS,
...MEMORY_OPS,
])

export function spanStartMs(span: AiSessionSpan): number {
Expand Down Expand Up @@ -77,7 +82,7 @@ export function classifyAiSpan(span: AiSessionSpan): AiSpanCategory {
if (operation !== undefined) {
if (INFERENCE_OPS.has(operation) || RETRIEVAL_OPS.has(operation)) return "inference"
if (TOOL_OPS.has(operation)) return "tool"
if (AGENT_OPS.has(operation)) return "agent"
if (AGENT_OPS.has(operation) || MEMORY_OPS.has(operation)) return "agent"

Copy link
Copy Markdown

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

🟡 Memory bookkeeping opens an extra turn

When a root memory span precedes an agent invocation, classifyAiSpan makes both eligible turn anchors. If they end over five seconds apart, findAnchors retains a phantom turn for bookkeeping.

Learn more

Agent-root anchors are AI spans classified as agent work with no AI ancestor. A root memory span now qualifies, although it does not start a user turn. The setup-merging rule only collapses workless anchors that end within five seconds of the next one buildSessionTurns. This makes memory operations produce extra turns when they are separate roots and more than five seconds apart.

Example: A search_memory span starts at second 0 and ends at second 1, followed by invoke_agent at second 10. Both are root AI spans. The session gets two turns instead of one.

Recommended fix: Keep memory operations classified as agent work for display, but exclude AI_MEMORY_OPERATIONS from the agent-root anchor selection in findAnchors. Add a test with a root memory span followed by an agent root beyond the setup window.

Devin Review


Was this helpful? React with 👍 or 👎 to provide feedback.

Copy link
Copy Markdown
Collaborator Author

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Valid, fixed in f3f338e. findAnchors now skips spans whose operation is in AI_MEMORY_OPERATIONS when it picks agent roots. They still classify as agent for display. Added test does not open a turn at a root memory operation: search_memory at 0-1s, then invoke_agent at 10s, gives one agent-root turn that holds all three spans.

Comment thread
maple-review-bot[bot] marked this conversation as resolved.
}
// Spans with no AI signal at all are the app's own HTTP/DB work, sharing the
// agent's traces. They are rendered, muted, and never colored as agent work.
Expand Down Expand Up @@ -436,8 +441,15 @@ function findAnchors(ordered: readonly AiSessionSpan[]): readonly TurnAnchor[] {
}
return false
}
// A memory operation is agent work, but bookkeeping around a turn rather than
// the start of one: as an anchor, a lookup ahead of the agent run would open a
// turn of its own.
const agentRoots = ordered.filter(
(span) => span.isAiSpan && classifyAiSpan(span) === "agent" && !underAiSpan(span),
(span) =>
span.isAiSpan &&
classifyAiSpan(span) === "agent" &&
!MEMORY_OPS.has(span.genAi.operationName ?? "") &&

Copy link
Copy Markdown

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Memory span with no gen_ai.operation.name still opens a turn

F1 · Warning · correctness

The anchor filter reads only the reported attribute, but classifyAiSpan falls back to the span name — and this file exists to serve the reporters that skip gen_ai.operation.name (GEN_AI_OPERATIONS, session-waterfall.tsx:758). A root span named create_memory with no model and no reported operation lands on session-turns.ts:91 as "agent", so MEMORY_OPS.has("") is false and it anchors a turn of its own: the session shows an extra turn whose only content is bookkeeping — exactly what f3f338e set out to prevent.

Derive the operation the same way `classifyAiSpan` does when the attribute is absent — the leading word of the span name when `GEN_AI_OPERATIONS` contains it — e.g. a small `spanOperation(span)` helper called from both places, and filter on its result here.
Prompt for an AI agent
In `packages/agent-sessions/src/session-turns.ts:425-426`: Memory span with no `gen_ai.operation.name` still opens a turn.

The anchor filter reads only the reported attribute, but `classifyAiSpan` falls back to the span name — and this file exists to serve the reporters that skip `gen_ai.operation.name` (`GEN_AI_OPERATIONS`, `session-waterfall.tsx:758`). A root span named `create_memory` with no model and no reported operation lands on `session-turns.ts:91` as `"agent"`, so `MEMORY_OPS.has("")` is false and it anchors a turn of its own: the session shows an extra turn whose only content is bookkeeping — exactly what f3f338e set out to prevent.

Suggested fix: Derive the operation the same way `classifyAiSpan` does when the attribute is absent — the leading word of the span name when `GEN_AI_OPERATIONS` contains it — e.g. a small `spanOperation(span)` helper called from both places, and filter on its result here.

Verify the problem exists at that location before changing it, and keep the fix to those lines.

@JeremyFunk JeremyFunk Sep 29, 2026 •

Copy link
Copy Markdown
Collaborator Author

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Leaving this as is. A memory operation is recognised by its operation everywhere: the ingest gateway (KNOWN_OPS against the reported operation, #1143) and classifyAiSpan. Reading it off a span name here only would make turn anchoring disagree with how the same span is classified.

!underAiSpan(span),
)
if (agentRoots.length > 0) {
return agentRoots.map((span) => ({
Expand Down
6 changes: 3 additions & 3 deletions packages/backend/src/services/warehouse/warehouse-catalog.ts
Original file line number Diff line number Diff line change
Expand Up @@ -32,9 +32,9 @@ const TABLE_NOTES: Record<string, ReadonlyArray<string>> = {
ai_trace_index: [
"GenAI agent spans ONLY (every row carries a non-empty `VendorId`), with the `maple_ai.*` identity pre-extracted to plain columns. ALWAYS prefer this over `traces` + `mapContains(SpanAttributes, 'maple_ai.…')` for finding agent traces/sessions — the raw-traces scan reads the full attribute Map per span and times out on day-plus windows.",
"`SessionId` is '' on most rows: vendors stamp the session key only on turn-owning spans. Resolve a trace's session as `max(SessionId) GROUP BY TraceId`, and treat a trace whose max is '' as a sessionless single-trace session.",
"`DeploymentEnv`, `Model`, `AgentName` and `ToolName` are the span's environment and GenAI identity, coalesced across dialects at insert (`gen_ai.*`, Vercel AI SDK `ai.*`, OpenInference `llm.*`/`tool.*`). '' where the span carries no such fact — a chat span has no tool — and on rows materialized before migration 0026. Filter and facet on these here rather than on `trace_detail_spans` attributes.",
"`IsLlmCall`, `IsToolCall`, `IsError` (UInt8 flags), `Tokens`, `Cost` (Float64) are the span's kind, failure and reported usage; a tool call paused for a human's approval is no call until it runs, so its paused copy has `IsToolCall` 0; `SpanId`/`ParentSpanId`/`Duration` are its own. `Tokens` is the span's billed total under the reporter's own convention — a prompt figure that already contains its cached tokens (OpenAI, OpenRouter, Gemini) or a completion figure that contains its reasoning is NOT double counted, so it can read below `input + cache_read + output + reasoning` summed off the raw attributes. Sum per session here for calls, failures, tokens and cost — but a wrapper span often repeats its children's usage, so subtract a child reporter's tokens from its parent (`ParentSpanId = SpanId`) before summing, or the total doubles.",
"`VendorVersion` (LowCardinality String) is the framework's version beside `VendorId`, and `InputTokens`, `CacheReadTokens`, `CacheWriteTokens`, `OutputTokens`, `ReasoningTokens` (Float64) are the disjoint split of `Tokens` — '' / 0 on rows materialized before migration 0031.",
"`DeploymentEnv`, `Model`, `AgentName` and `ToolName` are the span's environment and GenAI identity; `Model`, `AgentName` and `ToolName` are stamped by the ingest gateway (`maple_ai.model`, `maple_ai.agent.name`, `maple_ai.tool.name`), which reads every dialect (`gen_ai.*`, Vercel AI SDK `ai.*`, OpenInference `llm.*`/`tool.*`). '' where the span carries no such fact — a chat span has no tool — and on rows materialized before migration 0026. Filter and facet on these here rather than on `trace_detail_spans` attributes.",
"`IsLlmCall`, `IsToolCall`, `IsError` (UInt8 flags), `Tokens`, `Cost` (Float64) are the span's kind, failure and usage as the ingest gateway stamped them (`maple_ai.llm_call`, `maple_ai.tool_call`, `maple_ai.error`, `maple_ai.usage.*`); a tool call paused for a human's approval is no call until it runs, so its paused copy has `IsToolCall` 0; `SpanId`/`ParentSpanId`/`Duration` are its own. `Tokens` is the sum of the five disjoint buckets the gateway normalised into `maple_ai.usage.*` — a prompt figure that already contains its cached tokens (OpenAI, OpenRouter, Gemini) or a completion figure that contains its reasoning is NOT double counted, so it can read below `input + cache_read + output + reasoning` summed off the raw `gen_ai.usage.*` attributes. Only the model-call span carries usage, so sum per session here for calls, failures, tokens and cost. Rows materialized before migration 0035 keep the view's older figures, where a wrapper span could repeat its children's usage: net a child reporter's tokens off its parent (`ParentSpanId = SpanId`) on those rows only.",
"`VendorVersion` (LowCardinality String) is the framework's version beside `VendorId`, and `InputTokens`, `CacheReadTokens`, `CacheWriteTokens`, `OutputTokens`, `ReasoningTokens` (Float64) are the disjoint split of `Tokens`, read from the gateway's normalised `maple_ai.usage.*` — '' / 0 on rows materialized before migration 0031.",
"`ErrorType` (LowCardinality String, `error.type`), `StatusMessage` and `ToolDescription` (`gen_ai.tool.description`, tool spans only) are the span's own failure reason and tool documentation, truncated at insert — '' on rows materialized before migration 0032. Group failures by `ErrorType` here rather than reading `trace_detail_spans` for it. `FailedToolCallResult` (String) is a failed tool call's result, where many frameworks put the error — '' on every other row — and `ErrorFingerprint` (UInt64, select it as `toString(ErrorFingerprint)`) groups failures by that result, else the status message, with volatile values redacted; 0 on rows that did not fail.",
"Holds only the agent spans, and only their identity — for every span of a detected trace, or for anything it does not carry (`StatusCode`, the tool call's arguments and result, `gen_ai.usage.*` per key), collect `TraceId`s here first, then read `trace_detail_spans` with `TraceId IN (…)` AND a `Timestamp` window.",
"Sorting key: `(OrgId, Timestamp, TraceId)`; filled forward by its MV, so windows predating the cluster's schema apply under-report.",
Expand Down
15 changes: 14 additions & 1 deletion packages/domain/src/gen-ai.ts
Original file line number Diff line number Diff line change
Expand Up @@ -84,7 +84,7 @@ export const MAPLE_NATIVE_TURN_ID_ATTR = "maple_ai.turn.id"

// `gen_ai.operation.name` is an open set. These group the semantic convention's
// operation names — plus `agent_step`, which the Vercel AI SDK emits and
// production data carries — into the four readings the product distinguishes.
// production data carries — into the five readings the product distinguishes.
// Shared between the session summary query and the web's span classifier so an
// "llm call" is the same span on the server and on the page.
export const AI_INFERENCE_OPERATIONS = [
Expand All @@ -103,6 +103,19 @@ export const AI_AGENT_OPERATIONS = [
"plan",
"agent_step",
] as const
/** The convention's memory-store operations: the agent's own bookkeeping, never
* a model turn or a tool call. The ingest gateway decides this for the spans it
* stamps (`KNOWN_OPS` in `apps/ingest/src/ai_session/usage.rs`); the span
* classifier reads it for the spans ingested before. */
export const AI_MEMORY_OPERATIONS = [
"search_memory",
"create_memory",
"update_memory",
"upsert_memory",
"delete_memory",
"create_memory_store",
"delete_memory_store",
] as const
/**
* Count of whole oldest messages dropped from `gen_ai.input.messages` to fit
* the emitter's attribute budget. Write-only diagnostics: nothing decodes it,
Expand Down
Loading
Loading