diff --git a/packages/agent-sessions/src/session-summary.test.ts b/packages/agent-sessions/src/session-summary.test.ts index 5c0929a69..7a3381564 100644 --- a/packages/agent-sessions/src/session-summary.test.ts +++ b/packages/agent-sessions/src/session-summary.test.ts @@ -1566,6 +1566,40 @@ describe("per-model cost, tools and failure groups", () => { ]) }) + // The tests above read spans ingested before the gateway stamped them. On a + // stamped span its `maple_ai.tool_call` verdict decides, as on the list. + it("counts a stamped tool span by the gateway's verdict alone", () => { + const stamped = (spanId: string, startMs: number, mapleToolCall: number, toolCallResult?: string) => + toolSpan({ + spanId, + traceId: `trace-${spanId}`, + toolName: "delete_file", + startMs, + durationMs: 1, + genAi: { mapleLlmCall: 0, mapleToolCall, toolCallId: "call_a", toolCallResult }, + }) + const summary = summarize([ + // Google ADK's confirmation request: no call. + stamped( + "paused", + 0, + 0, + '{"error": "This tool call requires confirmation, please approve or reject."}', + ), + // LlamaIndex ends a step waiting for a human in error, which the + // gateway stamped neither a call nor a failure. + { ...stamped("waiting", 500, 0), statusCode: "Error" }, + // No result and no framework mark: a call, not merged into the one + // that follows under the same id. + stamped("no-result", 1_000, 1), + stamped("approved", 2_000, 1, "deleted /tmp/scratch-notes.txt"), + ]) + + expect(summary.work.toolCalls).toBe(2) + expect(summary.tools[0]!.events.map((event) => event.spanId)).toEqual(["no-result", "approved"]) + expect(summary.failures.errors).toBe(0) + }) + it("keeps two calls that share an id when both returned, and every call captured without payloads", () => { const lane = (spanId: string, traceId: string, startMs: number, result?: string) => toolSpan({ diff --git a/packages/agent-sessions/src/session-summary.ts b/packages/agent-sessions/src/session-summary.ts index 8dd0fea3a..a46150bb7 100644 --- a/packages/agent-sessions/src/session-summary.ts +++ b/packages/agent-sessions/src/session-summary.ts @@ -875,18 +875,19 @@ function modelUsage( } /** - * The tool calls the session made, in start order, a call paused for a human - * counted once. The interrupted call leaves a tool span that recorded no - * result and did not fail, and the resumed turn opens another under the same - * `gen_ai.tool.call.id` that carries the result on its attributes (Strands) - * — so the paused copy is dropped when a later copy with a result exists. - * Not covered, and still counted twice: Google ADK (the outcome is only in - * `gcp.vertex.agent.tool_response`, and the paused copy's confirmation - * request reads as a result), Strands versions that record results in span - * events, and OpenAI Agents (no call id). Nothing else is merged: two calls - * that merely share an id (parallel lanes, a provider numbering its calls per - * turn) both carry results, and a session captured without payloads keeps - * every span. + * The tool calls the session made, in start order. On a span the ingest + * gateway stamped, its verdict alone (`maple_ai.tool_call`, what the list + * sums): the copy a call paused for a human's approval leaves is stamped no + * call by its framework's explicit mark, and nothing is merged. + * + * The spans ingested before the gateway stamped them keep the old merge until + * they age out of the 30-day TTL: a call paused for a human leaves a tool span + * that recorded no result and did not fail, and the resumed turn opens + * another under the same `gen_ai.tool.call.id` that carries the result on its + * attributes (Strands), so the paused copy is dropped when a later copy with a + * result exists. Two calls that merely share an id (parallel lanes, a provider + * numbering its calls per turn) both carry results, and a session captured + * without payloads keeps every span. */ function countedToolCalls(ordered: readonly AiSessionSpan[]): readonly AiSessionSpan[] { const tools = ordered.filter((span) => classifyAiSpan(span) === "tool") @@ -897,6 +898,7 @@ function countedToolCalls(ordered: readonly AiSessionSpan[]): readonly AiSession if (callId !== undefined && callId !== "" && recorded(span)) resumedAt.set(callId, spanStartMs(span)) } return tools.filter((span) => { + if (span.genAi.mapleLlmCall !== undefined) return true const resumed = resumedAt.get(span.genAi.toolCallId ?? "") return resumed === undefined || resumed <= spanStartMs(span) || recorded(span) || spanFailed(span) }) diff --git a/packages/agent-sessions/src/session-turns.ts b/packages/agent-sessions/src/session-turns.ts index 28c9dd8e1..d166e52f4 100644 --- a/packages/agent-sessions/src/session-turns.ts +++ b/packages/agent-sessions/src/session-turns.ts @@ -127,11 +127,13 @@ const FAILED_RESPONSE_STATUSES = new Set(["failed", "error"]) * `gen_ai.response.status` counts too. Scoped to AI spans because HTTP * instrumentation legitimately stamps `error.type` on expected 4xx requests * whose span status is deliberately not `Error`. On a span the ingest gateway - * stamped, its verdict (`MAPLE_AI_STAMP_ATTRS.error`), which is this rule. + * stamped, its verdict (`MAPLE_AI_STAMP_ATTRS.error`), which is this rule + * less the copy a call paused for a human's approval leaves, which some + * frameworks end in error. */ export function spanFailed(span: AiSessionSpan): boolean { - if (span.statusCode === "Error") return true if (span.genAi.mapleLlmCall !== undefined) return span.genAi.mapleError === 1 + if (span.statusCode === "Error") return true if (!span.isAiSpan) return false const errorType = span.genAi.errorType if (errorType !== undefined && errorType !== "") return true diff --git a/packages/query-engine-integrations/src/__sql_baseline__/integrations.sql b/packages/query-engine-integrations/src/__sql_baseline__/integrations.sql index 0350d45fb..cd626fa77 100644 --- a/packages/query-engine-integrations/src/__sql_baseline__/integrations.sql +++ b/packages/query-engine-integrations/src/__sql_baseline__/integrations.sql @@ -819,7 +819,7 @@ SELECT countIf(SpanAttributes['maple_ai.vendor.id'] != '') AS aiSpanCount, countIf((SpanAttributes['maple_ai.llm_call'] = '1' OR (NOT (SpanAttributes['maple_ai.llm_call'] != '') AND (coalesce(nullIf(SpanAttributes['gen_ai.operation.name'], ''), '') IN ('chat', 'generate_content', 'text_completion', 'fetch_response') OR ((coalesce(nullIf(SpanAttributes['gen_ai.operation.name'], ''), '') NOT IN ('embeddings', 'retrieval', 'execute_tool', 'invoke_agent', 'create_agent', 'invoke_workflow', 'plan', 'agent_step') AND coalesce(nullIf(SpanAttributes['gen_ai.response.model'], ''), nullIf(SpanAttributes['ai.response.model'], ''), nullIf(SpanAttributes['gen_ai.request.model'], ''), nullIf(SpanAttributes['ai.model.id'], ''), nullIf(SpanAttributes['llm.model_name'], ''), '') != '') AND coalesce(nullIf(SpanAttributes['gen_ai.tool.name'], ''), nullIf(SpanAttributes['ai.toolCall.name'], ''), nullIf(SpanAttributes['tool.name'], ''), '') = ''))))) AS llmCalls, countIf((SpanAttributes['maple_ai.tool_call'] = '1' OR (NOT (SpanAttributes['maple_ai.llm_call'] != '') AND (coalesce(nullIf(SpanAttributes['gen_ai.operation.name'], ''), '') IN ('execute_tool') OR ((coalesce(nullIf(SpanAttributes['gen_ai.operation.name'], ''), '') = '' AND SpanAttributes['maple_ai.vendor.id'] != '') AND coalesce(nullIf(SpanAttributes['gen_ai.tool.name'], ''), nullIf(SpanAttributes['ai.toolCall.name'], ''), nullIf(SpanAttributes['tool.name'], ''), '') != ''))))) AS toolCalls, - countIf(((StatusCode = 'Error' OR SpanAttributes['maple_ai.error'] = '1') OR ((NOT (SpanAttributes['maple_ai.llm_call'] != '') AND SpanAttributes['maple_ai.vendor.id'] != '') AND (coalesce(nullIf(SpanAttributes['error.type'], ''), '') != '' OR SpanAttributes['gen_ai.response.status'] IN ('failed', 'error'))))) AS errorSpanCount, + countIf((SpanAttributes['maple_ai.error'] = '1' OR (NOT (SpanAttributes['maple_ai.llm_call'] != '') AND (StatusCode = 'Error' OR (SpanAttributes['maple_ai.vendor.id'] != '' AND (coalesce(nullIf(SpanAttributes['error.type'], ''), '') != '' OR SpanAttributes['gen_ai.response.status'] IN ('failed', 'error'))))))) AS errorSpanCount, ifNotFinite(sum(if(SpanAttributes['maple_ai.llm_call'] != '', toFloat64OrZero(coalesce(nullIf(SpanAttributes['maple_ai.usage.input_tokens'], ''), '')) + toFloat64OrZero(coalesce(nullIf(SpanAttributes['maple_ai.usage.cache_write_tokens'], ''), '')), toFloat64OrZero(coalesce(nullIf(SpanAttributes['gen_ai.usage.input_tokens'], ''), nullIf(SpanAttributes['gen_ai.usage.prompt_tokens'], ''), nullIf(SpanAttributes['ai.usage.inputTokens'], ''), nullIf(SpanAttributes['ai.usage.promptTokens'], ''), nullIf(SpanAttributes['llm.token_count.prompt'], ''), '')))), 0) AS inputTokens, ifNotFinite(sum(if(SpanAttributes['maple_ai.llm_call'] != '', toFloat64OrZero(coalesce(nullIf(SpanAttributes['maple_ai.usage.output_tokens'], ''), '')) + toFloat64OrZero(coalesce(nullIf(SpanAttributes['maple_ai.usage.reasoning_tokens'], ''), '')), toFloat64OrZero(coalesce(nullIf(SpanAttributes['gen_ai.usage.output_tokens'], ''), nullIf(SpanAttributes['gen_ai.usage.completion_tokens'], ''), nullIf(SpanAttributes['ai.usage.outputTokens'], ''), nullIf(SpanAttributes['ai.usage.completionTokens'], ''), nullIf(SpanAttributes['llm.token_count.completion'], ''), '')))), 0) AS outputTokens, ifNotFinite(sum(if(SpanAttributes['maple_ai.llm_call'] != '', toFloat64OrZero(coalesce(nullIf(SpanAttributes['maple_ai.usage.cache_read_tokens'], ''), '')), toFloat64OrZero(coalesce(nullIf(SpanAttributes['gen_ai.usage.cache_read.input_tokens'], ''), nullIf(SpanAttributes['gen_ai.usage.input_tokens.cached'], ''), nullIf(SpanAttributes['ai.usage.cachedInputTokens'], ''), nullIf(SpanAttributes['ai.usage.inputTokenDetails.cacheReadTokens'], ''), nullIf(SpanAttributes['llm.token_count.prompt_details.cache_read'], ''), '')))), 0) AS cacheReadTokens, @@ -858,7 +858,7 @@ SELECT countIf(SpanAttributes['maple_ai.vendor.id'] != '') AS aiSpanCount, countIf((SpanAttributes['maple_ai.llm_call'] = '1' OR (NOT (SpanAttributes['maple_ai.llm_call'] != '') AND (coalesce(nullIf(SpanAttributes['gen_ai.operation.name'], ''), '') IN ('chat', 'generate_content', 'text_completion', 'fetch_response') OR ((coalesce(nullIf(SpanAttributes['gen_ai.operation.name'], ''), '') NOT IN ('embeddings', 'retrieval', 'execute_tool', 'invoke_agent', 'create_agent', 'invoke_workflow', 'plan', 'agent_step') AND coalesce(nullIf(SpanAttributes['gen_ai.response.model'], ''), nullIf(SpanAttributes['ai.response.model'], ''), nullIf(SpanAttributes['gen_ai.request.model'], ''), nullIf(SpanAttributes['ai.model.id'], ''), nullIf(SpanAttributes['llm.model_name'], ''), '') != '') AND coalesce(nullIf(SpanAttributes['gen_ai.tool.name'], ''), nullIf(SpanAttributes['ai.toolCall.name'], ''), nullIf(SpanAttributes['tool.name'], ''), '') = ''))))) AS llmCalls, countIf((SpanAttributes['maple_ai.tool_call'] = '1' OR (NOT (SpanAttributes['maple_ai.llm_call'] != '') AND (coalesce(nullIf(SpanAttributes['gen_ai.operation.name'], ''), '') IN ('execute_tool') OR ((coalesce(nullIf(SpanAttributes['gen_ai.operation.name'], ''), '') = '' AND SpanAttributes['maple_ai.vendor.id'] != '') AND coalesce(nullIf(SpanAttributes['gen_ai.tool.name'], ''), nullIf(SpanAttributes['ai.toolCall.name'], ''), nullIf(SpanAttributes['tool.name'], ''), '') != ''))))) AS toolCalls, - countIf(((StatusCode = 'Error' OR SpanAttributes['maple_ai.error'] = '1') OR ((NOT (SpanAttributes['maple_ai.llm_call'] != '') AND SpanAttributes['maple_ai.vendor.id'] != '') AND (coalesce(nullIf(SpanAttributes['error.type'], ''), '') != '' OR SpanAttributes['gen_ai.response.status'] IN ('failed', 'error'))))) AS errorSpanCount, + countIf((SpanAttributes['maple_ai.error'] = '1' OR (NOT (SpanAttributes['maple_ai.llm_call'] != '') AND (StatusCode = 'Error' OR (SpanAttributes['maple_ai.vendor.id'] != '' AND (coalesce(nullIf(SpanAttributes['error.type'], ''), '') != '' OR SpanAttributes['gen_ai.response.status'] IN ('failed', 'error'))))))) AS errorSpanCount, ifNotFinite(sum(if(SpanAttributes['maple_ai.llm_call'] != '', toFloat64OrZero(coalesce(nullIf(SpanAttributes['maple_ai.usage.input_tokens'], ''), '')) + toFloat64OrZero(coalesce(nullIf(SpanAttributes['maple_ai.usage.cache_write_tokens'], ''), '')), toFloat64OrZero(coalesce(nullIf(SpanAttributes['gen_ai.usage.input_tokens'], ''), nullIf(SpanAttributes['gen_ai.usage.prompt_tokens'], ''), nullIf(SpanAttributes['ai.usage.inputTokens'], ''), nullIf(SpanAttributes['ai.usage.promptTokens'], ''), nullIf(SpanAttributes['llm.token_count.prompt'], ''), '')))), 0) AS inputTokens, ifNotFinite(sum(if(SpanAttributes['maple_ai.llm_call'] != '', toFloat64OrZero(coalesce(nullIf(SpanAttributes['maple_ai.usage.output_tokens'], ''), '')) + toFloat64OrZero(coalesce(nullIf(SpanAttributes['maple_ai.usage.reasoning_tokens'], ''), '')), toFloat64OrZero(coalesce(nullIf(SpanAttributes['gen_ai.usage.output_tokens'], ''), nullIf(SpanAttributes['gen_ai.usage.completion_tokens'], ''), nullIf(SpanAttributes['ai.usage.outputTokens'], ''), nullIf(SpanAttributes['ai.usage.completionTokens'], ''), nullIf(SpanAttributes['llm.token_count.completion'], ''), '')))), 0) AS outputTokens, ifNotFinite(sum(if(SpanAttributes['maple_ai.llm_call'] != '', toFloat64OrZero(coalesce(nullIf(SpanAttributes['maple_ai.usage.cache_read_tokens'], ''), '')), toFloat64OrZero(coalesce(nullIf(SpanAttributes['gen_ai.usage.cache_read.input_tokens'], ''), nullIf(SpanAttributes['gen_ai.usage.input_tokens.cached'], ''), nullIf(SpanAttributes['ai.usage.cachedInputTokens'], ''), nullIf(SpanAttributes['ai.usage.inputTokenDetails.cacheReadTokens'], ''), nullIf(SpanAttributes['llm.token_count.prompt_details.cache_read'], ''), '')))), 0) AS cacheReadTokens, @@ -952,7 +952,7 @@ SELECT countIf(SpanAttributes['maple_ai.vendor.id'] != '') AS aiSpanCount, countIf((SpanAttributes['maple_ai.llm_call'] = '1' OR (NOT (SpanAttributes['maple_ai.llm_call'] != '') AND (coalesce(nullIf(SpanAttributes['gen_ai.operation.name'], ''), '') IN ('chat', 'generate_content', 'text_completion', 'fetch_response') OR ((coalesce(nullIf(SpanAttributes['gen_ai.operation.name'], ''), '') NOT IN ('embeddings', 'retrieval', 'execute_tool', 'invoke_agent', 'create_agent', 'invoke_workflow', 'plan', 'agent_step') AND coalesce(nullIf(SpanAttributes['gen_ai.response.model'], ''), nullIf(SpanAttributes['ai.response.model'], ''), nullIf(SpanAttributes['gen_ai.request.model'], ''), nullIf(SpanAttributes['ai.model.id'], ''), nullIf(SpanAttributes['llm.model_name'], ''), '') != '') AND coalesce(nullIf(SpanAttributes['gen_ai.tool.name'], ''), nullIf(SpanAttributes['ai.toolCall.name'], ''), nullIf(SpanAttributes['tool.name'], ''), '') = ''))))) AS llmCalls, countIf((SpanAttributes['maple_ai.tool_call'] = '1' OR (NOT (SpanAttributes['maple_ai.llm_call'] != '') AND (coalesce(nullIf(SpanAttributes['gen_ai.operation.name'], ''), '') IN ('execute_tool') OR ((coalesce(nullIf(SpanAttributes['gen_ai.operation.name'], ''), '') = '' AND SpanAttributes['maple_ai.vendor.id'] != '') AND coalesce(nullIf(SpanAttributes['gen_ai.tool.name'], ''), nullIf(SpanAttributes['ai.toolCall.name'], ''), nullIf(SpanAttributes['tool.name'], ''), '') != ''))))) AS toolCalls, - countIf(((StatusCode = 'Error' OR SpanAttributes['maple_ai.error'] = '1') OR ((NOT (SpanAttributes['maple_ai.llm_call'] != '') AND SpanAttributes['maple_ai.vendor.id'] != '') AND (coalesce(nullIf(SpanAttributes['error.type'], ''), '') != '' OR SpanAttributes['gen_ai.response.status'] IN ('failed', 'error'))))) AS errorSpanCount, + countIf((SpanAttributes['maple_ai.error'] = '1' OR (NOT (SpanAttributes['maple_ai.llm_call'] != '') AND (StatusCode = 'Error' OR (SpanAttributes['maple_ai.vendor.id'] != '' AND (coalesce(nullIf(SpanAttributes['error.type'], ''), '') != '' OR SpanAttributes['gen_ai.response.status'] IN ('failed', 'error'))))))) AS errorSpanCount, ifNotFinite(sum(if(SpanAttributes['maple_ai.llm_call'] != '', toFloat64OrZero(coalesce(nullIf(SpanAttributes['maple_ai.usage.input_tokens'], ''), '')) + toFloat64OrZero(coalesce(nullIf(SpanAttributes['maple_ai.usage.cache_write_tokens'], ''), '')), toFloat64OrZero(coalesce(nullIf(SpanAttributes['gen_ai.usage.input_tokens'], ''), nullIf(SpanAttributes['gen_ai.usage.prompt_tokens'], ''), nullIf(SpanAttributes['ai.usage.inputTokens'], ''), nullIf(SpanAttributes['ai.usage.promptTokens'], ''), nullIf(SpanAttributes['llm.token_count.prompt'], ''), '')))), 0) AS inputTokens, ifNotFinite(sum(if(SpanAttributes['maple_ai.llm_call'] != '', toFloat64OrZero(coalesce(nullIf(SpanAttributes['maple_ai.usage.output_tokens'], ''), '')) + toFloat64OrZero(coalesce(nullIf(SpanAttributes['maple_ai.usage.reasoning_tokens'], ''), '')), toFloat64OrZero(coalesce(nullIf(SpanAttributes['gen_ai.usage.output_tokens'], ''), nullIf(SpanAttributes['gen_ai.usage.completion_tokens'], ''), nullIf(SpanAttributes['ai.usage.outputTokens'], ''), nullIf(SpanAttributes['ai.usage.completionTokens'], ''), nullIf(SpanAttributes['llm.token_count.completion'], ''), '')))), 0) AS outputTokens, ifNotFinite(sum(if(SpanAttributes['maple_ai.llm_call'] != '', toFloat64OrZero(coalesce(nullIf(SpanAttributes['maple_ai.usage.cache_read_tokens'], ''), '')), toFloat64OrZero(coalesce(nullIf(SpanAttributes['gen_ai.usage.cache_read.input_tokens'], ''), nullIf(SpanAttributes['gen_ai.usage.input_tokens.cached'], ''), nullIf(SpanAttributes['ai.usage.cachedInputTokens'], ''), nullIf(SpanAttributes['ai.usage.inputTokenDetails.cacheReadTokens'], ''), nullIf(SpanAttributes['llm.token_count.prompt_details.cache_read'], ''), '')))), 0) AS cacheReadTokens, @@ -984,7 +984,7 @@ SELECT countIf(SpanAttributes['maple_ai.vendor.id'] != '') AS aiSpanCount, countIf((SpanAttributes['maple_ai.llm_call'] = '1' OR (NOT (SpanAttributes['maple_ai.llm_call'] != '') AND (coalesce(nullIf(SpanAttributes['gen_ai.operation.name'], ''), '') IN ('chat', 'generate_content', 'text_completion', 'fetch_response') OR ((coalesce(nullIf(SpanAttributes['gen_ai.operation.name'], ''), '') NOT IN ('embeddings', 'retrieval', 'execute_tool', 'invoke_agent', 'create_agent', 'invoke_workflow', 'plan', 'agent_step') AND coalesce(nullIf(SpanAttributes['gen_ai.response.model'], ''), nullIf(SpanAttributes['ai.response.model'], ''), nullIf(SpanAttributes['gen_ai.request.model'], ''), nullIf(SpanAttributes['ai.model.id'], ''), nullIf(SpanAttributes['llm.model_name'], ''), '') != '') AND coalesce(nullIf(SpanAttributes['gen_ai.tool.name'], ''), nullIf(SpanAttributes['ai.toolCall.name'], ''), nullIf(SpanAttributes['tool.name'], ''), '') = ''))))) AS llmCalls, countIf((SpanAttributes['maple_ai.tool_call'] = '1' OR (NOT (SpanAttributes['maple_ai.llm_call'] != '') AND (coalesce(nullIf(SpanAttributes['gen_ai.operation.name'], ''), '') IN ('execute_tool') OR ((coalesce(nullIf(SpanAttributes['gen_ai.operation.name'], ''), '') = '' AND SpanAttributes['maple_ai.vendor.id'] != '') AND coalesce(nullIf(SpanAttributes['gen_ai.tool.name'], ''), nullIf(SpanAttributes['ai.toolCall.name'], ''), nullIf(SpanAttributes['tool.name'], ''), '') != ''))))) AS toolCalls, - countIf(((StatusCode = 'Error' OR SpanAttributes['maple_ai.error'] = '1') OR ((NOT (SpanAttributes['maple_ai.llm_call'] != '') AND SpanAttributes['maple_ai.vendor.id'] != '') AND (coalesce(nullIf(SpanAttributes['error.type'], ''), '') != '' OR SpanAttributes['gen_ai.response.status'] IN ('failed', 'error'))))) AS errorSpanCount, + countIf((SpanAttributes['maple_ai.error'] = '1' OR (NOT (SpanAttributes['maple_ai.llm_call'] != '') AND (StatusCode = 'Error' OR (SpanAttributes['maple_ai.vendor.id'] != '' AND (coalesce(nullIf(SpanAttributes['error.type'], ''), '') != '' OR SpanAttributes['gen_ai.response.status'] IN ('failed', 'error'))))))) AS errorSpanCount, ifNotFinite(sum(if(SpanAttributes['maple_ai.llm_call'] != '', toFloat64OrZero(coalesce(nullIf(SpanAttributes['maple_ai.usage.input_tokens'], ''), '')) + toFloat64OrZero(coalesce(nullIf(SpanAttributes['maple_ai.usage.cache_write_tokens'], ''), '')), toFloat64OrZero(coalesce(nullIf(SpanAttributes['gen_ai.usage.input_tokens'], ''), nullIf(SpanAttributes['gen_ai.usage.prompt_tokens'], ''), nullIf(SpanAttributes['ai.usage.inputTokens'], ''), nullIf(SpanAttributes['ai.usage.promptTokens'], ''), nullIf(SpanAttributes['llm.token_count.prompt'], ''), '')))), 0) AS inputTokens, ifNotFinite(sum(if(SpanAttributes['maple_ai.llm_call'] != '', toFloat64OrZero(coalesce(nullIf(SpanAttributes['maple_ai.usage.output_tokens'], ''), '')) + toFloat64OrZero(coalesce(nullIf(SpanAttributes['maple_ai.usage.reasoning_tokens'], ''), '')), toFloat64OrZero(coalesce(nullIf(SpanAttributes['gen_ai.usage.output_tokens'], ''), nullIf(SpanAttributes['gen_ai.usage.completion_tokens'], ''), nullIf(SpanAttributes['ai.usage.outputTokens'], ''), nullIf(SpanAttributes['ai.usage.completionTokens'], ''), nullIf(SpanAttributes['llm.token_count.completion'], ''), '')))), 0) AS outputTokens, ifNotFinite(sum(if(SpanAttributes['maple_ai.llm_call'] != '', toFloat64OrZero(coalesce(nullIf(SpanAttributes['maple_ai.usage.cache_read_tokens'], ''), '')), toFloat64OrZero(coalesce(nullIf(SpanAttributes['gen_ai.usage.cache_read.input_tokens'], ''), nullIf(SpanAttributes['gen_ai.usage.input_tokens.cached'], ''), nullIf(SpanAttributes['ai.usage.cachedInputTokens'], ''), nullIf(SpanAttributes['ai.usage.inputTokenDetails.cacheReadTokens'], ''), nullIf(SpanAttributes['llm.token_count.prompt_details.cache_read'], ''), '')))), 0) AS cacheReadTokens, diff --git a/packages/query-engine-integrations/src/ai/ai-sessions.test.ts b/packages/query-engine-integrations/src/ai/ai-sessions.test.ts index 88e26ed71..621a8fc22 100644 --- a/packages/query-engine-integrations/src/ai/ai-sessions.test.ts +++ b/packages/query-engine-integrations/src/ai/ai-sessions.test.ts @@ -1437,11 +1437,11 @@ describe("aiSessionSummaryQuery", () => { `if(${stamped}, toFloat64OrZero(coalesce(nullIf(SpanAttributes['maple_ai.usage.input_tokens'], ''), '')) + toFloat64OrZero(coalesce(nullIf(SpanAttributes['maple_ai.usage.cache_write_tokens'], ''), '')), toFloat64OrZero(coalesce(nullIf(SpanAttributes['gen_ai.usage.input_tokens'], '')`, ) // A model call and a tool call by the gateway's verdict, else by the - // op/model rules; a failure by its verdict or the span's status. + // op/model rules; a failure by its verdict, else by the span's status. expect(sql).toContain(`countIf((SpanAttributes['maple_ai.llm_call'] = '1' OR (NOT (${stamped}) AND `) expect(sql).toContain(`countIf((SpanAttributes['maple_ai.tool_call'] = '1' OR (NOT (${stamped}) AND `) expect(sql).toContain( - `countIf(((StatusCode = 'Error' OR SpanAttributes['maple_ai.error'] = '1') OR ((NOT (${stamped}) AND `, + `countIf((SpanAttributes['maple_ai.error'] = '1' OR (NOT (${stamped}) AND (StatusCode = 'Error' OR `, ) expect(sql).toContain(`if(${stamped}, SpanAttributes['maple_ai.model'], `) expect(sql).toContain(`if(${stamped}, SpanAttributes['maple_ai.agent.name'], `) diff --git a/packages/query-engine-integrations/src/ai/ai-sessions.ts b/packages/query-engine-integrations/src/ai/ai-sessions.ts index d28edb77c..eb689529e 100644 --- a/packages/query-engine-integrations/src/ai/ai-sessions.ts +++ b/packages/query-engine-integrations/src/ai/ai-sessions.ts @@ -1658,13 +1658,24 @@ const summaryMeasures_ = ($: SpanColumns) => { ), ) // The list's error rule, so the summary and the list badge agree. - const failed = $.StatusCode.eq("Error") - .or(stamp(MAPLE_AI_STAMP_ATTRS.error).eq("1")) + // On a stamped span the gateway's verdict alone, as `spanFailed`: it leaves + // out a paused tool call's copy that its framework ended in error. + const failed = stamp(MAPLE_AI_STAMP_ATTRS.error) + .eq("1") .or( - unstamped.and(isAi).and( - field("errorType") - .neq("") - .or(CH.inList($.SpanAttributes.get(RESPONSE_STATUS_ATTR), FAILED_RESPONSE_STATUSES)), + unstamped.and( + $.StatusCode.eq("Error").or( + isAi.and( + field("errorType") + .neq("") + .or( + CH.inList( + $.SpanAttributes.get(RESPONSE_STATUS_ATTR), + FAILED_RESPONSE_STATUSES, + ), + ), + ), + ), ), ) const conversationId = attr([