From 919f6cf4354ab9077bab93a6483e5fd902f43591 Mon Sep 17 00:00:00 2001 From: JeremyFunk Date: Tue, 29 Sep 2026 00:18:59 +0200 Subject: [PATCH 01/19] fix(agent-sessions): stop the prompt-cache check warning on uncacheable prompts Symptom: the Prompt cache check warned "Cache hit rate 0% over N calls; N missed the cache" on sessions whose prompts were all too short to cache. Seen on DSPy (prompts peak 870 tokens, OpenAI minimum 1024), Google ADK, LangChain, Haystack, CrewAI and Claude Agent SDK (Haiku 4.5 prompts of 1.2K-2.3K tokens against its 4,096-token minimum) test sessions. Cause: promptCacheCheck judged every call after the first, whatever its prompt size. Fix: calls under 1024 prompt tokens (the smallest minimum OpenAI, Anthropic and Gemini cache) are left out of the judgement, and a session whose calls report a cache-write bucket but never wrote or read the cache is skipped: nothing reached the model's minimum, or caching is off. The "too few calls" message now counts the calls actually judged. --- .../agent-sessions/src/session-checks.test.ts | 48 +++++++++++++++++++ packages/agent-sessions/src/session-checks.ts | 29 ++++++++++- 2 files changed, 75 insertions(+), 2 deletions(-) diff --git a/packages/agent-sessions/src/session-checks.test.ts b/packages/agent-sessions/src/session-checks.test.ts index 3c9490d4de..1ad695f5b0 100644 --- a/packages/agent-sessions/src/session-checks.test.ts +++ b/packages/agent-sessions/src/session-checks.test.ts @@ -1,3 +1,4 @@ +import type { AiSessionGenAiValues } from "@maple/domain/http" import { describe, expect, it } from "vitest" import { buildSessionChecks, type SessionCheck } from "./session-checks" @@ -397,6 +398,53 @@ describe("buildSessionChecks", () => { expect(byId(checks(firstTurn()), "prompt-cache").status).toBe("skipped") }) + // Short test conversations (DSPy peaked at 870 tokens; Claude Agent SDK on + // Haiku 4.5 sent 1.2K–2.3K against a 4,096-token minimum) cannot be cached, + // so reading them as misses warned "0% ... missed the cache" on every one. + it("skips the prompt cache when no prompt could have been cached", () => { + const call = (spanId: string, startMs: number, genAi: AiSessionGenAiValues) => + llmSpan({ + spanId, + parentSpanId: "a1", + startMs, + durationMs: SECOND, + genAi: { conversationId: "t1", usageOutputTokens: 50, ...genAi }, + }) + const session = (genAi: (i: number) => AiSessionGenAiValues) => + checks([ + agentSpan({ spanId: "a1", startMs: 0, durationMs: MINUTE, genAi: { conversationId: "t1" } }), + ...[0, 1, 2, 3, 4].map((i) => call(`c${i}`, (i + 1) * SECOND, genAi(i))), + ]) + + const short = byId( + session((i) => ({ usageInputTokens: 700 + 40 * i, usageCacheReadInputTokens: 0 })), + "prompt-cache", + ) + expect(short.status).toBe("skipped") + expect(short.headline).toBe( + "Only 0 model calls after the first had a prompt of 1024 tokens or more, the smallest a provider caches; at least 3 are needed to judge the prompt cache.", + ) + + const neverWritten = byId( + session((i) => ({ + usageInputTokens: 1_227 + 200 * i, + usageCacheReadInputTokens: 0, + usageCacheCreationInputTokens: 0, + })), + "prompt-cache", + ) + expect(neverWritten.status).toBe("skipped") + expect(neverWritten.headline).toMatch(/^No model call wrote to the prompt cache/) + + // Long enough and reporting no write bucket (OpenAI): a miss is a miss. + const missed = byId( + session(() => ({ usageInputTokens: 5_000, usageCacheReadInputTokens: 0 })), + "prompt-cache", + ) + expect(missed.status).toBe("warning") + expect(missed.headline).toBe("Cache hit rate 0% over 4 calls; 4 missed the cache") + }) + // The one rule behind red and amber, pinned per kind: a class that needs a // fix stays red when survived; everything else survived is amber. it("keeps a survived context overflow red and a survived provider error amber", () => { diff --git a/packages/agent-sessions/src/session-checks.ts b/packages/agent-sessions/src/session-checks.ts index ed215ef6c6..4b40c6032a 100644 --- a/packages/agent-sessions/src/session-checks.ts +++ b/packages/agent-sessions/src/session-checks.ts @@ -39,6 +39,9 @@ import { const CACHE_HIT_MIN_RATE = 0.5 /** Fewer calls than this say nothing about the cache either way. */ const CACHE_MIN_CALLS = 3 +/** The smallest prompt OpenAI, Anthropic or Gemini will cache: a shorter one + * cannot hit the cache, so missing it says nothing about the prefix. */ +const CACHE_MIN_PROMPT_TOKENS = 1024 /** A headline names this many findings before it counts the rest. */ const HEADLINE_MAX_CLAUSES = 3 @@ -627,6 +630,28 @@ function promptCacheCheck(llmCalls: readonly AiSessionSpan[]): SessionCheck { "No model call reported cache usage, so the prompt cache could not be checked.", ) } + if (reporting.length <= CACHE_MIN_CALLS) { + return check( + identity, + "skipped", + `Only ${plural(reporting.length, "model call")} reported cache usage; at least ${CACHE_MIN_CALLS + 1} are needed to judge the prompt cache.`, + ) + } + // A provider that reports cache writes (Anthropic) writes on the first call + // whose prompt is long enough for the model; none ever writing means no + // prompt reached that model's minimum, or caching was never requested. + const neverCached = reporting.every( + (span) => + span.genAi.usageCacheCreationInputTokens === 0 && + (span.genAi.usageCacheReadInputTokens ?? 0) === 0, + ) + if (neverCached) { + return check( + identity, + "skipped", + "No model call wrote to the prompt cache: the prompts are below the model's cacheable minimum, or caching is off.", + ) + } // The first call of a session cannot hit a cache nothing has written yet. const calls = reporting .slice(1) @@ -636,12 +661,12 @@ function promptCacheCheck(llmCalls: readonly AiSessionSpan[]): SessionCheck { read: buckets.cacheRead, prompt: buckets.input + buckets.cacheRead + buckets.cacheWrite, })) - .filter((call) => call.prompt > 0) + .filter((call) => call.prompt >= CACHE_MIN_PROMPT_TOKENS) if (calls.length < CACHE_MIN_CALLS) { return check( identity, "skipped", - `Only ${plural(reporting.length, "model call")} reported cache usage; at least ${CACHE_MIN_CALLS + 1} are needed to judge the prompt cache.`, + `Only ${plural(calls.length, "model call")} after the first had a prompt of ${CACHE_MIN_PROMPT_TOKENS} tokens or more, the smallest a provider caches; at least ${CACHE_MIN_CALLS} are needed to judge the prompt cache.`, ) } const rate = From e8d125d4489f6c6980db381c8d6cce1b13723c0b Mon Sep 17 00:00:00 2001 From: JeremyFunk Date: Tue, 29 Sep 2026 00:19:37 +0200 Subject: [PATCH 02/19] fix(agent-sessions): count model calls once in the provider-errors headline Symptom: the Provider errors check read "All 48 model calls were answered first time" on an OpenRouter Broadcast session of 16 calls ("All 30" for 10), and "All 16 model calls" on a Google ADK session with llmCalls 8. Cause: buildSessionChecks passed the count of every inference-classified span (session-checks.ts providerCheck call), so OpenRouter's `provider attempt N` and `generation` children and ADK's `call_llm` wrapper were counted beside the call they belong to. Fix: the headline uses summary.work.llmCalls, the netted count the session header already shows. --- .../agent-sessions/src/session-checks.test.ts | 41 +++++++++++++++++++ packages/agent-sessions/src/session-checks.ts | 2 +- 2 files changed, 42 insertions(+), 1 deletion(-) diff --git a/packages/agent-sessions/src/session-checks.test.ts b/packages/agent-sessions/src/session-checks.test.ts index 1ad695f5b0..70763c8733 100644 --- a/packages/agent-sessions/src/session-checks.test.ts +++ b/packages/agent-sessions/src/session-checks.test.ts @@ -445,6 +445,47 @@ describe("buildSessionChecks", () => { expect(missed.headline).toBe("Cache hit rate 0% over 4 calls; 4 missed the cache") }) + // OpenRouter Broadcast: every request is a `chat` root carrying the usage + // plus a `provider attempt N` and a `generation` child, all op `chat`. The + // headline read "All 48 model calls" for a session of 16. + it("counts model calls in the provider headline once per call, not per chat span", () => { + const request = (n: number) => { + const traceId = `trace-${n}` + const root = `gen-${n}` + const at = n * MINUTE + return [ + llmSpan({ + spanId: root, + traceId, + spanName: "LLM Generation", + startMs: at, + durationMs: 4_842, + genAi: { usageInputTokens: 58_797, usageOutputTokens: 220, responseId: `gen-${n}` }, + }), + llmSpan({ + spanId: `${root}-attempt`, + parentSpanId: root, + traceId, + spanName: "provider attempt 1: OpenAI", + startMs: at + 184, + durationMs: 2_019, + genAi: { responseId: `gen-${n}:attempt-0` }, + }), + llmSpan({ + spanId: `${root}-generation`, + parentSpanId: root, + traceId, + spanName: "generation", + startMs: at + 200, + durationMs: 4_402, + genAi: { responseId: `gen-${n}:generation` }, + }), + ] + } + const report = checks([...request(0), ...request(1)]) + expect(byId(report, "provider").headline).toBe("All 2 model calls were answered first time") + }) + // The one rule behind red and amber, pinned per kind: a class that needs a // fix stays red when survived; everything else survived is amber. it("keeps a survived context overflow red and a survived provider error amber", () => { diff --git a/packages/agent-sessions/src/session-checks.ts b/packages/agent-sessions/src/session-checks.ts index 4b40c6032a..1807cc2b29 100644 --- a/packages/agent-sessions/src/session-checks.ts +++ b/packages/agent-sessions/src/session-checks.ts @@ -138,7 +138,7 @@ export function buildSessionChecks( ...completionCheck(of("incomplete")), contextWindowCheck(of("contextExceeded"), llmCalls), rateLimitCheck(of("rateLimited")), - providerCheck(of("providerError"), of("providerRetry"), llmCalls.length), + providerCheck(of("providerError"), of("providerRetry"), summary.work.llmCalls), refusalCheck(of("refusal")), replyLengthCheck(of("truncation"), llmCalls), structuredOutputCheck(of("invalidOutput")), From b845e089e5819a9434203a5b2ea35e19a5d691d4 Mon Sep 17 00:00:00 2001 From: JeremyFunk Date: Tue, 29 Sep 2026 00:20:26 +0200 Subject: [PATCH 03/19] fix(agent-sessions): judge the prompt cache once per model call Symptom: on Google ADK sessions the Prompt cache check read "0% over 15 calls" (a1) and "over 19 calls" (b) while the session had 8 and 10 model calls: the checks did not net ADK's duplicate reporter pair. Cause: promptCacheCheck took every inference-classified span, so ADK's `call_llm` wrapper and the `generate_content` span beneath it, both stamped with the same usage, were judged as two calls each. Fix: session-summary exports sessionLlmCalls, the netted call list behind work.llmCalls, and the cache check reads it. The provider headline half of the same symptom is fixed in the previous commit. --- .../agent-sessions/src/session-checks.test.ts | 36 ++++++++++++++++++- packages/agent-sessions/src/session-checks.ts | 5 ++- .../agent-sessions/src/session-summary.ts | 8 +++++ 3 files changed, 47 insertions(+), 2 deletions(-) diff --git a/packages/agent-sessions/src/session-checks.test.ts b/packages/agent-sessions/src/session-checks.test.ts index 70763c8733..62fb68cc88 100644 --- a/packages/agent-sessions/src/session-checks.test.ts +++ b/packages/agent-sessions/src/session-checks.test.ts @@ -4,7 +4,7 @@ import { describe, expect, it } from "vitest" import { buildSessionChecks, type SessionCheck } from "./session-checks" import { buildSessionSummary } from "./session-summary" import { buildSessionTurns } from "./session-turns" -import { agentSpan, llmSpan, toolSpan } from "./span-test-support" +import { agentSpan, llmSpan, makeSpan, toolSpan } from "./span-test-support" const SECOND = 1000 const MINUTE = 60 * SECOND @@ -486,6 +486,40 @@ describe("buildSessionChecks", () => { expect(byId(report, "provider").headline).toBe("All 2 model calls were answered first time") }) + // Google ADK reports each call twice: `call_llm` (no operation, a model) + // over `generate_content`, both with the same usage. The cache check read + // "over 15 calls" for a session of 8. + it("judges the prompt cache once per model call when a wrapper repeats the usage", () => { + const usage = (i: number): AiSessionGenAiValues => ({ + requestModel: "openrouter/openai/gpt-4o-mini", + usageInputTokens: 2_000, + usageCacheReadInputTokens: i === 0 ? 0 : 1_536, + usageOutputTokens: 32, + }) + const report = checks([ + agentSpan({ spanId: "a1", startMs: 0, durationMs: MINUTE, agentName: "assistant" }), + ...[0, 1, 2, 3, 4].flatMap((i) => [ + makeSpan({ + spanId: `call-${i}`, + parentSpanId: "a1", + spanName: "call_llm", + startMs: (i + 1) * SECOND, + durationMs: 2 * SECOND, + genAi: usage(i), + }), + llmSpan({ + spanId: `gen-${i}`, + parentSpanId: `call-${i}`, + spanName: "generate_content", + startMs: (i + 1) * SECOND, + durationMs: 2 * SECOND, + genAi: { operationName: "generate_content", ...usage(i) }, + }), + ]), + ]) + expect(byId(report, "prompt-cache").headline).toBe("Cache hit rate 77% over 4 calls") + }) + // The one rule behind red and amber, pinned per kind: a class that needs a // fix stays red when survived; everything else survived is amber. it("keeps a survived context overflow red and a survived provider error amber", () => { diff --git a/packages/agent-sessions/src/session-checks.ts b/packages/agent-sessions/src/session-checks.ts index 1807cc2b29..e79f80aef1 100644 --- a/packages/agent-sessions/src/session-checks.ts +++ b/packages/agent-sessions/src/session-checks.ts @@ -21,6 +21,7 @@ import { type SessionVerdict, } from "./session-findings" import { + sessionLlmCalls, spanTokenBuckets, type SessionFailureKind, type SessionSummary, @@ -152,7 +153,9 @@ export function buildSessionChecks( ...otherErrorsCheck(errors.filter((finding) => finding.tool === undefined)), repetitionCheck(of("repetition"), summary, coverage), stallCheck(of("stall")), - promptCacheCheck(llmCalls), + // Per call, so a model call its framework also rolled up (ADK `call_llm` over + // `generate_content`) is judged once. + promptCacheCheck(sessionLlmCalls(spans)), ].sort( (a, b) => STATUS_RANK[a.status] - STATUS_RANK[b.status] || diff --git a/packages/agent-sessions/src/session-summary.ts b/packages/agent-sessions/src/session-summary.ts index a7e2e71aa8..52f5cf2d22 100644 --- a/packages/agent-sessions/src/session-summary.ts +++ b/packages/agent-sessions/src/session-summary.ts @@ -582,6 +582,14 @@ function countedLlmCalls( return [...unkeyed, ...byResponse.values()].sort((a, b) => spanStartMs(a) - spanStartMs(b)) } +/** The spans `work.llmCalls` counts, in start order: for a reading that needs + * one span per model call rather than every observation of it. */ +export function sessionLlmCalls(spans: readonly AiSessionSpan[]): readonly AiSessionSpan[] { + const byId = new Map(spans.map((span) => [span.spanId, span])) + const usage = countableUsageSpans(spans, byId) + return countedLlmCalls(spans, byId, usage.bySpan, usage.costs) +} + /** * Each reporter charged to the NEAREST ancestor that also reports, so a * two-level roll-up subtracts each figure once rather than at every level. From 0815f47f4dbc74eb85bdd81adc87c10457543c7f Mon Sep 17 00:00:00 2001 From: JeremyFunk Date: Tue, 29 Sep 2026 00:21:21 +0200 Subject: [PATCH 04/19] fix(agent-sessions): count a nested model call's time once in agentTime Symptom: vitals read 204.8 s of agent time in a 125.5 s OpenRouter Broadcast session with no parallel work. Cause: computeAgentTime charged every inference span its full duration, and Broadcast nests a `provider attempt N` and a `generation` span (both op `chat`) under each `LLM Generation`, so one call was charged up to three times. Google ADK's `call_llm` over `generate_content` doubles the same way. Fix: an inference span inside another inference span is the same call observed twice; only the outermost is charged. A TTFT that only a nested level reported still splits the call. --- .../src/session-summary.test.ts | 22 +++++++++++ .../agent-sessions/src/session-summary.ts | 38 +++++++++++++++---- 2 files changed, 52 insertions(+), 8 deletions(-) diff --git a/packages/agent-sessions/src/session-summary.test.ts b/packages/agent-sessions/src/session-summary.test.ts index 580572b624..344ef42ae8 100644 --- a/packages/agent-sessions/src/session-summary.test.ts +++ b/packages/agent-sessions/src/session-summary.test.ts @@ -114,6 +114,28 @@ describe("buildSessionSummary — time", () => { expect(summary.agentTime.segments.map((entry) => entry.kind)).toEqual(["inference", "tool"]) expect(summary.agentTime.totalMs).toBe(7 * SECOND) }) + + // OpenRouter Broadcast nests a `provider attempt` and a `generation` span, + // both op `chat`, under each `LLM Generation`: summing all three read 204.8s + // of agent time in a 125.5s session. + it("charges a model call observed at several levels once, at the outermost", () => { + const summary = summarize([ + llmSpan({ spanId: "root", spanName: "LLM Generation", startMs: 0, durationMs: 4_842 }), + llmSpan({ spanId: "attempt", parentSpanId: "root", startMs: 184, durationMs: 2_019 }), + llmSpan({ + spanId: "generation", + parentSpanId: "root", + startMs: 200, + durationMs: 4_402, + ttftSeconds: 3.6, + }), + ]) + + expect(summary.agentTime.totalMs).toBe(4_842) + // The TTFT only the nested span reported still splits the call. + expect(segment(summary.agentTime.segments, "ttft")).toBe(3_600) + expect(segment(summary.agentTime.segments, "inference")).toBe(1_242) + }) }) describe("buildSessionSummary — failed", () => { diff --git a/packages/agent-sessions/src/session-summary.ts b/packages/agent-sessions/src/session-summary.ts index 52f5cf2d22..7352f5c9d8 100644 --- a/packages/agent-sessions/src/session-summary.ts +++ b/packages/agent-sessions/src/session-summary.ts @@ -356,29 +356,51 @@ export function findIdleGaps(spans: readonly AiSessionSpan[]): readonly IdleGap[ * A TTFT splits its own span: the wait is not inference, and a session whose * time is mostly first-token latency is a different session from one that is * mostly generation. Agent and non-AI spans contribute nothing — an agent span - * covers its children, and adding it would count the same work twice. + * covers its children, and adding it would count the same work twice. For the + * same reason an inference span inside another is the same model call observed + * twice (OpenRouter's `LLM Generation` over its `generation`, ADK's `call_llm` + * over `generate_content`): the outermost carries the time, and lends its TTFT + * from the level that reported one. */ export function computeAgentTime(spans: readonly AiSessionSpan[]): SessionAgentTime { const totals = new Map() const add = (kind: AgentTimeKind, ms: number) => { if (ms > 0) totals.set(kind, (totals.get(kind) ?? 0) + ms) } + const byId = new Map(spans.map((span) => [span.spanId, span])) + const outermostCall = (span: AiSessionSpan): AiSessionSpan => { + let call = span + const seen = new Set([span.spanId]) + let parent = byId.get(span.parentSpanId) + while (parent !== undefined && !seen.has(parent.spanId)) { + seen.add(parent.spanId) + if (classifyAiSpan(parent) === "inference") call = parent + parent = byId.get(parent.parentSpanId) + } + return call + } + const calls: AiSessionSpan[] = [] + const nestedTtftMs = new Map() for (const span of spans) { - const spanStart = spanStartMs(span) - const spanEnd = spanEndMs(span) const category = classifyAiSpan(span) - if (category !== "tool" && category !== "inference") continue - if (category === "tool") { - add("tool", spanEnd - spanStart) + if (category === "tool") add("tool", span.durationMs) + if (category !== "inference") continue + const call = outermostCall(span) + if (call === span) { + calls.push(span) continue } const ttftMs = spanTtftMs(span) + if (ttftMs !== undefined && !nestedTtftMs.has(call.spanId)) nestedTtftMs.set(call.spanId, ttftMs) + } + for (const call of calls) { + const ttftMs = spanTtftMs(call) ?? nestedTtftMs.get(call.spanId) // A TTFT longer than the span itself is instrumentation disagreeing with // itself; the span's own duration is the one both classes must fit in. - const ttft = ttftMs === undefined ? 0 : Math.min(ttftMs, spanEnd - spanStart) + const ttft = ttftMs === undefined ? 0 : Math.min(ttftMs, call.durationMs) add("ttft", ttft) - add("inference", spanEnd - spanStart - ttft) + add("inference", call.durationMs - ttft) } const segments = AGENT_TIME_KIND_ORDER.map((kind) => ({ kind, ms: totals.get(kind) ?? 0 })).filter( From 008a9d52f434d01c4e6fa5acae9d4d8f98d67eed Mon Sep 17 00:00:00 2001 From: JeremyFunk Date: Tue, 29 Sep 2026 00:22:18 +0200 Subject: [PATCH 05/19] fix(agent-sessions): label a turn from its model calls before framework spans Symptom: every turn of a checkpointed LangGraph thread (OpenInference LangChain dual-write) was labeled with turn 1's prompt, e.g. all a1 turns 1-6 read "Hi! Briefly introduce yourself.". Cause: turnLabel (session-turns.ts) fell back to the first span in start order with any user message. The LangGraph `model` node CHAIN span starts before its ChatOpenAI call and carries only the thread's first message. Fix: after the anchor, model calls are asked first, as the transcript's userRows already does; other spans remain the last resort. --- .../agent-sessions/src/session-turns.test.ts | 43 +++++++++++++++++++ packages/agent-sessions/src/session-turns.ts | 7 ++- 2 files changed, 48 insertions(+), 2 deletions(-) diff --git a/packages/agent-sessions/src/session-turns.test.ts b/packages/agent-sessions/src/session-turns.test.ts index f73385d983..b3c3d8d8ec 100644 --- a/packages/agent-sessions/src/session-turns.test.ts +++ b/packages/agent-sessions/src/session-turns.test.ts @@ -298,6 +298,49 @@ describe("buildSessionTurns", () => { expect(turns[0]!.label).toBe("deploy the worker") }) + // LangGraph with a checkpointer, through the OpenInference dual-write: the + // `model` node (a CHAIN span, no operation) starts before its model call and + // carries only the thread's FIRST message, so every turn read turn 1's prompt. + it("labels from a model call before a framework span that started earlier", () => { + const turn = (n: number, prompts: readonly string[]) => { + const at = n * 60 * SECOND + return [ + agentSpan({ + spanId: `assistant-${n}`, + startMs: at, + durationMs: 2 * SECOND, + agentName: "assistant", + }), + makeSpan({ + spanId: `model-${n}`, + parentSpanId: `assistant-${n}`, + spanName: "model", + startMs: at + 10, + durationMs: SECOND, + vendorId: "unknown:openinference", + genAi: { inputMessages: userMessages(prompts[0]!) }, + }), + llmSpan({ + spanId: `chat-${n}`, + parentSpanId: `model-${n}`, + spanName: "ChatOpenAI", + startMs: at + 12, + durationMs: SECOND, + genAi: { inputMessages: userMessages(...prompts) }, + }), + ] + } + const turns = buildSessionTurns([ + ...turn(0, ["Hi! Briefly introduce yourself."]), + ...turn(1, ["Hi! Briefly introduce yourself.", "What's the weather in Berlin?"]), + ]) + + expect(turns.map((t) => t.label)).toEqual([ + "Hi! Briefly introduce yourself.", + "What's the weather in Berlin?", + ]) + }) + it("has no label when message content was not captured", () => { const turns = buildSessionTurns([agentSpan({ spanId: "agent", startMs: 0, durationMs: SECOND })]) diff --git a/packages/agent-sessions/src/session-turns.ts b/packages/agent-sessions/src/session-turns.ts index 1299a5be73..66e8ea1991 100644 --- a/packages/agent-sessions/src/session-turns.ts +++ b/packages/agent-sessions/src/session-turns.ts @@ -401,12 +401,15 @@ function findAnchors(ordered: readonly AiSessionSpan[]): readonly TurnAnchor[] { * * The anchor is asked first — on a `chat`-shaped span `gen_ai.input.messages` * is the whole history sent to the model, so a descendant several turns deep - * still carries turn 1's opening prompt. + * still carries turn 1's opening prompt. Model calls come next: the history a + * model was sent ends on the turn's prompt, while a framework's own node span + * may carry only what the thread started with (LangGraph's `model` node under a + * checkpointer holds the thread's first message on every turn). */ function turnLabel(anchor: AiSessionSpan, turnSpans: readonly AiSessionSpan[]): string | undefined { const fromAnchor = lastUserMessageText(anchor.genAi.inputMessages) if (fromAnchor !== undefined) return fromAnchor - for (const span of turnSpans) { + for (const span of [...turnSpans.filter(isLlmCall), ...turnSpans]) { const text = lastUserMessageText(span.genAi.inputMessages) if (text !== undefined) return text } From 2b3faa060b5b2aed10a4c135756a6b9030d74a69 Mon Sep 17 00:00:00 2001 From: JeremyFunk Date: Tue, 29 Sep 2026 00:22:56 +0200 Subject: [PATCH 06/19] fix(agent-sessions): read past a "New task:" lead-in in turn labels Symptom: every smolagents turn label and the session title read "New task:". Cause: proseLine (session-turns.ts) labels a turn with the first non-empty line of the user message, and smolagents sends every task as "New task:\n". Fix: a first line of at most three words ending in a colon is a lead-in; the line after it labels the turn. A longer sentence ending in a colon, or a lead-in with nothing after it, is kept as before. --- .../agent-sessions/src/session-turns.test.ts | 11 +++++++++ packages/agent-sessions/src/session-turns.ts | 23 +++++++++++++------ 2 files changed, 27 insertions(+), 7 deletions(-) diff --git a/packages/agent-sessions/src/session-turns.test.ts b/packages/agent-sessions/src/session-turns.test.ts index b3c3d8d8ec..f16eec7b9b 100644 --- a/packages/agent-sessions/src/session-turns.test.ts +++ b/packages/agent-sessions/src/session-turns.test.ts @@ -539,6 +539,17 @@ describe("turn labels", () => { expect(long?.endsWith("…")).toBe(true) }) + // smolagents sends every task as "New task:\n", so every turn and the + // session title read "New task:". + it("reads past a short lead-in ending in a colon", () => { + expect(labelFor([{ role: "user", content: "New task:\nWhat is 17 * 23?" }])).toBe("What is 17 * 23?") + expect(labelFor([{ role: "user", content: "New task:" }])).toBe("New task:") + // A sentence that ends in a colon is the prompt itself. + expect(labelFor([{ role: "user", content: "Rename these files to kebab case:\na.ts" }])).toBe( + "Rename these files to kebab case:", + ) + }) + // Vendors write "User" as readily as "user", and the transcript's own row // builders already read the role case-insensitively. it("reads a capitalised role", () => { diff --git a/packages/agent-sessions/src/session-turns.ts b/packages/agent-sessions/src/session-turns.ts index 66e8ea1991..cc9b016936 100644 --- a/packages/agent-sessions/src/session-turns.ts +++ b/packages/agent-sessions/src/session-turns.ts @@ -464,14 +464,23 @@ function messageText(value: unknown): string | undefined { return undefined } -/** The message's first non-empty line, collapsed to one line's worth of text. */ +/** The message's first non-empty line, collapsed to one line's worth of text — + * or the line after it, when the first is a lead-in. */ function proseLine(value: string): string | undefined { - for (const rawLine of value.split("\n")) { - const line = rawLine.trim().replace(/\s+/g, " ") - if (line.length === 0) continue - return line.length > MAX_LABEL_LENGTH ? `${line.slice(0, MAX_LABEL_LENGTH - 1)}…` : line - } - return undefined + const lines = value + .split("\n") + .map((line) => line.trim().replace(/\s+/g, " ")) + .filter((line) => line.length > 0) + const [first, second] = lines + if (first === undefined) return undefined + const line = second !== undefined && isLeadIn(first) ? second : first + return line.length > MAX_LABEL_LENGTH ? `${line.slice(0, MAX_LABEL_LENGTH - 1)}…` : line +} + +/** A short line ending in a colon names what follows rather than saying it: + * smolagents opens every task with "New task:". */ +function isLeadIn(line: string): boolean { + return line.endsWith(":") && line.split(" ").length <= 3 } function isRecord(value: unknown): value is Record { From ef98e2ca334cb33da3194ebf255d9d1f6554d5cd Mon Sep 17 00:00:00 2001 From: JeremyFunk Date: Tue, 29 Sep 2026 00:24:57 +0200 Subject: [PATCH 07/19] fix(agent-sessions): fold a turn that opened no work into the next one Symptom: a Microsoft Agent Framework workflow session showed its one-span `workflow.build` trace as Turn 1 (no label, no agent, 1 span), the real run as Turn 2, and get_agent_session returned no title. Cause: `workflow.build` carries the session's id in its own trace and is read as an agent root by its name, so buildSessionTurns opened a turn on it; the summary title is turn 1's label. Fix: on the agent-root and per-trace rules, where the boundary is a heuristic, an anchor whose spans hold no model or tool call, no user message and nothing failed joins the next turn, as spans before the first anchor join turn 1. Conversation-id turns are explicit and kept, and a session with no work anywhere keeps its anchors. --- .../agent-sessions/src/session-turns.test.ts | 56 +++++++++++++++++++ packages/agent-sessions/src/session-turns.ts | 24 ++++++++ 2 files changed, 80 insertions(+) diff --git a/packages/agent-sessions/src/session-turns.test.ts b/packages/agent-sessions/src/session-turns.test.ts index f16eec7b9b..4dce74f5b0 100644 --- a/packages/agent-sessions/src/session-turns.test.ts +++ b/packages/agent-sessions/src/session-turns.test.ts @@ -3,6 +3,7 @@ import { describe, expect, it } from "vitest" import type { AiSessionSpan } from "@maple/domain/http" import { agentSpan, llmSpan, makeSpan, toolSpan, userMessages } from "./span-test-support" +import { buildSessionSummary } from "./session-summary" import { buildSessionTurns, classifyAiSpan, isLlmCall, spanTtftMs } from "./session-turns" const SECOND = 1000 @@ -170,6 +171,61 @@ describe("buildSessionTurns", () => { expect(Number.isFinite(turns[0]!.startMs)).toBe(true) }) + // Microsoft Agent Framework workflows: `workflow.build` is a one-span trace + // of its own, read as an agent root by its name, ahead of `workflow.run`. + // It became an empty turn 1 with no label, and the session lost its title. + it("folds an anchor that opened no work into the turn after it", () => { + const maf = { + vendorId: "microsoft_agent_framework", + sessionId: "wf-1", + genAi: { conversationId: "wf-1" }, + } + const spans = [ + makeSpan({ + ...maf, + spanId: "build", + traceId: "trace-build", + spanName: "workflow.build", + startMs: 0, + durationMs: 2, + }), + makeSpan({ + ...maf, + spanId: "run", + traceId: "trace-run", + spanName: "workflow.run", + startMs: 50, + durationMs: 10 * SECOND, + }), + agentSpan({ + ...maf, + spanId: "orchestrator", + parentSpanId: "run", + traceId: "trace-run", + startMs: 60, + durationMs: 4 * SECOND, + }), + llmSpan({ + ...maf, + spanId: "chat", + parentSpanId: "orchestrator", + traceId: "trace-run", + startMs: 70, + durationMs: 4 * SECOND, + genAi: { + conversationId: "wf-1", + inputMessages: userMessages("Produce a mini briefing about Amsterdam"), + }, + }), + ] + const turns = buildSessionTurns(spans) + + expect(turns).toHaveLength(1) + expect(turns[0]!.spans.map((span) => span.spanId)).toEqual(["build", "run", "orchestrator", "chat"]) + expect(turns[0]!.label).toBe("Produce a mini briefing about Amsterdam") + expect(buildSessionSummary({ spans, turns }).title).toBe("Produce a mini briefing about Amsterdam") + }) + it("falls back to root agent invocations when no conversation id exists", () => { const turns = buildSessionTurns([ agentSpan({ spanId: "agent-1", startMs: 0, durationMs: 10 * SECOND }), diff --git a/packages/agent-sessions/src/session-turns.ts b/packages/agent-sessions/src/session-turns.ts index cc9b016936..ee0322d9e3 100644 --- a/packages/agent-sessions/src/session-turns.ts +++ b/packages/agent-sessions/src/session-turns.ts @@ -163,6 +163,8 @@ export interface SessionTurn { readonly traceIds: readonly string[] } +const WORK_CATEGORIES: ReadonlySet = new Set(["inference", "tool"]) + interface TurnAnchor { readonly span: AiSessionSpan readonly kind: TurnAnchorKind @@ -218,6 +220,28 @@ export function buildSessionTurns(spans: readonly AiSessionSpan[]): readonly Ses buckets[turnOf(span) ?? cursor].push(span) } + // On rules 2 and 3 the boundary is a guess, and an anchor that opened no + // work — no model or tool call, no prompt, nothing failed — is not a turn: + // Microsoft Agent Framework's one-span `workflow.build` trace ahead of its + // `workflow.run` became an empty turn 1 that also took the session's title. + // Its spans join the next turn, as spans before the first anchor join turn 1; + // the cursor filled the buckets in start order, so they stay in it. A session + // with no work anywhere keeps its anchors. + const opened = (bucket: readonly AiSessionSpan[]) => + bucket.some( + (span) => + WORK_CATEGORIES.has(classifyAiSpan(span)) || + spanFailed(span) || + lastUserMessageText(span.genAi.inputMessages) !== undefined, + ) + if (anchors[0]?.kind !== "conversation" && buckets.some(opened)) { + for (let i = 0; i < buckets.length - 1; i++) { + if (opened(buckets[i])) continue + buckets[i + 1] = [...buckets[i], ...buckets[i + 1]] + buckets[i] = [] + } + } + // A turn with no spans has no start, no end and nothing to draw. Rule 1 can no // longer produce one — an anchor always carries its own id, so it lands in its // own bucket even when another anchor shares its millisecond — but two From 94d0a272da4d063c9ab465099061f63294a7e350 Mon Sep 17 00:00:00 2001 From: JeremyFunk Date: Tue, 29 Sep 2026 00:25:27 +0200 Subject: [PATCH 08/19] fix(agent-sessions): stop at the second line when reading a turn label A captured prompt can be tens of kilobytes; the label reads at most two non-empty lines, so it stops collapsing whitespace after them. --- packages/agent-sessions/src/session-turns.ts | 9 +++++---- 1 file changed, 5 insertions(+), 4 deletions(-) diff --git a/packages/agent-sessions/src/session-turns.ts b/packages/agent-sessions/src/session-turns.ts index ee0322d9e3..f82bfa9b70 100644 --- a/packages/agent-sessions/src/session-turns.ts +++ b/packages/agent-sessions/src/session-turns.ts @@ -491,10 +491,11 @@ function messageText(value: unknown): string | undefined { /** The message's first non-empty line, collapsed to one line's worth of text — * or the line after it, when the first is a lead-in. */ function proseLine(value: string): string | undefined { - const lines = value - .split("\n") - .map((line) => line.trim().replace(/\s+/g, " ")) - .filter((line) => line.length > 0) + const lines: string[] = [] + for (const rawLine of value.split("\n")) { + const line = rawLine.trim().replace(/\s+/g, " ") + if (line.length > 0 && lines.push(line) === 2) break + } const [first, second] = lines if (first === undefined) return undefined const line = second !== undefined && isLeadIn(first) ? second : first From 202f7eb83e09bed0ae2d7f785f78669d2f777d08 Mon Sep 17 00:00:00 2001 From: JeremyFunk Date: Tue, 29 Sep 2026 00:29:14 +0200 Subject: [PATCH 09/19] fix(agent-sessions): leave out the first cacheable call, not the first call, in the prompt-cache check Follow-up to the uncacheable-prompt fix: the size filter ran after dropping the first call, so a short opening call (a title or router call) was dropped instead of the first long call, which writes the cache and then read as a miss. --- packages/agent-sessions/src/session-checks.test.ts | 14 +++++++++++++- packages/agent-sessions/src/session-checks.ts | 8 ++++---- 2 files changed, 17 insertions(+), 5 deletions(-) diff --git a/packages/agent-sessions/src/session-checks.test.ts b/packages/agent-sessions/src/session-checks.test.ts index 62fb68cc88..4649026093 100644 --- a/packages/agent-sessions/src/session-checks.test.ts +++ b/packages/agent-sessions/src/session-checks.test.ts @@ -422,7 +422,7 @@ describe("buildSessionChecks", () => { ) expect(short.status).toBe("skipped") expect(short.headline).toBe( - "Only 0 model calls after the first had a prompt of 1024 tokens or more, the smallest a provider caches; at least 3 are needed to judge the prompt cache.", + "Only 0 model calls had a prompt of 1024 tokens or more, the smallest a provider caches; at least 4 are needed to judge the prompt cache.", ) const neverWritten = byId( @@ -443,6 +443,18 @@ describe("buildSessionChecks", () => { ) expect(missed.status).toBe("warning") expect(missed.headline).toBe("Cache hit rate 0% over 4 calls; 4 missed the cache") + + // A short opening call (a title, a router) is not the one that wrote the + // cache: the first long call is, and it is not a miss. + const shortFirst = byId( + session((i) => + i === 0 + ? { usageInputTokens: 300, usageCacheReadInputTokens: 0 } + : { usageInputTokens: 2_000, usageCacheReadInputTokens: i === 1 ? 0 : 1_000 }, + ), + "prompt-cache", + ) + expect(shortFirst.headline).toBe("Cache hit rate 50% over 3 calls") }) // OpenRouter Broadcast: every request is a `chat` root carrying the usage diff --git a/packages/agent-sessions/src/session-checks.ts b/packages/agent-sessions/src/session-checks.ts index e79f80aef1..5a38c7c6c7 100644 --- a/packages/agent-sessions/src/session-checks.ts +++ b/packages/agent-sessions/src/session-checks.ts @@ -655,9 +655,8 @@ function promptCacheCheck(llmCalls: readonly AiSessionSpan[]): SessionCheck { "No model call wrote to the prompt cache: the prompts are below the model's cacheable minimum, or caching is off.", ) } - // The first call of a session cannot hit a cache nothing has written yet. - const calls = reporting - .slice(1) + // The first cacheable call cannot hit a cache nothing has written yet. + const cacheable = reporting .map(spanTokenBuckets) .filter((buckets) => buckets !== undefined) .map((buckets) => ({ @@ -665,11 +664,12 @@ function promptCacheCheck(llmCalls: readonly AiSessionSpan[]): SessionCheck { prompt: buckets.input + buckets.cacheRead + buckets.cacheWrite, })) .filter((call) => call.prompt >= CACHE_MIN_PROMPT_TOKENS) + const calls = cacheable.slice(1) if (calls.length < CACHE_MIN_CALLS) { return check( identity, "skipped", - `Only ${plural(calls.length, "model call")} after the first had a prompt of ${CACHE_MIN_PROMPT_TOKENS} tokens or more, the smallest a provider caches; at least ${CACHE_MIN_CALLS} are needed to judge the prompt cache.`, + `Only ${plural(cacheable.length, "model call")} had a prompt of ${CACHE_MIN_PROMPT_TOKENS} tokens or more, the smallest a provider caches; at least ${CACHE_MIN_CALLS + 1} are needed to judge the prompt cache.`, ) } const rate = From 9f76b7c8146760d7909a92db9d927e4d4207b398 Mon Sep 17 00:00:00 2001 From: JeremyFunk Date: Tue, 29 Sep 2026 00:29:15 +0200 Subject: [PATCH 10/19] fix(agent-sessions): match only smolagents' "New task:" lead-in in turn labels Follow-up to the lead-in fix: any first line of three words ending in a colon was skipped, so a user's "Fix this:" over pasted code labeled the turn with the code's first line. The rule now matches the framework's literal heading. --- packages/agent-sessions/src/session-turns.test.ts | 8 +++----- packages/agent-sessions/src/session-turns.ts | 13 ++++++------- 2 files changed, 9 insertions(+), 12 deletions(-) diff --git a/packages/agent-sessions/src/session-turns.test.ts b/packages/agent-sessions/src/session-turns.test.ts index 4dce74f5b0..4ad3645b96 100644 --- a/packages/agent-sessions/src/session-turns.test.ts +++ b/packages/agent-sessions/src/session-turns.test.ts @@ -597,13 +597,11 @@ describe("turn labels", () => { // smolagents sends every task as "New task:\n", so every turn and the // session title read "New task:". - it("reads past a short lead-in ending in a colon", () => { + it("reads past smolagents' lead-in", () => { expect(labelFor([{ role: "user", content: "New task:\nWhat is 17 * 23?" }])).toBe("What is 17 * 23?") expect(labelFor([{ role: "user", content: "New task:" }])).toBe("New task:") - // A sentence that ends in a colon is the prompt itself. - expect(labelFor([{ role: "user", content: "Rename these files to kebab case:\na.ts" }])).toBe( - "Rename these files to kebab case:", - ) + // A user's own line ending in a colon is the prompt itself. + expect(labelFor([{ role: "user", content: "Fix this:\n```ts" }])).toBe("Fix this:") }) // Vendors write "User" as readily as "user", and the transcript's own row diff --git a/packages/agent-sessions/src/session-turns.ts b/packages/agent-sessions/src/session-turns.ts index f82bfa9b70..e5b57637d3 100644 --- a/packages/agent-sessions/src/session-turns.ts +++ b/packages/agent-sessions/src/session-turns.ts @@ -488,6 +488,11 @@ function messageText(value: unknown): string | undefined { return undefined } +/** A framework's own heading over the prompt, which names what follows rather + * than saying it: smolagents opens every task with "New task:". Matched + * literally, since a user's "Fix this:" over pasted code is the prompt. */ +const LEAD_IN = /^new task:$/i + /** The message's first non-empty line, collapsed to one line's worth of text — * or the line after it, when the first is a lead-in. */ function proseLine(value: string): string | undefined { @@ -498,16 +503,10 @@ function proseLine(value: string): string | undefined { } const [first, second] = lines if (first === undefined) return undefined - const line = second !== undefined && isLeadIn(first) ? second : first + const line = second !== undefined && LEAD_IN.test(first) ? second : first return line.length > MAX_LABEL_LENGTH ? `${line.slice(0, MAX_LABEL_LENGTH - 1)}…` : line } -/** A short line ending in a colon names what follows rather than saying it: - * smolagents opens every task with "New task:". */ -function isLeadIn(line: string): boolean { - return line.endsWith(":") && line.split(" ").length <= 3 -} - function isRecord(value: unknown): value is Record { return typeof value === "object" && value !== null && !Array.isArray(value) } From 7e5ca580da6a3ec6d1c05825333d5af98949bc3a Mon Sep 17 00:00:00 2001 From: JeremyFunk Date: Tue, 29 Sep 2026 00:44:17 +0200 Subject: [PATCH 11/19] fix(agent-sessions): read a never-written prompt cache as uncacheable on Anthropic only Follow-up to the uncacheable-prompt fix (#25). The never-written skip read a zero cache-write bucket as "nothing was cacheable", but OpenAI- model emitters (CrewAI, Haystack, LiteLLM, Mastra, Microsoft Agent Framework, Strands, Vercel AI SDK) stamp a zero write bucket on every call, so an OpenAI session with long prompts and a real 0% hit rate was skipped instead of warned. The rule now applies only when every reporting call is an Anthropic call. The separate "too few calls" early return is dropped: the cacheable-call gate below it covers it. --- .../agent-sessions/src/session-checks.test.ts | 27 ++++++++++++------- packages/agent-sessions/src/session-checks.ts | 17 +++++------- 2 files changed, 24 insertions(+), 20 deletions(-) diff --git a/packages/agent-sessions/src/session-checks.test.ts b/packages/agent-sessions/src/session-checks.test.ts index 4649026093..79e6320fc4 100644 --- a/packages/agent-sessions/src/session-checks.test.ts +++ b/packages/agent-sessions/src/session-checks.test.ts @@ -427,6 +427,7 @@ describe("buildSessionChecks", () => { const neverWritten = byId( session((i) => ({ + providerName: "anthropic", usageInputTokens: 1_227 + 200 * i, usageCacheReadInputTokens: 0, usageCacheCreationInputTokens: 0, @@ -434,15 +435,23 @@ describe("buildSessionChecks", () => { "prompt-cache", ) expect(neverWritten.status).toBe("skipped") - expect(neverWritten.headline).toMatch(/^No model call wrote to the prompt cache/) - - // Long enough and reporting no write bucket (OpenAI): a miss is a miss. - const missed = byId( - session(() => ({ usageInputTokens: 5_000, usageCacheReadInputTokens: 0 })), - "prompt-cache", - ) - expect(missed.status).toBe("warning") - expect(missed.headline).toBe("Cache hit rate 0% over 4 calls; 4 missed the cache") + expect(neverWritten.headline).toMatch(/^No Anthropic model call wrote to the prompt cache/) + + // Long enough on OpenAI: a miss is a miss, whether or not the emitter + // reported a write bucket (most stamp a zero one on every call). + for (const write of [undefined, 0]) { + const missed = byId( + session(() => ({ + providerName: "openai", + usageInputTokens: 2_000, + usageCacheReadInputTokens: 0, + usageCacheCreationInputTokens: write, + })), + "prompt-cache", + ) + expect(missed.status).toBe("warning") + expect(missed.headline).toBe("Cache hit rate 0% over 4 calls; 4 missed the cache") + } // A short opening call (a title, a router) is not the one that wrote the // cache: the first long call is, and it is not a miss. diff --git a/packages/agent-sessions/src/session-checks.ts b/packages/agent-sessions/src/session-checks.ts index 5a38c7c6c7..a414c8b23b 100644 --- a/packages/agent-sessions/src/session-checks.ts +++ b/packages/agent-sessions/src/session-checks.ts @@ -633,18 +633,13 @@ function promptCacheCheck(llmCalls: readonly AiSessionSpan[]): SessionCheck { "No model call reported cache usage, so the prompt cache could not be checked.", ) } - if (reporting.length <= CACHE_MIN_CALLS) { - return check( - identity, - "skipped", - `Only ${plural(reporting.length, "model call")} reported cache usage; at least ${CACHE_MIN_CALLS + 1} are needed to judge the prompt cache.`, - ) - } - // A provider that reports cache writes (Anthropic) writes on the first call - // whose prompt is long enough for the model; none ever writing means no - // prompt reached that model's minimum, or caching was never requested. + // Anthropic writes the cache on the first call whose prompt reaches the + // model's minimum, so calls that never wrote or read it had nothing + // cacheable, or never asked. Other providers stamp a zero write bucket on + // every call whatever happened, so only Anthropic's zero says this. const neverCached = reporting.every( (span) => + span.genAi.providerName === "anthropic" && span.genAi.usageCacheCreationInputTokens === 0 && (span.genAi.usageCacheReadInputTokens ?? 0) === 0, ) @@ -652,7 +647,7 @@ function promptCacheCheck(llmCalls: readonly AiSessionSpan[]): SessionCheck { return check( identity, "skipped", - "No model call wrote to the prompt cache: the prompts are below the model's cacheable minimum, or caching is off.", + "No Anthropic model call wrote to the prompt cache: the prompts are below the model's cacheable minimum, or caching is off.", ) } // The first cacheable call cannot hit a cache nothing has written yet. From d9c9fa77a3451c862ab2ec7a96bc8d8365ecb9ab Mon Sep 17 00:00:00 2001 From: JeremyFunk Date: Tue, 29 Sep 2026 00:44:26 +0200 Subject: [PATCH 12/19] fix(agent-sessions): say why nested inference time is charged once, not that it is one call Follow-up to the agentTime fix (#39): an inference span inside another is only the same model call under direct nesting; the rule holds because the nested span's time is inside the outer span's. Comment only. --- packages/agent-sessions/src/session-summary.ts | 8 ++++---- 1 file changed, 4 insertions(+), 4 deletions(-) diff --git a/packages/agent-sessions/src/session-summary.ts b/packages/agent-sessions/src/session-summary.ts index 7352f5c9d8..336ddb73da 100644 --- a/packages/agent-sessions/src/session-summary.ts +++ b/packages/agent-sessions/src/session-summary.ts @@ -357,10 +357,10 @@ export function findIdleGaps(spans: readonly AiSessionSpan[]): readonly IdleGap[ * time is mostly first-token latency is a different session from one that is * mostly generation. Agent and non-AI spans contribute nothing — an agent span * covers its children, and adding it would count the same work twice. For the - * same reason an inference span inside another is the same model call observed - * twice (OpenRouter's `LLM Generation` over its `generation`, ADK's `call_llm` - * over `generate_content`): the outermost carries the time, and lends its TTFT - * from the level that reported one. + * same reason an inference span inside another is charged nothing: its time is + * already inside the outer span's (OpenRouter's `LLM Generation` over its + * `generation`, ADK's `call_llm` over `generate_content`). The outermost + * carries the time, and borrows a TTFT from a nested span when it reported none. */ export function computeAgentTime(spans: readonly AiSessionSpan[]): SessionAgentTime { const totals = new Map() From 19f316b238c7ffea5c69757b706cb9fec4f1ccf5 Mon Sep 17 00:00:00 2001 From: JeremyFunk Date: Tue, 29 Sep 2026 00:45:53 +0200 Subject: [PATCH 13/19] fix(agent-sessions): match finish reasons without case or separators Symptom: Strands TS replies cut off at the output limit were never flagged; the Reply length check passed. Cause: Strands TS writes its finish reasons camelCase (`maxTokens`), and the truncation and refusal matchers (session-findings.ts TRUNCATION_FINISH_REASONS, session-summary.ts REFUSAL_FINISH_REASONS) only lowercased before comparing against `max_tokens`/`content_filter`. Fix: a shared finishReasonsIn helper matches on the reason lowercased with separators removed, so `maxTokens`, `MAX_TOKENS` and `max_tokens` are one reason. --- .../agent-sessions/src/session-checks.test.ts | 20 +++++++++++++++++++ .../agent-sessions/src/session-findings.ts | 11 +++++----- .../agent-sessions/src/session-summary.ts | 14 +++++++++++-- 3 files changed, 37 insertions(+), 8 deletions(-) diff --git a/packages/agent-sessions/src/session-checks.test.ts b/packages/agent-sessions/src/session-checks.test.ts index 79e6320fc4..f8e1f26888 100644 --- a/packages/agent-sessions/src/session-checks.test.ts +++ b/packages/agent-sessions/src/session-checks.test.ts @@ -507,6 +507,26 @@ describe("buildSessionChecks", () => { expect(byId(report, "provider").headline).toBe("All 2 model calls were answered first time") }) + // Strands TS writes its finish reasons camelCase (`maxTokens`), which the + // lowercased match against `max_tokens` never caught. + it("reads a cut-off reply whatever the finish reason's spelling", () => { + for (const reason of ["maxTokens", "MAX_TOKENS", "max_tokens"]) { + const report = checks([ + agentSpan({ spanId: "a1", startMs: 0, durationMs: 10 * SECOND }), + llmSpan({ + spanId: "l1", + parentSpanId: "a1", + startMs: SECOND, + durationMs: SECOND, + genAi: { requestMaxTokens: 600, responseFinishReasons: [reason] }, + }), + ]) + const replyLength = byId(report, "reply-length") + expect(replyLength.status).toBe("warning") + expect(replyLength.headline).toMatch(/^1 reply hit the output token limit \(max_tokens 600\)/) + } + }) + // Google ADK reports each call twice: `call_llm` (no operation, a model) // over `generate_content`, both with the same usage. The cache check read // "over 15 calls" for a session of 8. diff --git a/packages/agent-sessions/src/session-findings.ts b/packages/agent-sessions/src/session-findings.ts index 3478c6c0d5..8c2d44a056 100644 --- a/packages/agent-sessions/src/session-findings.ts +++ b/packages/agent-sessions/src/session-findings.ts @@ -13,6 +13,7 @@ import { canonicalJSON } from "@maple/query-engine" import { clipDetail, failureDetailText } from "./failure-text" import { failureEvents, + finishReasonsIn, findIdleGaps, isProviderAttempt, shadowedAncestorIds, @@ -46,8 +47,9 @@ const IDENTICAL_RUN_MIN_CALLS = 3 */ const MID_TURN_STALL_MIN_MS = 30_000 -/** Finish reasons that mean the reply was cut off at the output token limit. */ -const TRUNCATION_FINISH_REASONS = new Set(["length", "max_tokens", "max_output_tokens"]) +/** Finish reasons that mean the reply was cut off at the output token limit, + * as `finishReasonsIn` keys. */ +const TRUNCATION_FINISH_REASONS = new Set(["length", "maxtokens", "maxoutputtokens"]) /** * Failure kinds that need a fix whether or not the run survived them: the @@ -351,10 +353,7 @@ function truncationFindings( } function truncationSignal(span: AiSessionSpan): string | undefined { - const reasons = (span.genAi.responseFinishReasons ?? []) - .map((reason) => reason.toLowerCase()) - .filter((reason) => TRUNCATION_FINISH_REASONS.has(reason)) - return reasons.length === 0 ? undefined : reasons.join(",") + return finishReasonsIn(span, TRUNCATION_FINISH_REASONS) } function repetitionFindings(turns: readonly SessionTurn[]): SessionFinding[] { diff --git a/packages/agent-sessions/src/session-summary.ts b/packages/agent-sessions/src/session-summary.ts index 336ddb73da..f45497a23b 100644 --- a/packages/agent-sessions/src/session-summary.ts +++ b/packages/agent-sessions/src/session-summary.ts @@ -241,7 +241,7 @@ export interface SessionSummary { const RATE_LIMIT_PATTERN = /\b429\b|rate.?limit|too.many.requests|resource.exhausted|overloaded/i const CONTEXT_EXCEEDED_PATTERN = /context.{0,16}(length|window|limit)|maximum.context|prompt is too long|too many tokens/i -const REFUSAL_FINISH_REASONS = new Set(["refusal", "content_filter"]) +const REFUSAL_FINISH_REASONS = new Set(["refusal", "contentfilter"]) export function buildSessionSummary({ spans, @@ -881,9 +881,19 @@ function failureSignal(span: AiSessionSpan): string | undefined { } function refusalSignal(span: AiSessionSpan): string | undefined { + return finishReasonsIn(span, REFUSAL_FINISH_REASONS) +} + +/** + * The span's finish reasons that are one of `keys`, lowercased and joined — or + * `undefined`. Matched without case or separators, since vendors spell one + * reason `max_tokens`, `MAX_TOKENS` and `maxTokens` (Strands TS); `keys` are + * written that way too. + */ +export function finishReasonsIn(span: AiSessionSpan, keys: ReadonlySet): string | undefined { const reasons = (span.genAi.responseFinishReasons ?? []) .map((reason) => reason.toLowerCase()) - .filter((reason) => REFUSAL_FINISH_REASONS.has(reason)) + .filter((reason) => keys.has(reason.replace(/[^a-z0-9]/g, ""))) return reasons.length === 0 ? undefined : reasons.join(",") } From efd5c0615e228fd608da0d6ed8b225cf5fba073b Mon Sep 17 00:00:00 2001 From: JeremyFunk Date: Tue, 29 Sep 2026 00:47:36 +0200 Subject: [PATCH 14/19] fix(agent-sessions): fold a workless anchor only when it leads straight into the next turn Follow-up to the empty-turn fix (#41), from review. A workless agent invocation followed by a pause was folded into the next turn, which then read the pause as a stall inside it. The fold now needs the next turn to start within 5 s of the workless anchor ending (setup such as MAF's `workflow.build`), and merging moves the accumulated bucket forward in place instead of re-copying it for each workless anchor in a row. --- .../agent-sessions/src/session-turns.test.ts | 15 ++++++++++ packages/agent-sessions/src/session-turns.ts | 28 +++++++++++++------ 2 files changed, 35 insertions(+), 8 deletions(-) diff --git a/packages/agent-sessions/src/session-turns.test.ts b/packages/agent-sessions/src/session-turns.test.ts index 4ad3645b96..685e9e2bdb 100644 --- a/packages/agent-sessions/src/session-turns.test.ts +++ b/packages/agent-sessions/src/session-turns.test.ts @@ -226,6 +226,21 @@ describe("buildSessionTurns", () => { expect(buildSessionSummary({ spans, turns }).title).toBe("Produce a mini briefing about Amsterdam") }) + // A pause after it says the invocation stood on its own: folding it would + // read the pause as a stall inside the next turn. + it("keeps a workless anchor followed by a pause as its own turn", () => { + const turns = buildSessionTurns([ + agentSpan({ spanId: "quiet", startMs: 0, durationMs: SECOND }), + agentSpan({ spanId: "busy", startMs: 60 * SECOND, durationMs: 10 * SECOND }), + llmSpan({ spanId: "chat", parentSpanId: "busy", startMs: 61 * SECOND, durationMs: SECOND }), + ]) + + expect(turns.map((turn) => turn.spans.map((span) => span.spanId))).toEqual([ + ["quiet"], + ["busy", "chat"], + ]) + }) + it("falls back to root agent invocations when no conversation id exists", () => { const turns = buildSessionTurns([ agentSpan({ spanId: "agent-1", startMs: 0, durationMs: 10 * SECOND }), diff --git a/packages/agent-sessions/src/session-turns.ts b/packages/agent-sessions/src/session-turns.ts index e5b57637d3..4b85cc9fc3 100644 --- a/packages/agent-sessions/src/session-turns.ts +++ b/packages/agent-sessions/src/session-turns.ts @@ -164,6 +164,9 @@ export interface SessionTurn { } const WORK_CATEGORIES: ReadonlySet = new Set(["inference", "tool"]) +/** How soon before the next turn a workless anchor must end to be its setup: + * the pause the session summary starts calling idle. */ +const SETUP_LEAD_MAX_MS = 5_000 interface TurnAnchor { readonly span: AiSessionSpan @@ -221,12 +224,13 @@ export function buildSessionTurns(spans: readonly AiSessionSpan[]): readonly Ses } // On rules 2 and 3 the boundary is a guess, and an anchor that opened no - // work — no model or tool call, no prompt, nothing failed — is not a turn: - // Microsoft Agent Framework's one-span `workflow.build` trace ahead of its - // `workflow.run` became an empty turn 1 that also took the session's title. - // Its spans join the next turn, as spans before the first anchor join turn 1; - // the cursor filled the buckets in start order, so they stay in it. A session - // with no work anywhere keeps its anchors. + // work — no model or tool call, no prompt, nothing failed — right before the + // next turn is that turn's setup: Microsoft Agent Framework's one-span + // `workflow.build` trace ahead of its `workflow.run` became an empty turn 1 + // that also took the session's title. Its spans join the next turn, as spans + // before the first anchor join turn 1; the cursor filled the buckets in start + // order, so they stay in it. One followed by a pause is left alone, and a + // session with no work anywhere keeps its anchors. const opened = (bucket: readonly AiSessionSpan[]) => bucket.some( (span) => @@ -236,8 +240,16 @@ export function buildSessionTurns(spans: readonly AiSessionSpan[]): readonly Ses ) if (anchors[0]?.kind !== "conversation" && buckets.some(opened)) { for (let i = 0; i < buckets.length - 1; i++) { - if (opened(buckets[i])) continue - buckets[i + 1] = [...buckets[i], ...buckets[i + 1]] + const bucket = buckets[i] + const next = buckets[i + 1] + if (opened(bucket)) continue + const endMs = bucket.reduce( + (max, span) => Math.max(max, spanEndMs(span)), + Number.NEGATIVE_INFINITY, + ) + if (next[0] !== undefined && spanStartMs(next[0]) - endMs > SETUP_LEAD_MAX_MS) continue + for (const span of next) bucket.push(span) + buckets[i + 1] = bucket buckets[i] = [] } } From 429379ce40c5d2b30c0393c4ab84b5521101bbee Mon Sep 17 00:00:00 2001 From: JeremyFunk Date: Tue, 29 Sep 2026 00:48:02 +0200 Subject: [PATCH 15/19] fix(agent-sessions): charge a model call made under a tool or agent in agentTime Follow-up to the agentTime fix (#39), from review. The walk to the outermost inference span crossed agent and tool spans, so a model call a tool or sub-agent made inside another call was charged nothing. The walk now stops at an agent or tool: only inference spans nested directly (or through non-work spans) share the outer span's time. --- .../agent-sessions/src/session-summary.test.ts | 10 ++++++++++ packages/agent-sessions/src/session-summary.ts | 14 +++++++++----- 2 files changed, 19 insertions(+), 5 deletions(-) diff --git a/packages/agent-sessions/src/session-summary.test.ts b/packages/agent-sessions/src/session-summary.test.ts index 344ef42ae8..a188b861ac 100644 --- a/packages/agent-sessions/src/session-summary.test.ts +++ b/packages/agent-sessions/src/session-summary.test.ts @@ -136,6 +136,16 @@ describe("buildSessionSummary — time", () => { expect(segment(summary.agentTime.segments, "ttft")).toBe(3_600) expect(segment(summary.agentTime.segments, "inference")).toBe(1_242) }) + + it("still charges a model call a tool made inside another call", () => { + const summary = summarize([ + llmSpan({ spanId: "outer", startMs: 0, durationMs: 10 * SECOND }), + toolSpan({ spanId: "tool", parentSpanId: "outer", startMs: SECOND, durationMs: 4 * SECOND }), + llmSpan({ spanId: "inner", parentSpanId: "tool", startMs: 2 * SECOND, durationMs: 2 * SECOND }), + ]) + + expect(segment(summary.agentTime.segments, "inference")).toBe(12 * SECOND) + }) }) describe("buildSessionSummary — failed", () => { diff --git a/packages/agent-sessions/src/session-summary.ts b/packages/agent-sessions/src/session-summary.ts index f45497a23b..9c41affc31 100644 --- a/packages/agent-sessions/src/session-summary.ts +++ b/packages/agent-sessions/src/session-summary.ts @@ -357,10 +357,11 @@ export function findIdleGaps(spans: readonly AiSessionSpan[]): readonly IdleGap[ * time is mostly first-token latency is a different session from one that is * mostly generation. Agent and non-AI spans contribute nothing — an agent span * covers its children, and adding it would count the same work twice. For the - * same reason an inference span inside another is charged nothing: its time is - * already inside the outer span's (OpenRouter's `LLM Generation` over its - * `generation`, ADK's `call_llm` over `generate_content`). The outermost - * carries the time, and borrows a TTFT from a nested span when it reported none. + * same reason an inference span inside another, with no agent or tool between + * them, is charged nothing: its time is already inside the outer span's + * (OpenRouter's `LLM Generation` over its `generation`, ADK's `call_llm` over + * `generate_content`). The outermost carries the time, and borrows a TTFT from + * a nested span when it reported none. */ export function computeAgentTime(spans: readonly AiSessionSpan[]): SessionAgentTime { const totals = new Map() @@ -374,7 +375,10 @@ export function computeAgentTime(spans: readonly AiSessionSpan[]): SessionAgentT let parent = byId.get(span.parentSpanId) while (parent !== undefined && !seen.has(parent.spanId)) { seen.add(parent.spanId) - if (classifyAiSpan(parent) === "inference") call = parent + const category = classifyAiSpan(parent) + // An agent or a tool in between ran calls of its own. + if (category === "agent" || category === "tool") break + if (category === "inference") call = parent parent = byId.get(parent.parentSpanId) } return call From 3b9a6e5c33990f058a015aa01102942c9c1bae5a Mon Sep 17 00:00:00 2001 From: JeremyFunk Date: Tue, 29 Sep 2026 00:49:16 +0200 Subject: [PATCH 16/19] fix(agent-sessions): judge a netted call's cache off the observation that reported it Follow-up to the per-call cache fix (#40), from review. When an app span and its gateway's mirror share a response id, the netted call list can keep the app span, which may carry no cache fields, and the cache check then skipped a session the mirror measured. For such a call the check now reads the same response's observation that reported cache usage. --- .../agent-sessions/src/session-checks.test.ts | 37 +++++++++++++++++++ packages/agent-sessions/src/session-checks.ts | 32 ++++++++++++---- 2 files changed, 61 insertions(+), 8 deletions(-) diff --git a/packages/agent-sessions/src/session-checks.test.ts b/packages/agent-sessions/src/session-checks.test.ts index f8e1f26888..c11efbc2c7 100644 --- a/packages/agent-sessions/src/session-checks.test.ts +++ b/packages/agent-sessions/src/session-checks.test.ts @@ -527,6 +527,43 @@ describe("buildSessionChecks", () => { } }) + // The app's own span reports no cache fields; its gateway's mirror, in a + // trace of its own with the same response id, does. The call is counted + // once, and judged by the observation that measured the cache. + it("judges the prompt cache off a gateway mirror when the counted span has none", () => { + const report = checks( + [0, 1, 2, 3, 4].flatMap((i) => [ + llmSpan({ + spanId: `app-${i}`, + traceId: `trace-app-${i}`, + startMs: i * MINUTE, + durationMs: 2 * SECOND, + genAi: { + providerName: "openai", + usageInputTokens: 2_000, + usageOutputTokens: 50, + responseId: `gen-${i}`, + }, + }), + llmSpan({ + spanId: `mirror-${i}`, + traceId: `trace-gateway-${i}`, + spanName: "LLM Generation", + startMs: i * MINUTE + 10, + durationMs: 2 * SECOND, + genAi: { + providerName: "openai", + usageInputTokens: 2_000, + usageCacheReadInputTokens: i === 0 ? 0 : 1_500, + usageOutputTokens: 50, + responseId: `gen-${i}`, + }, + }), + ]), + ) + expect(byId(report, "prompt-cache").headline).toBe("Cache hit rate 75% over 4 calls") + }) + // Google ADK reports each call twice: `call_llm` (no operation, a model) // over `generate_content`, both with the same usage. The cache check read // "over 15 calls" for a session of 8. diff --git a/packages/agent-sessions/src/session-checks.ts b/packages/agent-sessions/src/session-checks.ts index a414c8b23b..77f612eed2 100644 --- a/packages/agent-sessions/src/session-checks.ts +++ b/packages/agent-sessions/src/session-checks.ts @@ -153,9 +153,7 @@ export function buildSessionChecks( ...otherErrorsCheck(errors.filter((finding) => finding.tool === undefined)), repetitionCheck(of("repetition"), summary, coverage), stallCheck(of("stall")), - // Per call, so a model call its framework also rolled up (ADK `call_llm` over - // `generate_content`) is judged once. - promptCacheCheck(sessionLlmCalls(spans)), + promptCacheCheck(cacheObservations(spans)), ].sort( (a, b) => STATUS_RANK[a.status] - STATUS_RANK[b.status] || @@ -619,13 +617,31 @@ function stallCheck(findings: readonly SessionFinding[]): SessionCheck { ) } +const reportsCache = (span: AiSessionSpan): boolean => + span.genAi.usageCacheReadInputTokens !== undefined || + span.genAi.usageCacheCreationInputTokens !== undefined + +/** + * One span per model call, so a call its framework also rolled up (ADK + * `call_llm` over `generate_content`) is judged once — and, when the span + * counted for it reported no cache usage, another observation of the same + * response that did (an app span beside its gateway's mirror). + */ +function cacheObservations(spans: readonly AiSessionSpan[]): readonly AiSessionSpan[] { + const byResponse = new Map() + for (const span of spans) { + const responseId = span.genAi.responseId + if (responseId !== undefined && responseId !== "" && reportsCache(span)) + byResponse.set(responseId, span) + } + return sessionLlmCalls(spans).map((call) => + reportsCache(call) ? call : (byResponse.get(call.genAi.responseId ?? "") ?? call), + ) +} + function promptCacheCheck(llmCalls: readonly AiSessionSpan[]): SessionCheck { const identity: CheckIdentity = { id: "prompt-cache", name: "Prompt cache", fixArea: "prompt" } - const reporting = llmCalls.filter( - (span) => - span.genAi.usageCacheReadInputTokens !== undefined || - span.genAi.usageCacheCreationInputTokens !== undefined, - ) + const reporting = llmCalls.filter(reportsCache) if (reporting.length === 0) { return check( identity, From 258b7355aa61a4922f8d61df79770b6dc5dd119c Mon Sep 17 00:00:00 2001 From: JeremyFunk Date: Tue, 29 Sep 2026 01:05:23 +0200 Subject: [PATCH 17/19] fix(agent-sessions): leave out each uncached Claude call from the prompt-cache check Follow-up to the uncacheable-prompt fix (#25). The never-written rule was session-wide and keyed on provider `anthropic` with a write bucket of 0, so it missed Claude through OpenRouter (flue: provider `openrouter`, prompts 1,043-1,392 tokens under Haiku 4.5's 4,096 minimum) and OpenRouter Broadcast (provider `anthropic`, write key absent, prompts 1.2K-2.6K), and a mixed session fell back to judging every call. A call is now left out on its own when it is Claude (provider `anthropic` or a `claude` model) and neither wrote nor read the cache, like the 1024-token cut; the rest of the session is still judged. --- .../agent-sessions/src/session-checks.test.ts | 55 +++++++++++++++---- packages/agent-sessions/src/session-checks.ts | 32 +++++------ 2 files changed, 59 insertions(+), 28 deletions(-) diff --git a/packages/agent-sessions/src/session-checks.test.ts b/packages/agent-sessions/src/session-checks.test.ts index c11efbc2c7..c3811e353e 100644 --- a/packages/agent-sessions/src/session-checks.test.ts +++ b/packages/agent-sessions/src/session-checks.test.ts @@ -365,7 +365,7 @@ describe("buildSessionChecks", () => { parentSpanId: "a1", startMs, durationMs: SECOND, - model: "claude-opus-5", + model: "gpt-5", genAi: { conversationId: "t1", // Inclusive of the cache read, as the default convention counts it. @@ -422,20 +422,55 @@ describe("buildSessionChecks", () => { ) expect(short.status).toBe("skipped") expect(short.headline).toBe( - "Only 0 model calls had a prompt of 1024 tokens or more, the smallest a provider caches; at least 4 are needed to judge the prompt cache.", + "Only 0 model calls had a prompt long enough to cache; at least 4 are needed to judge the prompt cache.", ) - const neverWritten = byId( - session((i) => ({ - providerName: "anthropic", - usageInputTokens: 1_227 + 200 * i, - usageCacheReadInputTokens: 0, + // Claude calls that neither wrote nor read the cache were below the + // model's minimum (Haiku 4.5: 4,096): Claude Agent SDK, flue through + // OpenRouter (write stamped 0), OpenRouter Broadcast (write key absent). + for (const claude of [ + { providerName: "anthropic", requestModel: "claude-haiku-4-5", usageCacheCreationInputTokens: 0 }, + { + providerName: "openrouter", + requestModel: "anthropic/claude-haiku-4.5", usageCacheCreationInputTokens: 0, - })), + }, + { providerName: "anthropic", requestModel: "claude-sonnet-4.5" }, + ]) { + const neverWritten = byId( + session((i) => ({ + ...claude, + usageInputTokens: 1_227 + 200 * i, + usageCacheReadInputTokens: 0, + })), + "prompt-cache", + ) + expect(neverWritten.status).toBe("skipped") + } + + // Per call: a session mixing uncached Claude calls with OpenAI misses is + // judged on the OpenAI calls. + const mixed = byId( + session((i) => + i % 2 === 0 + ? { + providerName: "openai", + requestModel: "gpt-4o-mini", + usageInputTokens: 2_000, + usageCacheReadInputTokens: 0, + } + : { + providerName: "openrouter", + requestModel: "anthropic/claude-haiku-4.5", + usageInputTokens: 1_300, + usageCacheReadInputTokens: 0, + usageCacheCreationInputTokens: 0, + }, + ), "prompt-cache", ) - expect(neverWritten.status).toBe("skipped") - expect(neverWritten.headline).toMatch(/^No Anthropic model call wrote to the prompt cache/) + expect(mixed.status).toBe("skipped") + expect(mixed.headline).toMatch(/^Only 3 model calls had a prompt long enough to cache/) // Long enough on OpenAI: a miss is a miss, whether or not the emitter // reported a write bucket (most stamp a zero one on every call). diff --git a/packages/agent-sessions/src/session-checks.ts b/packages/agent-sessions/src/session-checks.ts index 77f612eed2..029ad9e989 100644 --- a/packages/agent-sessions/src/session-checks.ts +++ b/packages/agent-sessions/src/session-checks.ts @@ -30,6 +30,7 @@ import { import { classifyAiSpan, isLlmCall, + spanModel, spanStartMs, type SessionTurn, type TurnAnchorKind, @@ -617,6 +618,17 @@ function stallCheck(findings: readonly SessionFinding[]): SessionCheck { ) } +/** + * A Claude call that neither wrote nor read the cache: its prompt was below the + * model's minimum (up to 4,096 tokens), or it never asked. Anthropic writes as + * soon as a prompt qualifies, so this says nothing about the prefix. Only + * Claude's zero means this: other providers stamp a zero write on every call. + */ +const uncachedClaudeCall = (span: AiSessionSpan): boolean => + (span.genAi.providerName === "anthropic" || /claude/i.test(spanModel(span) ?? "")) && + (span.genAi.usageCacheCreationInputTokens ?? 0) === 0 && + (span.genAi.usageCacheReadInputTokens ?? 0) === 0 + const reportsCache = (span: AiSessionSpan): boolean => span.genAi.usageCacheReadInputTokens !== undefined || span.genAi.usageCacheCreationInputTokens !== undefined @@ -649,25 +661,9 @@ function promptCacheCheck(llmCalls: readonly AiSessionSpan[]): SessionCheck { "No model call reported cache usage, so the prompt cache could not be checked.", ) } - // Anthropic writes the cache on the first call whose prompt reaches the - // model's minimum, so calls that never wrote or read it had nothing - // cacheable, or never asked. Other providers stamp a zero write bucket on - // every call whatever happened, so only Anthropic's zero says this. - const neverCached = reporting.every( - (span) => - span.genAi.providerName === "anthropic" && - span.genAi.usageCacheCreationInputTokens === 0 && - (span.genAi.usageCacheReadInputTokens ?? 0) === 0, - ) - if (neverCached) { - return check( - identity, - "skipped", - "No Anthropic model call wrote to the prompt cache: the prompts are below the model's cacheable minimum, or caching is off.", - ) - } // The first cacheable call cannot hit a cache nothing has written yet. const cacheable = reporting + .filter((span) => !uncachedClaudeCall(span)) .map(spanTokenBuckets) .filter((buckets) => buckets !== undefined) .map((buckets) => ({ @@ -680,7 +676,7 @@ function promptCacheCheck(llmCalls: readonly AiSessionSpan[]): SessionCheck { return check( identity, "skipped", - `Only ${plural(cacheable.length, "model call")} had a prompt of ${CACHE_MIN_PROMPT_TOKENS} tokens or more, the smallest a provider caches; at least ${CACHE_MIN_CALLS + 1} are needed to judge the prompt cache.`, + `Only ${plural(cacheable.length, "model call")} had a prompt long enough to cache; at least ${CACHE_MIN_CALLS + 1} are needed to judge the prompt cache.`, ) } const rate = From 80bff228a40b5e3e23d0e61eeab1cc883c65c767 Mon Sep 17 00:00:00 2001 From: JeremyFunk Date: Tue, 29 Sep 2026 01:05:46 +0200 Subject: [PATCH 18/19] fix(agent-sessions): cover Vercel AI SDK's content-filter refusal in the finish-reason match Follow-up to the finish-reason fix. The case it fixes today is Vercel AI SDK's kebab-case ai.response.finishReason: a `content-filter` refusal read as passed and now warns. Strands TS camelCase reasons live in output-message parts and reach the checks only once those are read. --- .../agent-sessions/src/session-checks.test.ts | 17 +++++++++++++++++ 1 file changed, 17 insertions(+) diff --git a/packages/agent-sessions/src/session-checks.test.ts b/packages/agent-sessions/src/session-checks.test.ts index c3811e353e..148989cfaf 100644 --- a/packages/agent-sessions/src/session-checks.test.ts +++ b/packages/agent-sessions/src/session-checks.test.ts @@ -562,6 +562,23 @@ describe("buildSessionChecks", () => { } }) + // Vercel AI SDK's `ai.response.finishReason` is kebab-case: a + // `content-filter` refusal read as passed. + it("reads a filtered reply whatever the finish reason's spelling", () => { + const report = checks([ + agentSpan({ spanId: "a1", startMs: 0, durationMs: 10 * SECOND }), + llmSpan({ + spanId: "l1", + parentSpanId: "a1", + startMs: SECOND, + durationMs: SECOND, + genAi: { responseFinishReasons: ["content-filter"] }, + }), + ]) + expect(byId(report, "refusals").status).not.toBe("passed") + expect(byId(report, "refusals").headline).toMatch(/^1 reply was refused or filtered/) + }) + // The app's own span reports no cache fields; its gateway's mirror, in a // trace of its own with the same response id, does. The call is counted // once, and judged by the observation that measured the cache. From 840e50e17bdca45332f5b1f512031ad34c472d9b Mon Sep 17 00:00:00 2001 From: JeremyFunk Date: Tue, 29 Sep 2026 01:12:18 +0200 Subject: [PATCH 19/19] fix(agent-sessions): judge an uncached Claude call once its prompt passes every Claude minimum Follow-up to the uncacheable-prompt fix (#25). Leaving out every Claude call that neither wrote nor read the cache also hid real misses when the emitter does not report writes: 8 Claude calls with 10K-token prompts and 4 misses read "passed 90% over 3 calls" instead of "51% over 7". Such a call is now left out only when its prompt is also under 4,096 tokens, the largest Claude minimum; above it, reading nothing is a miss. The cold/warm cache fixture is back on a Claude model with no write bucket and 10K prompts, and must warn. --- .../agent-sessions/src/session-checks.test.ts | 4 ++- packages/agent-sessions/src/session-checks.ts | 35 +++++++++---------- 2 files changed, 19 insertions(+), 20 deletions(-) diff --git a/packages/agent-sessions/src/session-checks.test.ts b/packages/agent-sessions/src/session-checks.test.ts index 148989cfaf..b04f6097cd 100644 --- a/packages/agent-sessions/src/session-checks.test.ts +++ b/packages/agent-sessions/src/session-checks.test.ts @@ -365,10 +365,12 @@ describe("buildSessionChecks", () => { parentSpanId: "a1", startMs, durationMs: SECOND, - model: "gpt-5", + model: "claude-opus-5", genAi: { conversationId: "t1", // Inclusive of the cache read, as the default convention counts it. + // No write reported: a Claude prompt this long that read nothing + // missed the cache, rather than being too short to cache. usageInputTokens: 10_000, usageCacheReadInputTokens: cacheRead, usageOutputTokens: 100, diff --git a/packages/agent-sessions/src/session-checks.ts b/packages/agent-sessions/src/session-checks.ts index 029ad9e989..c7b08805d1 100644 --- a/packages/agent-sessions/src/session-checks.ts +++ b/packages/agent-sessions/src/session-checks.ts @@ -44,6 +44,11 @@ const CACHE_MIN_CALLS = 3 /** The smallest prompt OpenAI, Anthropic or Gemini will cache: a shorter one * cannot hit the cache, so missing it says nothing about the prefix. */ const CACHE_MIN_PROMPT_TOKENS = 1024 +/** The largest Claude minimum (Haiku 4.5, Opus 4.5). Anthropic writes as soon + * as a prompt qualifies, so a Claude call that neither wrote nor read the + * cache under it may just have been too short; one above it missed. Only + * Claude's zero says this: other providers stamp a zero write on every call. */ +const CLAUDE_CACHE_MIN_PROMPT_TOKENS = 4096 /** A headline names this many findings before it counts the rest. */ const HEADLINE_MAX_CLAUSES = 3 @@ -618,16 +623,9 @@ function stallCheck(findings: readonly SessionFinding[]): SessionCheck { ) } -/** - * A Claude call that neither wrote nor read the cache: its prompt was below the - * model's minimum (up to 4,096 tokens), or it never asked. Anthropic writes as - * soon as a prompt qualifies, so this says nothing about the prefix. Only - * Claude's zero means this: other providers stamp a zero write on every call. - */ -const uncachedClaudeCall = (span: AiSessionSpan): boolean => - (span.genAi.providerName === "anthropic" || /claude/i.test(spanModel(span) ?? "")) && - (span.genAi.usageCacheCreationInputTokens ?? 0) === 0 && - (span.genAi.usageCacheReadInputTokens ?? 0) === 0 +/** Claude, by provider or through any gateway by model. */ +const isClaude = (span: AiSessionSpan): boolean => + span.genAi.providerName === "anthropic" || /claude/i.test(spanModel(span) ?? "") const reportsCache = (span: AiSessionSpan): boolean => span.genAi.usageCacheReadInputTokens !== undefined || @@ -662,15 +660,14 @@ function promptCacheCheck(llmCalls: readonly AiSessionSpan[]): SessionCheck { ) } // The first cacheable call cannot hit a cache nothing has written yet. - const cacheable = reporting - .filter((span) => !uncachedClaudeCall(span)) - .map(spanTokenBuckets) - .filter((buckets) => buckets !== undefined) - .map((buckets) => ({ - read: buckets.cacheRead, - prompt: buckets.input + buckets.cacheRead + buckets.cacheWrite, - })) - .filter((call) => call.prompt >= CACHE_MIN_PROMPT_TOKENS) + const cacheable = reporting.flatMap((span) => { + const buckets = spanTokenBuckets(span) + if (buckets === undefined) return [] + const prompt = buckets.input + buckets.cacheRead + buckets.cacheWrite + const uncachedClaude = isClaude(span) && buckets.cacheRead + buckets.cacheWrite === 0 + const minimum = uncachedClaude ? CLAUDE_CACHE_MIN_PROMPT_TOKENS : CACHE_MIN_PROMPT_TOKENS + return prompt >= minimum ? [{ read: buckets.cacheRead, prompt }] : [] + }) const calls = cacheable.slice(1) if (calls.length < CACHE_MIN_CALLS) { return check(