Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
Show all changes
20 commits
Select commit Hold shift + click to select a range
919f6cf
fix(agent-sessions): stop the prompt-cache check warning on uncacheab…
JeremyFunk Sep 28, 2026
e8d125d
fix(agent-sessions): count model calls once in the provider-errors he…
JeremyFunk Sep 28, 2026
b845e08
fix(agent-sessions): judge the prompt cache once per model call
JeremyFunk Sep 28, 2026
0815f47
fix(agent-sessions): count a nested model call's time once in agentTime
JeremyFunk Sep 28, 2026
008a9d5
fix(agent-sessions): label a turn from its model calls before framewo…
JeremyFunk Sep 28, 2026
2b3faa0
fix(agent-sessions): read past a "New task:" lead-in in turn labels
JeremyFunk Sep 28, 2026
ef98e2c
fix(agent-sessions): fold a turn that opened no work into the next one
JeremyFunk Sep 28, 2026
94d0a27
fix(agent-sessions): stop at the second line when reading a turn label
JeremyFunk Sep 28, 2026
202f7eb
fix(agent-sessions): leave out the first cacheable call, not the firs…
JeremyFunk Sep 28, 2026
9f76b7c
fix(agent-sessions): match only smolagents' "New task:" lead-in in tu…
JeremyFunk Sep 28, 2026
7e5ca58
fix(agent-sessions): read a never-written prompt cache as uncacheable…
JeremyFunk Sep 28, 2026
d9c9fa7
fix(agent-sessions): say why nested inference time is charged once, n…
JeremyFunk Sep 28, 2026
19f316b
fix(agent-sessions): match finish reasons without case or separators
JeremyFunk Sep 28, 2026
efd5c06
fix(agent-sessions): fold a workless anchor only when it leads straig…
JeremyFunk Sep 28, 2026
429379c
fix(agent-sessions): charge a model call made under a tool or agent i…
JeremyFunk Sep 28, 2026
3b9a6e5
fix(agent-sessions): judge a netted call's cache off the observation …
JeremyFunk Sep 28, 2026
38bbca1
Merge remote-tracking branch 'origin/main' into fix/agent-sessions-ch…
JeremyFunk Sep 28, 2026
258b735
fix(agent-sessions): leave out each uncached Claude call from the pro…
JeremyFunk Sep 28, 2026
80bff22
fix(agent-sessions): cover Vercel AI SDK's content-filter refusal in …
JeremyFunk Sep 28, 2026
840e50e
fix(agent-sessions): judge an uncached Claude call once its prompt pa…
JeremyFunk Sep 28, 2026
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
257 changes: 256 additions & 1 deletion packages/agent-sessions/src/session-checks.test.ts
Original file line number Diff line number Diff line change
@@ -1,9 +1,10 @@
import type { AiSessionGenAiValues } from "@maple/domain/http"
import { describe, expect, it } from "vitest"

import { buildSessionChecks, type SessionCheck } from "./session-checks"
import { buildSessionSummary } from "./session-summary"
import { buildSessionTurns } from "./session-turns"
import { agentSpan, llmSpan, toolSpan } from "./span-test-support"
import { agentSpan, llmSpan, makeSpan, toolSpan } from "./span-test-support"

const SECOND = 1000
const MINUTE = 60 * SECOND
Expand Down Expand Up @@ -368,6 +369,8 @@ describe("buildSessionChecks", () => {
genAi: {
conversationId: "t1",
// Inclusive of the cache read, as the default convention counts it.
// No write reported: a Claude prompt this long that read nothing
// missed the cache, rather than being too short to cache.
usageInputTokens: 10_000,
usageCacheReadInputTokens: cacheRead,
usageOutputTokens: 100,
Expand Down Expand Up @@ -397,6 +400,258 @@ describe("buildSessionChecks", () => {
expect(byId(checks(firstTurn()), "prompt-cache").status).toBe("skipped")
})

// Short test conversations (DSPy peaked at 870 tokens; Claude Agent SDK on
// Haiku 4.5 sent 1.2K–2.3K against a 4,096-token minimum) cannot be cached,
// so reading them as misses warned "0% ... missed the cache" on every one.
it("skips the prompt cache when no prompt could have been cached", () => {
const call = (spanId: string, startMs: number, genAi: AiSessionGenAiValues) =>
llmSpan({
spanId,
parentSpanId: "a1",
startMs,
durationMs: SECOND,
genAi: { conversationId: "t1", usageOutputTokens: 50, ...genAi },
})
const session = (genAi: (i: number) => AiSessionGenAiValues) =>
checks([
agentSpan({ spanId: "a1", startMs: 0, durationMs: MINUTE, genAi: { conversationId: "t1" } }),
...[0, 1, 2, 3, 4].map((i) => call(`c${i}`, (i + 1) * SECOND, genAi(i))),
])

const short = byId(
session((i) => ({ usageInputTokens: 700 + 40 * i, usageCacheReadInputTokens: 0 })),
"prompt-cache",
)
expect(short.status).toBe("skipped")
expect(short.headline).toBe(
"Only 0 model calls had a prompt long enough to cache; at least 4 are needed to judge the prompt cache.",
)

// Claude calls that neither wrote nor read the cache were below the
// model's minimum (Haiku 4.5: 4,096): Claude Agent SDK, flue through
// OpenRouter (write stamped 0), OpenRouter Broadcast (write key absent).
for (const claude of [
{ providerName: "anthropic", requestModel: "claude-haiku-4-5", usageCacheCreationInputTokens: 0 },
{
providerName: "openrouter",
requestModel: "anthropic/claude-haiku-4.5",
usageCacheCreationInputTokens: 0,
},
{ providerName: "anthropic", requestModel: "claude-sonnet-4.5" },
]) {
const neverWritten = byId(
session((i) => ({
...claude,
usageInputTokens: 1_227 + 200 * i,
usageCacheReadInputTokens: 0,
})),
"prompt-cache",
)
expect(neverWritten.status).toBe("skipped")
}

// Per call: a session mixing uncached Claude calls with OpenAI misses is
// judged on the OpenAI calls.
const mixed = byId(
session((i) =>
i % 2 === 0
? {
providerName: "openai",
requestModel: "gpt-4o-mini",
usageInputTokens: 2_000,
usageCacheReadInputTokens: 0,
}
: {
providerName: "openrouter",
requestModel: "anthropic/claude-haiku-4.5",
usageInputTokens: 1_300,
usageCacheReadInputTokens: 0,
usageCacheCreationInputTokens: 0,
},
),
"prompt-cache",
)
expect(mixed.status).toBe("skipped")
expect(mixed.headline).toMatch(/^Only 3 model calls had a prompt long enough to cache/)

// Long enough on OpenAI: a miss is a miss, whether or not the emitter
// reported a write bucket (most stamp a zero one on every call).
for (const write of [undefined, 0]) {
const missed = byId(
session(() => ({
providerName: "openai",
usageInputTokens: 2_000,
usageCacheReadInputTokens: 0,
usageCacheCreationInputTokens: write,
})),
"prompt-cache",
)
expect(missed.status).toBe("warning")
expect(missed.headline).toBe("Cache hit rate 0% over 4 calls; 4 missed the cache")
}

// A short opening call (a title, a router) is not the one that wrote the
// cache: the first long call is, and it is not a miss.
const shortFirst = byId(
session((i) =>
i === 0
? { usageInputTokens: 300, usageCacheReadInputTokens: 0 }
: { usageInputTokens: 2_000, usageCacheReadInputTokens: i === 1 ? 0 : 1_000 },
),
"prompt-cache",
)
expect(shortFirst.headline).toBe("Cache hit rate 50% over 3 calls")
})

// OpenRouter Broadcast: every request is a `chat` root carrying the usage
// plus a `provider attempt N` and a `generation` child, all op `chat`. The
// headline read "All 48 model calls" for a session of 16.
it("counts model calls in the provider headline once per call, not per chat span", () => {
const request = (n: number) => {
const traceId = `trace-${n}`
const root = `gen-${n}`
const at = n * MINUTE
return [
llmSpan({
spanId: root,
traceId,
spanName: "LLM Generation",
startMs: at,
durationMs: 4_842,
genAi: { usageInputTokens: 58_797, usageOutputTokens: 220, responseId: `gen-${n}` },
}),
llmSpan({
spanId: `${root}-attempt`,
parentSpanId: root,
traceId,
spanName: "provider attempt 1: OpenAI",
startMs: at + 184,
durationMs: 2_019,
genAi: { responseId: `gen-${n}:attempt-0` },
}),
llmSpan({
spanId: `${root}-generation`,
parentSpanId: root,
traceId,
spanName: "generation",
startMs: at + 200,
durationMs: 4_402,
genAi: { responseId: `gen-${n}:generation` },
}),
]
}
const report = checks([...request(0), ...request(1)])
expect(byId(report, "provider").headline).toBe("All 2 model calls were answered first time")
})

// Strands TS writes its finish reasons camelCase (`maxTokens`), which the
// lowercased match against `max_tokens` never caught.
it("reads a cut-off reply whatever the finish reason's spelling", () => {
for (const reason of ["maxTokens", "MAX_TOKENS", "max_tokens"]) {
const report = checks([
agentSpan({ spanId: "a1", startMs: 0, durationMs: 10 * SECOND }),
llmSpan({
spanId: "l1",
parentSpanId: "a1",
startMs: SECOND,
durationMs: SECOND,
genAi: { requestMaxTokens: 600, responseFinishReasons: [reason] },
}),
])
const replyLength = byId(report, "reply-length")
expect(replyLength.status).toBe("warning")
expect(replyLength.headline).toMatch(/^1 reply hit the output token limit \(max_tokens 600\)/)
}
})

// Vercel AI SDK's `ai.response.finishReason` is kebab-case: a
// `content-filter` refusal read as passed.
it("reads a filtered reply whatever the finish reason's spelling", () => {
const report = checks([
agentSpan({ spanId: "a1", startMs: 0, durationMs: 10 * SECOND }),
llmSpan({
spanId: "l1",
parentSpanId: "a1",
startMs: SECOND,
durationMs: SECOND,
genAi: { responseFinishReasons: ["content-filter"] },
}),
])
expect(byId(report, "refusals").status).not.toBe("passed")
expect(byId(report, "refusals").headline).toMatch(/^1 reply was refused or filtered/)
})

// The app's own span reports no cache fields; its gateway's mirror, in a
// trace of its own with the same response id, does. The call is counted
// once, and judged by the observation that measured the cache.
it("judges the prompt cache off a gateway mirror when the counted span has none", () => {
const report = checks(
[0, 1, 2, 3, 4].flatMap((i) => [
llmSpan({
spanId: `app-${i}`,
traceId: `trace-app-${i}`,
startMs: i * MINUTE,
durationMs: 2 * SECOND,
genAi: {
providerName: "openai",
usageInputTokens: 2_000,
usageOutputTokens: 50,
responseId: `gen-${i}`,
},
}),
llmSpan({
spanId: `mirror-${i}`,
traceId: `trace-gateway-${i}`,
spanName: "LLM Generation",
startMs: i * MINUTE + 10,
durationMs: 2 * SECOND,
genAi: {
providerName: "openai",
usageInputTokens: 2_000,
usageCacheReadInputTokens: i === 0 ? 0 : 1_500,
usageOutputTokens: 50,
responseId: `gen-${i}`,
},
}),
]),
)
expect(byId(report, "prompt-cache").headline).toBe("Cache hit rate 75% over 4 calls")
})

// Google ADK reports each call twice: `call_llm` (no operation, a model)
// over `generate_content`, both with the same usage. The cache check read
// "over 15 calls" for a session of 8.
it("judges the prompt cache once per model call when a wrapper repeats the usage", () => {
const usage = (i: number): AiSessionGenAiValues => ({
requestModel: "openrouter/openai/gpt-4o-mini",
usageInputTokens: 2_000,
usageCacheReadInputTokens: i === 0 ? 0 : 1_536,
usageOutputTokens: 32,
})
const report = checks([
agentSpan({ spanId: "a1", startMs: 0, durationMs: MINUTE, agentName: "assistant" }),
...[0, 1, 2, 3, 4].flatMap((i) => [
makeSpan({
spanId: `call-${i}`,
parentSpanId: "a1",
spanName: "call_llm",
startMs: (i + 1) * SECOND,
durationMs: 2 * SECOND,
genAi: usage(i),
}),
llmSpan({
spanId: `gen-${i}`,
parentSpanId: `call-${i}`,
spanName: "generate_content",
startMs: (i + 1) * SECOND,
durationMs: 2 * SECOND,
genAi: { operationName: "generate_content", ...usage(i) },
}),
]),
])
expect(byId(report, "prompt-cache").headline).toBe("Cache hit rate 77% over 4 calls")
})

// The one rule behind red and amber, pinned per kind: a class that needs a
// fix stays red when survived; everything else survived is amber.
it("keeps a survived context overflow red and a survived provider error amber", () => {
Expand Down
Loading
Loading