feat(coding-agent): add prompt cache miss tracking (#6427)

Detect prompt cache misses per turn by comparing each assistant message's
cache reads against the previous request's prompt tokens (core/cache-stats.ts).
Significant misses emit a warning-colored transcript notice at the turn they
occur, noting idle gaps past the cache TTL and model switches when relevant.

/session gained cache statistics: a compact token/cache breakdown with hit
rate, a $-prefixed cost section with per-model cost breakdown, and the
cumulative cost re-billed due to cache misses.
This commit is contained in:
Armin Ronacher
2026-07-09 14:12:57 +02:00
committed by GitHub
parent 57d96d72ed
commit 3f9aa5d10b
11 changed files with 471 additions and 36 deletions
@@ -109,7 +109,8 @@ describe("AgentSession.getSessionStats", () => {
syncAgentMessages(session, sessionManager);
const stats = session.getSessionStats();
expect(stats.tokens.input).toBe(195_000);
// Totals cover ALL entries, including history compacted away (180k + 195k).
expect(stats.tokens.input).toBe(375_000);
expect(stats.contextUsage).toBeDefined();
expect(stats.contextUsage?.tokens).toBeNull();
expect(stats.contextUsage?.percent).toBeNull();
@@ -132,7 +133,8 @@ describe("AgentSession.getSessionStats", () => {
syncAgentMessages(session, sessionManager);
const stats = session.getSessionStats();
expect(stats.tokens.input).toBe(220_000);
// Totals cover ALL entries, including history compacted away (180k + 195k + 25k).
expect(stats.tokens.input).toBe(400_000);
expect(stats.contextUsage).toBeDefined();
expect(stats.contextUsage?.tokens).toBe(25_000);
expect(stats.contextUsage?.percent).toBe((25_000 / model.contextWindow) * 100);