From f8b74a4507a0505cc8275a8f5afcdd9072cc75f0 Mon Sep 17 00:00:00 2001 From: Mario Zechner Date: Mon, 20 Jul 2026 17:04:12 +0200 Subject: [PATCH] fix: complete extension usage accounting closes #6509 --- packages/agent/CHANGELOG.md | 4 ++ .../src/harness/compaction/compaction.ts | 28 ++++++++++- packages/agent/src/index.ts | 1 + .../agent/test/harness/compaction.test.ts | 11 ++++- packages/ai/CHANGELOG.md | 1 + packages/coding-agent/CHANGELOG.md | 1 + packages/coding-agent/README.md | 2 +- packages/coding-agent/docs/compaction.md | 11 +++-- packages/coding-agent/docs/extensions.md | 18 ++++++-- packages/coding-agent/docs/rpc.md | 30 +++++++++++- packages/coding-agent/docs/session-format.md | 3 ++ packages/coding-agent/docs/usage.md | 2 +- .../examples/extensions/custom-compaction.ts | 1 + .../src/core/compaction/compaction.ts | 35 +++++++++++++- .../coding-agent/src/core/usage-totals.ts | 43 +++++++++++++++++ packages/coding-agent/src/index.ts | 1 + .../src/modes/interactive/interactive-mode.ts | 24 +++------- .../test/agent-session-stats.test.ts | 26 +++++++++++ .../test/compaction-summary-reasoning.test.ts | 15 +++++- .../session-manager/tree-traversal.test.ts | 46 +++++++++++++++++++ 20 files changed, 267 insertions(+), 36 deletions(-) diff --git a/packages/agent/CHANGELOG.md b/packages/agent/CHANGELOG.md index eba2a62a..3e8ec724 100644 --- a/packages/agent/CHANGELOG.md +++ b/packages/agent/CHANGELOG.md @@ -6,6 +6,10 @@ - Moved the `uuidv7` export to `@earendil-works/pi-ai`. +### Added + +- Added usage metadata to tool results, compaction entries, and branch summaries in the agent harness ([#6671](https://github.com/earendil-works/pi/pull/6671) by [@davidbrai](https://github.com/davidbrai)). + ## [0.80.10] - 2026-07-16 ## [0.80.9] - 2026-07-16 diff --git a/packages/agent/src/harness/compaction/compaction.ts b/packages/agent/src/harness/compaction/compaction.ts index 58938c13..6c21a492 100644 --- a/packages/agent/src/harness/compaction/compaction.ts +++ b/packages/agent/src/harness/compaction/compaction.ts @@ -499,6 +499,30 @@ export async function generateSummary( customInstructions?: string, previousSummary?: string, thinkingLevel?: ThinkingLevel, +): Promise> { + const result = await generateSummaryWithUsage( + currentMessages, + models, + model, + reserveTokens, + signal, + customInstructions, + previousSummary, + thinkingLevel, + ); + return result.ok ? ok(result.value.text) : err(result.error); +} + +/** Generate or update a conversation summary and return its provider usage. */ +export async function generateSummaryWithUsage( + currentMessages: AgentMessage[], + models: Models, + model: Model, + reserveTokens: number, + signal?: AbortSignal, + customInstructions?: string, + previousSummary?: string, + thinkingLevel?: ThinkingLevel, ): Promise> { const maxTokens = Math.min( Math.floor(0.8 * reserveTokens), @@ -687,7 +711,7 @@ export async function compact( let historyText = "No prior history."; let historyUsage: Usage | undefined; if (messagesToSummarize.length > 0) { - const historyResult = await generateSummary( + const historyResult = await generateSummaryWithUsage( messagesToSummarize, models, model, @@ -715,7 +739,7 @@ export async function compact( ? combineUsage(historyUsage, turnPrefixResult.value.usage) : turnPrefixResult.value.usage; } else { - const summaryResult = await generateSummary( + const summaryResult = await generateSummaryWithUsage( messagesToSummarize, models, model, diff --git a/packages/agent/src/index.ts b/packages/agent/src/index.ts index 49baf081..b5ec7807 100644 --- a/packages/agent/src/index.ts +++ b/packages/agent/src/index.ts @@ -20,6 +20,7 @@ export { findCutPoint, findTurnStartIndex, generateSummary, + generateSummaryWithUsage, getLastAssistantUsage, prepareCompaction, serializeConversation, diff --git a/packages/agent/test/harness/compaction.test.ts b/packages/agent/test/harness/compaction.test.ts index 7cdea0c9..c8160307 100644 --- a/packages/agent/test/harness/compaction.test.ts +++ b/packages/agent/test/harness/compaction.test.ts @@ -20,6 +20,7 @@ import { findCutPoint, findTurnStartIndex, generateSummary, + generateSummaryWithUsage, getLastAssistantUsage, prepareCompaction, serializeConversation, @@ -510,7 +511,7 @@ describe("harness compaction", () => { ]); const summary = getOrThrow( - await generateSummary(messages, models, model, 2000, undefined, "focus", "old summary"), + await generateSummaryWithUsage(messages, models, model, 2000, undefined, "focus", "old summary"), ); expect(summary.text).toContain("Test summary"); @@ -523,6 +524,14 @@ describe("harness compaction", () => { expect(promptText).toContain("Additional focus: focus"); }); + it("preserves the string result from generateSummary", async () => { + const messages: AgentMessage[] = [createUserMessage("Summarize this.")]; + const { faux, model } = createFauxModel(false); + faux.setResponses([fauxAssistantMessage("## Goal\nTest summary")]); + + expect(getOrThrow(await generateSummary(messages, models, model, 2000))).toBe("## Goal\nTest summary"); + }); + it("returns error results for failed or aborted summary generations", async () => { const messages: AgentMessage[] = [createUserMessage("Summarize this.")]; const { faux: errorFaux, model: errorModel } = createFauxModel(false); diff --git a/packages/ai/CHANGELOG.md b/packages/ai/CHANGELOG.md index 7739597b..56693a28 100644 --- a/packages/ai/CHANGELOG.md +++ b/packages/ai/CHANGELOG.md @@ -6,6 +6,7 @@ - Added `contentText` for extracting joined text from message content. - Added a shared `uuidv7` utility for time-ordered identifiers. +- Added optional usage metadata to tool result messages ([#6671](https://github.com/earendil-works/pi/pull/6671) by [@davidbrai](https://github.com/davidbrai)). ### Fixed diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index 15bdbf80..cac4d019 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -6,6 +6,7 @@ - Added built-in llama.cpp router support with `/login` connection setup and `/llama` Hugging Face model search and downloads, explicit loading, unloading, and live progress. See [llama.cpp](docs/llama-cpp.md). - Added extension registration for complete pi-ai providers, including native authentication, model refresh, filtering, and streaming behavior. +- Added usage accounting for tools, compaction, and branch summaries in persisted sessions, footer totals, and session statistics ([#6671](https://github.com/earendil-works/pi/pull/6671) by [@davidbrai](https://github.com/davidbrai)). ### Fixed diff --git a/packages/coding-agent/README.md b/packages/coding-agent/README.md index 3454fac2..2198d724 100644 --- a/packages/coding-agent/README.md +++ b/packages/coding-agent/README.md @@ -152,7 +152,7 @@ The interface from top to bottom: - **Startup header** - Shows shortcuts (`/hotkeys` for all), loaded AGENTS.md files, prompt templates, skills, and extensions - **Messages** - Your messages, assistant responses, tool calls and results, notifications, errors, and extension UI - **Editor** - Where you type; border color indicates thinking level -- **Footer** - Working directory, session name, total token/cache usage (`↑` input, `↓` output, `R` cache read, `W` cache write, `CH` latest cache hit rate), cost, context usage, current model +- **Footer** - Working directory, session name, total token/cache usage (`↑` input, `↓` output, `R` cache read, `W` cache write, `CH` latest cache hit rate), cost, context usage, current model. Totals include assistant responses, usage reported by tools, and summary generation. The editor can be temporarily replaced by other UI, like built-in `/settings` or custom UI from extensions (e.g., a Q&A tool that lets the user answer model questions in a structured format). [Extensions](#extensions) can also replace the editor, add widgets above/below it, a status line, custom footer, or overlays. diff --git a/packages/coding-agent/docs/compaction.md b/packages/coding-agent/docs/compaction.md index 5e0d4eef..d55c2136 100644 --- a/packages/coding-agent/docs/compaction.md +++ b/packages/coding-agent/docs/compaction.md @@ -129,6 +129,7 @@ interface CompactionEntry { summary: string; firstKeptEntryId: string; tokensBefore: number; + usage?: Usage; // LLM usage that generated the summary fromHook?: boolean; // true if provided by extension (legacy field name) details?: T; // implementation-specific data } @@ -140,9 +141,9 @@ interface CompactionDetails { } ``` -Extensions can store any JSON-serializable data in `details`. The default compaction tracks file operations, but custom extension implementations can use their own structure. +Extensions can store any JSON-serializable data in `details`. The default compaction tracks file operations, but custom extension implementations can use their own structure. Generated and extension-provided summaries store their LLM `usage` when available so session totals include summarization work. -See [`prepareCompaction()`](https://github.com/earendil-works/pi-mono/blob/main/packages/coding-agent/src/core/compaction/compaction.ts) and [`compact()`](https://github.com/earendil-works/pi-mono/blob/main/packages/coding-agent/src/core/compaction/compaction.ts) for the implementation. +See [`prepareCompaction()`](https://github.com/earendil-works/pi-mono/blob/main/packages/coding-agent/src/core/compaction/compaction.ts) and [`compact()`](https://github.com/earendil-works/pi-mono/blob/main/packages/coding-agent/src/core/compaction/compaction.ts) for the implementation. For direct programmatic summarization, `generateSummary()` returns the summary text and `generateSummaryWithUsage()` returns `{ text, usage }`. ## Branch Summarization @@ -195,6 +196,7 @@ interface BranchSummaryEntry { timestamp: number; summary: string; fromId: string; // Entry we navigated from + usage?: Usage; // LLM usage that generated the summary fromHook?: boolean; // true if provided by extension (legacy field name) details?: T; // implementation-specific data } @@ -300,6 +302,7 @@ pi.on("session_before_compact", async (event, ctx) => { summary: "Your summary...", firstKeptEntryId: preparation.firstKeptEntryId, tokensBefore: preparation.tokensBefore, + // usage: summaryResponse.usage, // Optional; included in session totals details: { /* custom data */ }, } }; @@ -328,13 +331,14 @@ pi.on("session_before_compact", async (event, ctx) => { // [Tool result]: output text // Now send to your model for summarization - const summary = await myModel.summarize(conversationText); + const { summary, usage } = await myModel.summarize(conversationText); return { compaction: { summary, firstKeptEntryId: preparation.firstKeptEntryId, tokensBefore: preparation.tokensBefore, + usage, } }; }); @@ -364,6 +368,7 @@ pi.on("session_before_tree", async (event, ctx) => { return { summary: { summary: "Your summary...", + // usage: summaryResponse.usage, // Optional; included in session totals details: { /* custom data */ }, } }; diff --git a/packages/coding-agent/docs/extensions.md b/packages/coding-agent/docs/extensions.md index e51b056e..6aec1196 100644 --- a/packages/coding-agent/docs/extensions.md +++ b/packages/coding-agent/docs/extensions.md @@ -468,6 +468,7 @@ pi.on("session_before_compact", async (event, ctx) => { summary: "...", firstKeptEntryId: preparation.firstKeptEntryId, tokensBefore: preparation.tokensBefore, + // usage: summaryResponse.usage, // Optional; included in session totals } }; }); @@ -489,7 +490,13 @@ pi.on("session_before_tree", async (event, ctx) => { const { preparation, signal } = event; return { cancel: true }; // OR provide custom summary: - return { summary: { summary: "...", details: {} } }; + return { + summary: { + summary: "...", + // usage: summaryResponse.usage, // Optional; included in session totals + details: {}, + }, + }; }); pi.on("session_tree", async (event, ctx) => { @@ -813,7 +820,7 @@ In parallel tool mode, `tool_result` and `tool_execution_end` may interleave in `tool_result` handlers chain like middleware: - Handlers run in extension load order - Each handler sees the latest result after previous handler changes -- Handlers can return partial patches (`content`, `details`, or `isError`); omitted fields keep their current values +- Handlers can return partial patches (`content`, `details`, `isError`, or `usage`); omitted fields keep their current values Use `ctx.signal` for nested async work inside the handler. This lets Esc cancel model calls, `fetch()`, and other abort-aware operations started by the extension. @@ -822,7 +829,7 @@ import { isBashToolResult } from "@earendil-works/pi-coding-agent"; pi.on("tool_result", async (event, ctx) => { // event.toolName, event.toolCallId, event.input - // event.content, event.details, event.isError + // event.content, event.details, event.isError, event.usage if (isBashToolResult(event)) { // event.details is typed as BashToolDetails @@ -835,7 +842,7 @@ pi.on("tool_result", async (event, ctx) => { }); // Modify result: - return { content: [...], details: {...}, isError: false }; + return { content: [...], details: {...}, isError: false, usage: nestedModelUsage }; }); ``` @@ -1932,6 +1939,7 @@ pi.registerTool({ return { content: [{ type: "text", text: "Done" }], // Sent to LLM details: { data: result }, // For rendering & state + // usage: nestedModelResponse.usage, // Optional nested LLM usage // Optional: stop after this tool batch when every finalized tool result // in the batch also returns terminate: true. terminate: true, @@ -1944,6 +1952,8 @@ pi.registerTool({ }); ``` +**Usage accounting:** If a tool makes nested LLM calls, return their combined `Usage` as `usage`. Pi persists it on the tool result and includes it in footer, `/session`, and RPC session totals. `tool_result` handlers can inspect or replace this value. + **Signaling errors:** To mark a tool execution as failed (sets `isError: true` on the result and reports it to the LLM), throw an error from `execute`. Returning a value never sets the error flag regardless of what properties you include in the return object. **Early termination:** Return `terminate: true` from `execute()` to hint that the automatic follow-up LLM call should be skipped after the current tool batch. This only takes effect when every finalized tool result in that batch is terminating. See [examples/extensions/structured-output.ts](../examples/extensions/structured-output.ts) for a minimal example where the agent ends on a final structured-output tool call. diff --git a/packages/coding-agent/docs/rpc.md b/packages/coding-agent/docs/rpc.md index 8074ceb9..2151b4b3 100644 --- a/packages/coding-agent/docs/rpc.md +++ b/packages/coding-agent/docs/rpc.md @@ -395,12 +395,20 @@ Response: "firstKeptEntryId": "abc123", "tokensBefore": 150000, "estimatedTokensAfter": 32000, + "usage": { + "input": 32000, + "output": 1200, + "cacheRead": 0, + "cacheWrite": 0, + "totalTokens": 33200, + "cost": {"input": 0.01, "output": 0.02, "cacheRead": 0, "cacheWrite": 0, "total": 0.03} + }, "details": {} } } ``` -`estimatedTokensAfter` is a heuristic estimate over the rebuilt message context immediately after compaction, not a provider-exact token count. +`estimatedTokensAfter` is a heuristic estimate over the rebuilt message context immediately after compaction, not a provider-exact token count. `usage` reports the LLM call or calls that generated the summary and may be omitted by custom compaction handlers. #### set_auto_compaction @@ -557,7 +565,7 @@ Response: } ``` -`tokens` contains assistant usage totals for the current session state. `contextUsage` contains the actual current context-window estimate used for compaction and footer display. +`tokens` and `cost` include assistant messages, usage reported by tools, and compaction/branch-summary generation across the full session. `contextUsage` contains the actual current context-window estimate used for compaction and footer display. `contextUsage` is omitted when no model or context window is available. `contextUsage.tokens` and `contextUsage.percent` are `null` immediately after compaction until a fresh post-compaction assistant response provides valid usage data. @@ -1016,6 +1024,14 @@ The `reason` field is `"manual"`, `"threshold"`, or `"overflow"`. "firstKeptEntryId": "abc123", "tokensBefore": 150000, "estimatedTokensAfter": 32000, + "usage": { + "input": 32000, + "output": 1200, + "cacheRead": 0, + "cacheWrite": 0, + "totalTokens": 33200, + "cost": {"input": 0.01, "output": 0.02, "cacheRead": 0, "cacheWrite": 0, "total": 0.03} + }, "details": {} }, "aborted": false, @@ -1368,11 +1384,21 @@ Stop reasons: `"stop"`, `"length"`, `"toolUse"`, `"error"`, `"aborted"` "toolCallId": "call_123", "toolName": "bash", "content": [{"type": "text", "text": "total 48\ndrwxr-xr-x ..."}], + "usage": { + "input": 100, + "output": 50, + "cacheRead": 0, + "cacheWrite": 0, + "totalTokens": 150, + "cost": {"input": 0.0003, "output": 0.00075, "cacheRead": 0, "cacheWrite": 0, "total": 0.00105} + }, "isError": false, "timestamp": 1733234567890 } ``` +`usage` is optional and reports nested LLM work performed by the tool. When present, it contributes to session token and cost totals. + ### BashExecutionMessage Created by the `bash` RPC command (not by LLM tool calls): diff --git a/packages/coding-agent/docs/session-format.md b/packages/coding-agent/docs/session-format.md index a4afb318..b27b8b70 100644 --- a/packages/coding-agent/docs/session-format.md +++ b/packages/coding-agent/docs/session-format.md @@ -96,6 +96,7 @@ interface ToolResultMessage { toolName: string; content: (TextContent | ImageContent)[]; details?: any; // Tool-specific metadata + usage?: Usage; // Nested LLM work performed by the tool isError: boolean; timestamp: number; } @@ -232,6 +233,7 @@ Created when context is compacted. Stores a summary of earlier messages. ``` Optional fields: +- `usage`: LLM usage from generating the summary; included in session token and cost totals - `details`: Implementation-specific data (e.g., `{ readFiles: string[], modifiedFiles: string[] }` for default, or custom data for extensions) - `fromHook`: `true` if generated by an extension, `false`/`undefined` if pi-generated (legacy field name) @@ -244,6 +246,7 @@ Created when switching branches via `/tree` with an LLM generated summary of the ``` Optional fields: +- `usage`: LLM usage from generating the summary; included in session token and cost totals - `details`: File tracking data (`{ readFiles: string[], modifiedFiles: string[] }`) for default, or custom data for extensions - `fromHook`: `true` if generated by an extension, `false`/`undefined` if pi-generated (legacy field name) diff --git a/packages/coding-agent/docs/usage.md b/packages/coding-agent/docs/usage.md index 48cf16b1..c9df34ef 100644 --- a/packages/coding-agent/docs/usage.md +++ b/packages/coding-agent/docs/usage.md @@ -11,7 +11,7 @@ The interface has four main areas: - **Startup header** - shortcuts, loaded context files, prompt templates, skills, and extensions - **Messages** - user messages, assistant responses, tool calls, tool results, notifications, errors, and extension UI - **Editor** - where you type; border color indicates the current thinking level -- **Footer** - working directory, session name, token/cache usage, cost, context usage, and current model +- **Footer** - working directory, session name, token/cache usage, cost, context usage, and current model. Totals include assistant responses, usage reported by tools, and summary generation. The editor can be replaced temporarily by built-in UI such as `/settings` or by custom extension UI. diff --git a/packages/coding-agent/examples/extensions/custom-compaction.ts b/packages/coding-agent/examples/extensions/custom-compaction.ts index fdc2128c..e37ce976 100644 --- a/packages/coding-agent/examples/extensions/custom-compaction.ts +++ b/packages/coding-agent/examples/extensions/custom-compaction.ts @@ -116,6 +116,7 @@ ${conversationText} summary, firstKeptEntryId, tokensBefore, + usage: response.usage, }, }; } catch (error) { diff --git a/packages/coding-agent/src/core/compaction/compaction.ts b/packages/coding-agent/src/core/compaction/compaction.ts index dafcb09e..ffad75e8 100644 --- a/packages/coding-agent/src/core/compaction/compaction.ts +++ b/packages/coding-agent/src/core/compaction/compaction.ts @@ -581,6 +581,37 @@ export async function generateSummary( thinkingLevel?: ThinkingLevel, streamFn?: StreamFn, env?: Record, +): Promise { + return ( + await generateSummaryWithUsage( + currentMessages, + model, + reserveTokens, + apiKey, + headers, + signal, + customInstructions, + previousSummary, + thinkingLevel, + streamFn, + env, + ) + ).text; +} + +/** Generate or update a conversation summary and return its provider usage. */ +export async function generateSummaryWithUsage( + currentMessages: AgentMessage[], + model: Model, + reserveTokens: number, + apiKey: string | undefined, + headers?: Record, + signal?: AbortSignal, + customInstructions?: string, + previousSummary?: string, + thinkingLevel?: ThinkingLevel, + streamFn?: StreamFn, + env?: Record, ): Promise<{ text: string; usage: Usage }> { const maxTokens = Math.min( Math.floor(0.8 * reserveTokens), @@ -790,7 +821,7 @@ export async function compact( let historyText = "No prior history."; let historyUsage: Usage | undefined; if (messagesToSummarize.length > 0) { - const historyResult = await generateSummary( + const historyResult = await generateSummaryWithUsage( messagesToSummarize, model, settings.reserveTokens, @@ -822,7 +853,7 @@ export async function compact( summaryUsage = historyUsage ? combineUsage(historyUsage, turnPrefixResult.usage) : turnPrefixResult.usage; } else { // Just generate history summary - const result = await generateSummary( + const result = await generateSummaryWithUsage( messagesToSummarize, model, settings.reserveTokens, diff --git a/packages/coding-agent/src/core/usage-totals.ts b/packages/coding-agent/src/core/usage-totals.ts index e65e2eb4..4ae55d0d 100644 --- a/packages/coding-agent/src/core/usage-totals.ts +++ b/packages/coding-agent/src/core/usage-totals.ts @@ -1,4 +1,5 @@ import type { Usage } from "@earendil-works/pi-ai/compat"; +import type { SessionEntry } from "./session-manager.ts"; export interface UsageTotals { input: number; @@ -25,3 +26,45 @@ export function addUsageToTotals(totals: UsageTotals, usage: Usage): void { totals.cacheWrite += usage.cacheWrite; totals.cost += usage.cost.total; } + +export interface UsageCostBreakdownEntry { + key: string; + cost: number; + tokens: number; +} + +/** Group attributable assistant usage by model and all other usage into a separate bucket. */ +export function getUsageCostBreakdown(entries: SessionEntry[]): UsageCostBreakdownEntry[] { + const totalsByKey = new Map(); + + for (const entry of entries) { + let key: string | undefined; + let usage: Usage | undefined; + if (entry.type === "message" && entry.message.role === "assistant") { + key = `${entry.message.provider}/${entry.message.responseModel ?? entry.message.model}`; + usage = entry.message.usage; + } else if (entry.type === "message" && entry.message.role === "toolResult" && entry.message.usage) { + key = "Tools/summaries"; + usage = entry.message.usage; + } else if ((entry.type === "branch_summary" || entry.type === "compaction") && entry.usage) { + key = "Tools/summaries"; + usage = entry.usage; + } + if (!key || !usage) continue; + + let totals = totalsByKey.get(key); + if (!totals) { + totals = createUsageTotals(); + totalsByKey.set(key, totals); + } + addUsageToTotals(totals, usage); + } + + return Array.from(totalsByKey, ([key, totals]) => ({ + key, + cost: totals.cost, + tokens: totals.input + totals.output + totals.cacheRead + totals.cacheWrite, + })) + .filter((entry) => entry.cost > 0 || entry.tokens > 0) + .sort((a, b) => b.cost - a.cost); +} diff --git a/packages/coding-agent/src/index.ts b/packages/coding-agent/src/index.ts index 196fc8ca..e223d314 100644 --- a/packages/coding-agent/src/index.ts +++ b/packages/coding-agent/src/index.ts @@ -42,6 +42,7 @@ export { type GenerateBranchSummaryOptions, generateBranchSummary, generateSummary, + generateSummaryWithUsage, getLastAssistantUsage, prepareBranchEntries, serializeConversation, diff --git a/packages/coding-agent/src/modes/interactive/interactive-mode.ts b/packages/coding-agent/src/modes/interactive/interactive-mode.ts index b7dc08fc..54415407 100644 --- a/packages/coding-agent/src/modes/interactive/interactive-mode.ts +++ b/packages/coding-agent/src/modes/interactive/interactive-mode.ts @@ -86,6 +86,7 @@ import type { SourceInfo } from "../../core/source-info.ts"; import { isInstallTelemetryEnabled } from "../../core/telemetry.ts"; import type { TruncationResult } from "../../core/tools/truncate.ts"; import { hasTrustRequiringProjectResources, ProjectTrustStore } from "../../core/trust-manager.ts"; +import { getUsageCostBreakdown } from "../../core/usage-totals.ts"; import { getChangelogPath, getNewEntries, normalizeChangelogLinks, parseChangelog } from "../../utils/changelog.ts"; import { copyToClipboard, readClipboardText } from "../../utils/clipboard.ts"; import { extensionForImageMimeType, readClipboardImage } from "../../utils/clipboard-image.ts"; @@ -5597,22 +5598,9 @@ export class InteractiveMode { const cacheWaste = computeCacheWaste(entries, this.session.modelRuntime); // Cost/token totals per provider/model actually used (e.g. OpenRouter `auto` - // resolves to a concrete responseModel), sorted by cost descending. - const perModelMap = new Map(); - for (const entry of entries) { - if (entry.type !== "message" || entry.message.role !== "assistant") continue; - const message = entry.message; - const usage = message.usage; - const key = `${message.provider}/${message.responseModel ?? message.model}`; - let bucket = perModelMap.get(key); - if (!bucket) { - bucket = { key, cost: 0, tokens: 0 }; - perModelMap.set(key, bucket); - } - bucket.cost += usage.cost.total; - bucket.tokens += usage.input + usage.output + usage.cacheRead + usage.cacheWrite; - } - const perModel = Array.from(perModelMap.values()).sort((a, b) => b.cost - a.cost); + // resolves to a concrete responseModel). Usage without model attribution is + // grouped separately so the breakdown reconciles with the session total. + const usageBreakdown = getUsageCostBreakdown(entries); let info = `${theme.bold("Session Info")}\n\n`; if (sessionName) { @@ -5646,8 +5634,8 @@ export class InteractiveMode { if (stats.cost > 0 || cacheWaste.missedTokens > 0) { info += `\n${theme.bold("Cost")}\n`; info += `${theme.fg("dim", "Total:")} $${stats.cost.toFixed(3)}`; - if (perModel.length > 1) { - for (const entry of perModel) { + if (usageBreakdown.length > 1) { + for (const entry of usageBreakdown) { info += `\n ${theme.fg("dim", `${entry.key}:`)} $${entry.cost.toFixed(3)} ${theme.fg("dim", `(${formatTokens(entry.tokens)} tokens)`)}`; } } diff --git a/packages/coding-agent/test/agent-session-stats.test.ts b/packages/coding-agent/test/agent-session-stats.test.ts index 684c5e1c..2e9d6bf4 100644 --- a/packages/coding-agent/test/agent-session-stats.test.ts +++ b/packages/coding-agent/test/agent-session-stats.test.ts @@ -5,6 +5,7 @@ import { AgentSession } from "../src/core/agent-session.ts"; import { AuthStorage } from "../src/core/auth-storage.ts"; import { SessionManager } from "../src/core/session-manager.ts"; import { SettingsManager } from "../src/core/settings-manager.ts"; +import { getUsageCostBreakdown } from "../src/core/usage-totals.ts"; import { createInMemoryModelRegistry, getModelRuntime } from "./model-runtime-test-utils.ts"; import { createTestResourceLoader } from "./utilities.ts"; @@ -224,6 +225,31 @@ describe("AgentSession.getSessionStats", () => { } }); + it("groups tool and summary usage separately from model-attributed usage", () => { + const sessionManager = SessionManager.inMemory(); + const rootId = sessionManager.appendMessage(createUserMessage("hello", 1)); + sessionManager.appendMessage({ + ...createAssistantMessage("response", 100, 2), + usage: { ...createUsage(100), cost: { ...createUsage(100).cost, total: 0.5 } }, + }); + sessionManager.appendMessage( + createToolResultMessage({ ...createUsage(100), cost: { ...createUsage(100).cost, total: 1 } }), + ); + sessionManager.appendCompaction("summary", rootId, 100, undefined, false, { + ...createUsage(100), + cost: { ...createUsage(100).cost, total: 2 }, + }); + sessionManager.branchWithSummary(null, "branch summary", undefined, false, { + ...createUsage(100), + cost: { ...createUsage(100).cost, total: 3 }, + }); + + expect(getUsageCostBreakdown(sessionManager.getEntries())).toEqual([ + { key: "Tools/summaries", cost: 6, tokens: 300 }, + { key: `${model.provider}/${model.id}`, cost: 0.5, tokens: 100 }, + ]); + }); + it("ignores zero-usage messages when checking for post-compaction context usage", async () => { const { session, sessionManager } = await createSession(); diff --git a/packages/coding-agent/test/compaction-summary-reasoning.test.ts b/packages/coding-agent/test/compaction-summary-reasoning.test.ts index eb1d3579..41766238 100644 --- a/packages/coding-agent/test/compaction-summary-reasoning.test.ts +++ b/packages/coding-agent/test/compaction-summary-reasoning.test.ts @@ -1,7 +1,12 @@ import type { AgentMessage } from "@earendil-works/pi-agent-core"; import type { AssistantMessage, Model } from "@earendil-works/pi-ai"; import { beforeEach, describe, expect, it, vi } from "vitest"; -import { type CompactionPreparation, compact, generateSummary } from "../src/core/compaction/index.ts"; +import { + type CompactionPreparation, + compact, + generateSummary, + generateSummaryWithUsage, +} from "../src/core/compaction/index.ts"; const { completeSimpleMock } = vi.hoisted(() => ({ completeSimpleMock: vi.fn(), @@ -57,7 +62,7 @@ describe("generateSummary reasoning options", () => { }); it("uses the provided thinking level for reasoning-capable models", async () => { - const result = await generateSummary( + const result = await generateSummaryWithUsage( messages, createModel(true), 2000, @@ -79,6 +84,12 @@ describe("generateSummary reasoning options", () => { }); }); + it("preserves the string result from generateSummary", async () => { + await expect(generateSummary(messages, createModel(false), 2000, "test-key")).resolves.toBe( + "## Goal\nTest summary", + ); + }); + it("does not set reasoning when thinking is off", async () => { await generateSummary( messages, diff --git a/packages/coding-agent/test/session-manager/tree-traversal.test.ts b/packages/coding-agent/test/session-manager/tree-traversal.test.ts index 9f447de1..d6e20f04 100644 --- a/packages/coding-agent/test/session-manager/tree-traversal.test.ts +++ b/packages/coding-agent/test/session-manager/tree-traversal.test.ts @@ -523,6 +523,52 @@ describe("createBranchedSession", () => { } }); + it("preserves tool and summary usage across a file-backed reload", () => { + const tempDir = join(tmpdir(), `session-usage-roundtrip-${Date.now()}`); + mkdirSync(tempDir, { recursive: true }); + + try { + const session = SessionManager.create(tempDir, tempDir); + const rootId = session.appendMessage(userMsg("question")); + session.appendMessage(assistantMsg("answer")); + const usage = { + input: 10, + output: 20, + cacheRead: 30, + cacheWrite: 40, + totalTokens: 100, + cost: { input: 0.1, output: 0.2, cacheRead: 0.3, cacheWrite: 0.4, total: 1 }, + }; + session.appendMessage({ + role: "toolResult", + toolCallId: "call-1", + toolName: "nested-model", + content: [{ type: "text", text: "result" }], + isError: false, + usage, + timestamp: Date.now(), + }); + session.appendCompaction("summary", rootId, 100, undefined, false, usage); + session.branchWithSummary(rootId, "branch summary", undefined, false, usage); + + const file = session.getSessionFile(); + expect(file).toBeDefined(); + const reopened = SessionManager.open(file!, tempDir); + expect(reopened.getEntries()).toEqual( + expect.arrayContaining([ + expect.objectContaining({ type: "compaction", usage }), + expect.objectContaining({ type: "branch_summary", usage }), + expect.objectContaining({ + type: "message", + message: expect.objectContaining({ role: "toolResult", usage }), + }), + ]), + ); + } finally { + rmSync(tempDir, { recursive: true, force: true }); + } + }); + it("writes file immediately when forking from a point with assistant messages", () => { const tempDir = join(tmpdir(), `session-fork-with-assistant-${Date.now()}`); mkdirSync(tempDir, { recursive: true });