@@ -6,6 +6,10 @@
|
|||||||
|
|
||||||
- Moved the `uuidv7` export to `@earendil-works/pi-ai`.
|
- Moved the `uuidv7` export to `@earendil-works/pi-ai`.
|
||||||
|
|
||||||
|
### Added
|
||||||
|
|
||||||
|
- Added usage metadata to tool results, compaction entries, and branch summaries in the agent harness ([#6671](https://github.com/earendil-works/pi/pull/6671) by [@davidbrai](https://github.com/davidbrai)).
|
||||||
|
|
||||||
## [0.80.10] - 2026-07-16
|
## [0.80.10] - 2026-07-16
|
||||||
|
|
||||||
## [0.80.9] - 2026-07-16
|
## [0.80.9] - 2026-07-16
|
||||||
|
|||||||
@@ -499,6 +499,30 @@ export async function generateSummary(
|
|||||||
customInstructions?: string,
|
customInstructions?: string,
|
||||||
previousSummary?: string,
|
previousSummary?: string,
|
||||||
thinkingLevel?: ThinkingLevel,
|
thinkingLevel?: ThinkingLevel,
|
||||||
|
): Promise<Result<string, CompactionError>> {
|
||||||
|
const result = await generateSummaryWithUsage(
|
||||||
|
currentMessages,
|
||||||
|
models,
|
||||||
|
model,
|
||||||
|
reserveTokens,
|
||||||
|
signal,
|
||||||
|
customInstructions,
|
||||||
|
previousSummary,
|
||||||
|
thinkingLevel,
|
||||||
|
);
|
||||||
|
return result.ok ? ok(result.value.text) : err(result.error);
|
||||||
|
}
|
||||||
|
|
||||||
|
/** Generate or update a conversation summary and return its provider usage. */
|
||||||
|
export async function generateSummaryWithUsage(
|
||||||
|
currentMessages: AgentMessage[],
|
||||||
|
models: Models,
|
||||||
|
model: Model<any>,
|
||||||
|
reserveTokens: number,
|
||||||
|
signal?: AbortSignal,
|
||||||
|
customInstructions?: string,
|
||||||
|
previousSummary?: string,
|
||||||
|
thinkingLevel?: ThinkingLevel,
|
||||||
): Promise<Result<{ text: string; usage: Usage }, CompactionError>> {
|
): Promise<Result<{ text: string; usage: Usage }, CompactionError>> {
|
||||||
const maxTokens = Math.min(
|
const maxTokens = Math.min(
|
||||||
Math.floor(0.8 * reserveTokens),
|
Math.floor(0.8 * reserveTokens),
|
||||||
@@ -687,7 +711,7 @@ export async function compact(
|
|||||||
let historyText = "No prior history.";
|
let historyText = "No prior history.";
|
||||||
let historyUsage: Usage | undefined;
|
let historyUsage: Usage | undefined;
|
||||||
if (messagesToSummarize.length > 0) {
|
if (messagesToSummarize.length > 0) {
|
||||||
const historyResult = await generateSummary(
|
const historyResult = await generateSummaryWithUsage(
|
||||||
messagesToSummarize,
|
messagesToSummarize,
|
||||||
models,
|
models,
|
||||||
model,
|
model,
|
||||||
@@ -715,7 +739,7 @@ export async function compact(
|
|||||||
? combineUsage(historyUsage, turnPrefixResult.value.usage)
|
? combineUsage(historyUsage, turnPrefixResult.value.usage)
|
||||||
: turnPrefixResult.value.usage;
|
: turnPrefixResult.value.usage;
|
||||||
} else {
|
} else {
|
||||||
const summaryResult = await generateSummary(
|
const summaryResult = await generateSummaryWithUsage(
|
||||||
messagesToSummarize,
|
messagesToSummarize,
|
||||||
models,
|
models,
|
||||||
model,
|
model,
|
||||||
|
|||||||
@@ -20,6 +20,7 @@ export {
|
|||||||
findCutPoint,
|
findCutPoint,
|
||||||
findTurnStartIndex,
|
findTurnStartIndex,
|
||||||
generateSummary,
|
generateSummary,
|
||||||
|
generateSummaryWithUsage,
|
||||||
getLastAssistantUsage,
|
getLastAssistantUsage,
|
||||||
prepareCompaction,
|
prepareCompaction,
|
||||||
serializeConversation,
|
serializeConversation,
|
||||||
|
|||||||
@@ -20,6 +20,7 @@ import {
|
|||||||
findCutPoint,
|
findCutPoint,
|
||||||
findTurnStartIndex,
|
findTurnStartIndex,
|
||||||
generateSummary,
|
generateSummary,
|
||||||
|
generateSummaryWithUsage,
|
||||||
getLastAssistantUsage,
|
getLastAssistantUsage,
|
||||||
prepareCompaction,
|
prepareCompaction,
|
||||||
serializeConversation,
|
serializeConversation,
|
||||||
@@ -510,7 +511,7 @@ describe("harness compaction", () => {
|
|||||||
]);
|
]);
|
||||||
|
|
||||||
const summary = getOrThrow(
|
const summary = getOrThrow(
|
||||||
await generateSummary(messages, models, model, 2000, undefined, "focus", "old summary"),
|
await generateSummaryWithUsage(messages, models, model, 2000, undefined, "focus", "old summary"),
|
||||||
);
|
);
|
||||||
|
|
||||||
expect(summary.text).toContain("Test summary");
|
expect(summary.text).toContain("Test summary");
|
||||||
@@ -523,6 +524,14 @@ describe("harness compaction", () => {
|
|||||||
expect(promptText).toContain("Additional focus: focus");
|
expect(promptText).toContain("Additional focus: focus");
|
||||||
});
|
});
|
||||||
|
|
||||||
|
it("preserves the string result from generateSummary", async () => {
|
||||||
|
const messages: AgentMessage[] = [createUserMessage("Summarize this.")];
|
||||||
|
const { faux, model } = createFauxModel(false);
|
||||||
|
faux.setResponses([fauxAssistantMessage("## Goal\nTest summary")]);
|
||||||
|
|
||||||
|
expect(getOrThrow(await generateSummary(messages, models, model, 2000))).toBe("## Goal\nTest summary");
|
||||||
|
});
|
||||||
|
|
||||||
it("returns error results for failed or aborted summary generations", async () => {
|
it("returns error results for failed or aborted summary generations", async () => {
|
||||||
const messages: AgentMessage[] = [createUserMessage("Summarize this.")];
|
const messages: AgentMessage[] = [createUserMessage("Summarize this.")];
|
||||||
const { faux: errorFaux, model: errorModel } = createFauxModel(false);
|
const { faux: errorFaux, model: errorModel } = createFauxModel(false);
|
||||||
|
|||||||
@@ -6,6 +6,7 @@
|
|||||||
|
|
||||||
- Added `contentText` for extracting joined text from message content.
|
- Added `contentText` for extracting joined text from message content.
|
||||||
- Added a shared `uuidv7` utility for time-ordered identifiers.
|
- Added a shared `uuidv7` utility for time-ordered identifiers.
|
||||||
|
- Added optional usage metadata to tool result messages ([#6671](https://github.com/earendil-works/pi/pull/6671) by [@davidbrai](https://github.com/davidbrai)).
|
||||||
|
|
||||||
### Fixed
|
### Fixed
|
||||||
|
|
||||||
|
|||||||
@@ -6,6 +6,7 @@
|
|||||||
|
|
||||||
- Added built-in llama.cpp router support with `/login` connection setup and `/llama` Hugging Face model search and downloads, explicit loading, unloading, and live progress. See [llama.cpp](docs/llama-cpp.md).
|
- Added built-in llama.cpp router support with `/login` connection setup and `/llama` Hugging Face model search and downloads, explicit loading, unloading, and live progress. See [llama.cpp](docs/llama-cpp.md).
|
||||||
- Added extension registration for complete pi-ai providers, including native authentication, model refresh, filtering, and streaming behavior.
|
- Added extension registration for complete pi-ai providers, including native authentication, model refresh, filtering, and streaming behavior.
|
||||||
|
- Added usage accounting for tools, compaction, and branch summaries in persisted sessions, footer totals, and session statistics ([#6671](https://github.com/earendil-works/pi/pull/6671) by [@davidbrai](https://github.com/davidbrai)).
|
||||||
|
|
||||||
### Fixed
|
### Fixed
|
||||||
|
|
||||||
|
|||||||
@@ -152,7 +152,7 @@ The interface from top to bottom:
|
|||||||
- **Startup header** - Shows shortcuts (`/hotkeys` for all), loaded AGENTS.md files, prompt templates, skills, and extensions
|
- **Startup header** - Shows shortcuts (`/hotkeys` for all), loaded AGENTS.md files, prompt templates, skills, and extensions
|
||||||
- **Messages** - Your messages, assistant responses, tool calls and results, notifications, errors, and extension UI
|
- **Messages** - Your messages, assistant responses, tool calls and results, notifications, errors, and extension UI
|
||||||
- **Editor** - Where you type; border color indicates thinking level
|
- **Editor** - Where you type; border color indicates thinking level
|
||||||
- **Footer** - Working directory, session name, total token/cache usage (`↑` input, `↓` output, `R` cache read, `W` cache write, `CH` latest cache hit rate), cost, context usage, current model
|
- **Footer** - Working directory, session name, total token/cache usage (`↑` input, `↓` output, `R` cache read, `W` cache write, `CH` latest cache hit rate), cost, context usage, current model. Totals include assistant responses, usage reported by tools, and summary generation.
|
||||||
|
|
||||||
The editor can be temporarily replaced by other UI, like built-in `/settings` or custom UI from extensions (e.g., a Q&A tool that lets the user answer model questions in a structured format). [Extensions](#extensions) can also replace the editor, add widgets above/below it, a status line, custom footer, or overlays.
|
The editor can be temporarily replaced by other UI, like built-in `/settings` or custom UI from extensions (e.g., a Q&A tool that lets the user answer model questions in a structured format). [Extensions](#extensions) can also replace the editor, add widgets above/below it, a status line, custom footer, or overlays.
|
||||||
|
|
||||||
|
|||||||
@@ -129,6 +129,7 @@ interface CompactionEntry<T = unknown> {
|
|||||||
summary: string;
|
summary: string;
|
||||||
firstKeptEntryId: string;
|
firstKeptEntryId: string;
|
||||||
tokensBefore: number;
|
tokensBefore: number;
|
||||||
|
usage?: Usage; // LLM usage that generated the summary
|
||||||
fromHook?: boolean; // true if provided by extension (legacy field name)
|
fromHook?: boolean; // true if provided by extension (legacy field name)
|
||||||
details?: T; // implementation-specific data
|
details?: T; // implementation-specific data
|
||||||
}
|
}
|
||||||
@@ -140,9 +141,9 @@ interface CompactionDetails {
|
|||||||
}
|
}
|
||||||
```
|
```
|
||||||
|
|
||||||
Extensions can store any JSON-serializable data in `details`. The default compaction tracks file operations, but custom extension implementations can use their own structure.
|
Extensions can store any JSON-serializable data in `details`. The default compaction tracks file operations, but custom extension implementations can use their own structure. Generated and extension-provided summaries store their LLM `usage` when available so session totals include summarization work.
|
||||||
|
|
||||||
See [`prepareCompaction()`](https://github.com/earendil-works/pi-mono/blob/main/packages/coding-agent/src/core/compaction/compaction.ts) and [`compact()`](https://github.com/earendil-works/pi-mono/blob/main/packages/coding-agent/src/core/compaction/compaction.ts) for the implementation.
|
See [`prepareCompaction()`](https://github.com/earendil-works/pi-mono/blob/main/packages/coding-agent/src/core/compaction/compaction.ts) and [`compact()`](https://github.com/earendil-works/pi-mono/blob/main/packages/coding-agent/src/core/compaction/compaction.ts) for the implementation. For direct programmatic summarization, `generateSummary()` returns the summary text and `generateSummaryWithUsage()` returns `{ text, usage }`.
|
||||||
|
|
||||||
## Branch Summarization
|
## Branch Summarization
|
||||||
|
|
||||||
@@ -195,6 +196,7 @@ interface BranchSummaryEntry<T = unknown> {
|
|||||||
timestamp: number;
|
timestamp: number;
|
||||||
summary: string;
|
summary: string;
|
||||||
fromId: string; // Entry we navigated from
|
fromId: string; // Entry we navigated from
|
||||||
|
usage?: Usage; // LLM usage that generated the summary
|
||||||
fromHook?: boolean; // true if provided by extension (legacy field name)
|
fromHook?: boolean; // true if provided by extension (legacy field name)
|
||||||
details?: T; // implementation-specific data
|
details?: T; // implementation-specific data
|
||||||
}
|
}
|
||||||
@@ -300,6 +302,7 @@ pi.on("session_before_compact", async (event, ctx) => {
|
|||||||
summary: "Your summary...",
|
summary: "Your summary...",
|
||||||
firstKeptEntryId: preparation.firstKeptEntryId,
|
firstKeptEntryId: preparation.firstKeptEntryId,
|
||||||
tokensBefore: preparation.tokensBefore,
|
tokensBefore: preparation.tokensBefore,
|
||||||
|
// usage: summaryResponse.usage, // Optional; included in session totals
|
||||||
details: { /* custom data */ },
|
details: { /* custom data */ },
|
||||||
}
|
}
|
||||||
};
|
};
|
||||||
@@ -328,13 +331,14 @@ pi.on("session_before_compact", async (event, ctx) => {
|
|||||||
// [Tool result]: output text
|
// [Tool result]: output text
|
||||||
|
|
||||||
// Now send to your model for summarization
|
// Now send to your model for summarization
|
||||||
const summary = await myModel.summarize(conversationText);
|
const { summary, usage } = await myModel.summarize(conversationText);
|
||||||
|
|
||||||
return {
|
return {
|
||||||
compaction: {
|
compaction: {
|
||||||
summary,
|
summary,
|
||||||
firstKeptEntryId: preparation.firstKeptEntryId,
|
firstKeptEntryId: preparation.firstKeptEntryId,
|
||||||
tokensBefore: preparation.tokensBefore,
|
tokensBefore: preparation.tokensBefore,
|
||||||
|
usage,
|
||||||
}
|
}
|
||||||
};
|
};
|
||||||
});
|
});
|
||||||
@@ -364,6 +368,7 @@ pi.on("session_before_tree", async (event, ctx) => {
|
|||||||
return {
|
return {
|
||||||
summary: {
|
summary: {
|
||||||
summary: "Your summary...",
|
summary: "Your summary...",
|
||||||
|
// usage: summaryResponse.usage, // Optional; included in session totals
|
||||||
details: { /* custom data */ },
|
details: { /* custom data */ },
|
||||||
}
|
}
|
||||||
};
|
};
|
||||||
|
|||||||
@@ -468,6 +468,7 @@ pi.on("session_before_compact", async (event, ctx) => {
|
|||||||
summary: "...",
|
summary: "...",
|
||||||
firstKeptEntryId: preparation.firstKeptEntryId,
|
firstKeptEntryId: preparation.firstKeptEntryId,
|
||||||
tokensBefore: preparation.tokensBefore,
|
tokensBefore: preparation.tokensBefore,
|
||||||
|
// usage: summaryResponse.usage, // Optional; included in session totals
|
||||||
}
|
}
|
||||||
};
|
};
|
||||||
});
|
});
|
||||||
@@ -489,7 +490,13 @@ pi.on("session_before_tree", async (event, ctx) => {
|
|||||||
const { preparation, signal } = event;
|
const { preparation, signal } = event;
|
||||||
return { cancel: true };
|
return { cancel: true };
|
||||||
// OR provide custom summary:
|
// OR provide custom summary:
|
||||||
return { summary: { summary: "...", details: {} } };
|
return {
|
||||||
|
summary: {
|
||||||
|
summary: "...",
|
||||||
|
// usage: summaryResponse.usage, // Optional; included in session totals
|
||||||
|
details: {},
|
||||||
|
},
|
||||||
|
};
|
||||||
});
|
});
|
||||||
|
|
||||||
pi.on("session_tree", async (event, ctx) => {
|
pi.on("session_tree", async (event, ctx) => {
|
||||||
@@ -813,7 +820,7 @@ In parallel tool mode, `tool_result` and `tool_execution_end` may interleave in
|
|||||||
`tool_result` handlers chain like middleware:
|
`tool_result` handlers chain like middleware:
|
||||||
- Handlers run in extension load order
|
- Handlers run in extension load order
|
||||||
- Each handler sees the latest result after previous handler changes
|
- Each handler sees the latest result after previous handler changes
|
||||||
- Handlers can return partial patches (`content`, `details`, or `isError`); omitted fields keep their current values
|
- Handlers can return partial patches (`content`, `details`, `isError`, or `usage`); omitted fields keep their current values
|
||||||
|
|
||||||
Use `ctx.signal` for nested async work inside the handler. This lets Esc cancel model calls, `fetch()`, and other abort-aware operations started by the extension.
|
Use `ctx.signal` for nested async work inside the handler. This lets Esc cancel model calls, `fetch()`, and other abort-aware operations started by the extension.
|
||||||
|
|
||||||
@@ -822,7 +829,7 @@ import { isBashToolResult } from "@earendil-works/pi-coding-agent";
|
|||||||
|
|
||||||
pi.on("tool_result", async (event, ctx) => {
|
pi.on("tool_result", async (event, ctx) => {
|
||||||
// event.toolName, event.toolCallId, event.input
|
// event.toolName, event.toolCallId, event.input
|
||||||
// event.content, event.details, event.isError
|
// event.content, event.details, event.isError, event.usage
|
||||||
|
|
||||||
if (isBashToolResult(event)) {
|
if (isBashToolResult(event)) {
|
||||||
// event.details is typed as BashToolDetails
|
// event.details is typed as BashToolDetails
|
||||||
@@ -835,7 +842,7 @@ pi.on("tool_result", async (event, ctx) => {
|
|||||||
});
|
});
|
||||||
|
|
||||||
// Modify result:
|
// Modify result:
|
||||||
return { content: [...], details: {...}, isError: false };
|
return { content: [...], details: {...}, isError: false, usage: nestedModelUsage };
|
||||||
});
|
});
|
||||||
```
|
```
|
||||||
|
|
||||||
@@ -1932,6 +1939,7 @@ pi.registerTool({
|
|||||||
return {
|
return {
|
||||||
content: [{ type: "text", text: "Done" }], // Sent to LLM
|
content: [{ type: "text", text: "Done" }], // Sent to LLM
|
||||||
details: { data: result }, // For rendering & state
|
details: { data: result }, // For rendering & state
|
||||||
|
// usage: nestedModelResponse.usage, // Optional nested LLM usage
|
||||||
// Optional: stop after this tool batch when every finalized tool result
|
// Optional: stop after this tool batch when every finalized tool result
|
||||||
// in the batch also returns terminate: true.
|
// in the batch also returns terminate: true.
|
||||||
terminate: true,
|
terminate: true,
|
||||||
@@ -1944,6 +1952,8 @@ pi.registerTool({
|
|||||||
});
|
});
|
||||||
```
|
```
|
||||||
|
|
||||||
|
**Usage accounting:** If a tool makes nested LLM calls, return their combined `Usage` as `usage`. Pi persists it on the tool result and includes it in footer, `/session`, and RPC session totals. `tool_result` handlers can inspect or replace this value.
|
||||||
|
|
||||||
**Signaling errors:** To mark a tool execution as failed (sets `isError: true` on the result and reports it to the LLM), throw an error from `execute`. Returning a value never sets the error flag regardless of what properties you include in the return object.
|
**Signaling errors:** To mark a tool execution as failed (sets `isError: true` on the result and reports it to the LLM), throw an error from `execute`. Returning a value never sets the error flag regardless of what properties you include in the return object.
|
||||||
|
|
||||||
**Early termination:** Return `terminate: true` from `execute()` to hint that the automatic follow-up LLM call should be skipped after the current tool batch. This only takes effect when every finalized tool result in that batch is terminating. See [examples/extensions/structured-output.ts](../examples/extensions/structured-output.ts) for a minimal example where the agent ends on a final structured-output tool call.
|
**Early termination:** Return `terminate: true` from `execute()` to hint that the automatic follow-up LLM call should be skipped after the current tool batch. This only takes effect when every finalized tool result in that batch is terminating. See [examples/extensions/structured-output.ts](../examples/extensions/structured-output.ts) for a minimal example where the agent ends on a final structured-output tool call.
|
||||||
|
|||||||
@@ -395,12 +395,20 @@ Response:
|
|||||||
"firstKeptEntryId": "abc123",
|
"firstKeptEntryId": "abc123",
|
||||||
"tokensBefore": 150000,
|
"tokensBefore": 150000,
|
||||||
"estimatedTokensAfter": 32000,
|
"estimatedTokensAfter": 32000,
|
||||||
|
"usage": {
|
||||||
|
"input": 32000,
|
||||||
|
"output": 1200,
|
||||||
|
"cacheRead": 0,
|
||||||
|
"cacheWrite": 0,
|
||||||
|
"totalTokens": 33200,
|
||||||
|
"cost": {"input": 0.01, "output": 0.02, "cacheRead": 0, "cacheWrite": 0, "total": 0.03}
|
||||||
|
},
|
||||||
"details": {}
|
"details": {}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
```
|
```
|
||||||
|
|
||||||
`estimatedTokensAfter` is a heuristic estimate over the rebuilt message context immediately after compaction, not a provider-exact token count.
|
`estimatedTokensAfter` is a heuristic estimate over the rebuilt message context immediately after compaction, not a provider-exact token count. `usage` reports the LLM call or calls that generated the summary and may be omitted by custom compaction handlers.
|
||||||
|
|
||||||
#### set_auto_compaction
|
#### set_auto_compaction
|
||||||
|
|
||||||
@@ -557,7 +565,7 @@ Response:
|
|||||||
}
|
}
|
||||||
```
|
```
|
||||||
|
|
||||||
`tokens` contains assistant usage totals for the current session state. `contextUsage` contains the actual current context-window estimate used for compaction and footer display.
|
`tokens` and `cost` include assistant messages, usage reported by tools, and compaction/branch-summary generation across the full session. `contextUsage` contains the actual current context-window estimate used for compaction and footer display.
|
||||||
|
|
||||||
`contextUsage` is omitted when no model or context window is available. `contextUsage.tokens` and `contextUsage.percent` are `null` immediately after compaction until a fresh post-compaction assistant response provides valid usage data.
|
`contextUsage` is omitted when no model or context window is available. `contextUsage.tokens` and `contextUsage.percent` are `null` immediately after compaction until a fresh post-compaction assistant response provides valid usage data.
|
||||||
|
|
||||||
@@ -1016,6 +1024,14 @@ The `reason` field is `"manual"`, `"threshold"`, or `"overflow"`.
|
|||||||
"firstKeptEntryId": "abc123",
|
"firstKeptEntryId": "abc123",
|
||||||
"tokensBefore": 150000,
|
"tokensBefore": 150000,
|
||||||
"estimatedTokensAfter": 32000,
|
"estimatedTokensAfter": 32000,
|
||||||
|
"usage": {
|
||||||
|
"input": 32000,
|
||||||
|
"output": 1200,
|
||||||
|
"cacheRead": 0,
|
||||||
|
"cacheWrite": 0,
|
||||||
|
"totalTokens": 33200,
|
||||||
|
"cost": {"input": 0.01, "output": 0.02, "cacheRead": 0, "cacheWrite": 0, "total": 0.03}
|
||||||
|
},
|
||||||
"details": {}
|
"details": {}
|
||||||
},
|
},
|
||||||
"aborted": false,
|
"aborted": false,
|
||||||
@@ -1368,11 +1384,21 @@ Stop reasons: `"stop"`, `"length"`, `"toolUse"`, `"error"`, `"aborted"`
|
|||||||
"toolCallId": "call_123",
|
"toolCallId": "call_123",
|
||||||
"toolName": "bash",
|
"toolName": "bash",
|
||||||
"content": [{"type": "text", "text": "total 48\ndrwxr-xr-x ..."}],
|
"content": [{"type": "text", "text": "total 48\ndrwxr-xr-x ..."}],
|
||||||
|
"usage": {
|
||||||
|
"input": 100,
|
||||||
|
"output": 50,
|
||||||
|
"cacheRead": 0,
|
||||||
|
"cacheWrite": 0,
|
||||||
|
"totalTokens": 150,
|
||||||
|
"cost": {"input": 0.0003, "output": 0.00075, "cacheRead": 0, "cacheWrite": 0, "total": 0.00105}
|
||||||
|
},
|
||||||
"isError": false,
|
"isError": false,
|
||||||
"timestamp": 1733234567890
|
"timestamp": 1733234567890
|
||||||
}
|
}
|
||||||
```
|
```
|
||||||
|
|
||||||
|
`usage` is optional and reports nested LLM work performed by the tool. When present, it contributes to session token and cost totals.
|
||||||
|
|
||||||
### BashExecutionMessage
|
### BashExecutionMessage
|
||||||
|
|
||||||
Created by the `bash` RPC command (not by LLM tool calls):
|
Created by the `bash` RPC command (not by LLM tool calls):
|
||||||
|
|||||||
@@ -96,6 +96,7 @@ interface ToolResultMessage {
|
|||||||
toolName: string;
|
toolName: string;
|
||||||
content: (TextContent | ImageContent)[];
|
content: (TextContent | ImageContent)[];
|
||||||
details?: any; // Tool-specific metadata
|
details?: any; // Tool-specific metadata
|
||||||
|
usage?: Usage; // Nested LLM work performed by the tool
|
||||||
isError: boolean;
|
isError: boolean;
|
||||||
timestamp: number;
|
timestamp: number;
|
||||||
}
|
}
|
||||||
@@ -232,6 +233,7 @@ Created when context is compacted. Stores a summary of earlier messages.
|
|||||||
```
|
```
|
||||||
|
|
||||||
Optional fields:
|
Optional fields:
|
||||||
|
- `usage`: LLM usage from generating the summary; included in session token and cost totals
|
||||||
- `details`: Implementation-specific data (e.g., `{ readFiles: string[], modifiedFiles: string[] }` for default, or custom data for extensions)
|
- `details`: Implementation-specific data (e.g., `{ readFiles: string[], modifiedFiles: string[] }` for default, or custom data for extensions)
|
||||||
- `fromHook`: `true` if generated by an extension, `false`/`undefined` if pi-generated (legacy field name)
|
- `fromHook`: `true` if generated by an extension, `false`/`undefined` if pi-generated (legacy field name)
|
||||||
|
|
||||||
@@ -244,6 +246,7 @@ Created when switching branches via `/tree` with an LLM generated summary of the
|
|||||||
```
|
```
|
||||||
|
|
||||||
Optional fields:
|
Optional fields:
|
||||||
|
- `usage`: LLM usage from generating the summary; included in session token and cost totals
|
||||||
- `details`: File tracking data (`{ readFiles: string[], modifiedFiles: string[] }`) for default, or custom data for extensions
|
- `details`: File tracking data (`{ readFiles: string[], modifiedFiles: string[] }`) for default, or custom data for extensions
|
||||||
- `fromHook`: `true` if generated by an extension, `false`/`undefined` if pi-generated (legacy field name)
|
- `fromHook`: `true` if generated by an extension, `false`/`undefined` if pi-generated (legacy field name)
|
||||||
|
|
||||||
|
|||||||
@@ -11,7 +11,7 @@ The interface has four main areas:
|
|||||||
- **Startup header** - shortcuts, loaded context files, prompt templates, skills, and extensions
|
- **Startup header** - shortcuts, loaded context files, prompt templates, skills, and extensions
|
||||||
- **Messages** - user messages, assistant responses, tool calls, tool results, notifications, errors, and extension UI
|
- **Messages** - user messages, assistant responses, tool calls, tool results, notifications, errors, and extension UI
|
||||||
- **Editor** - where you type; border color indicates the current thinking level
|
- **Editor** - where you type; border color indicates the current thinking level
|
||||||
- **Footer** - working directory, session name, token/cache usage, cost, context usage, and current model
|
- **Footer** - working directory, session name, token/cache usage, cost, context usage, and current model. Totals include assistant responses, usage reported by tools, and summary generation.
|
||||||
|
|
||||||
The editor can be replaced temporarily by built-in UI such as `/settings` or by custom extension UI.
|
The editor can be replaced temporarily by built-in UI such as `/settings` or by custom extension UI.
|
||||||
|
|
||||||
|
|||||||
@@ -116,6 +116,7 @@ ${conversationText}
|
|||||||
summary,
|
summary,
|
||||||
firstKeptEntryId,
|
firstKeptEntryId,
|
||||||
tokensBefore,
|
tokensBefore,
|
||||||
|
usage: response.usage,
|
||||||
},
|
},
|
||||||
};
|
};
|
||||||
} catch (error) {
|
} catch (error) {
|
||||||
|
|||||||
@@ -581,6 +581,37 @@ export async function generateSummary(
|
|||||||
thinkingLevel?: ThinkingLevel,
|
thinkingLevel?: ThinkingLevel,
|
||||||
streamFn?: StreamFn,
|
streamFn?: StreamFn,
|
||||||
env?: Record<string, string>,
|
env?: Record<string, string>,
|
||||||
|
): Promise<string> {
|
||||||
|
return (
|
||||||
|
await generateSummaryWithUsage(
|
||||||
|
currentMessages,
|
||||||
|
model,
|
||||||
|
reserveTokens,
|
||||||
|
apiKey,
|
||||||
|
headers,
|
||||||
|
signal,
|
||||||
|
customInstructions,
|
||||||
|
previousSummary,
|
||||||
|
thinkingLevel,
|
||||||
|
streamFn,
|
||||||
|
env,
|
||||||
|
)
|
||||||
|
).text;
|
||||||
|
}
|
||||||
|
|
||||||
|
/** Generate or update a conversation summary and return its provider usage. */
|
||||||
|
export async function generateSummaryWithUsage(
|
||||||
|
currentMessages: AgentMessage[],
|
||||||
|
model: Model<any>,
|
||||||
|
reserveTokens: number,
|
||||||
|
apiKey: string | undefined,
|
||||||
|
headers?: Record<string, string>,
|
||||||
|
signal?: AbortSignal,
|
||||||
|
customInstructions?: string,
|
||||||
|
previousSummary?: string,
|
||||||
|
thinkingLevel?: ThinkingLevel,
|
||||||
|
streamFn?: StreamFn,
|
||||||
|
env?: Record<string, string>,
|
||||||
): Promise<{ text: string; usage: Usage }> {
|
): Promise<{ text: string; usage: Usage }> {
|
||||||
const maxTokens = Math.min(
|
const maxTokens = Math.min(
|
||||||
Math.floor(0.8 * reserveTokens),
|
Math.floor(0.8 * reserveTokens),
|
||||||
@@ -790,7 +821,7 @@ export async function compact(
|
|||||||
let historyText = "No prior history.";
|
let historyText = "No prior history.";
|
||||||
let historyUsage: Usage | undefined;
|
let historyUsage: Usage | undefined;
|
||||||
if (messagesToSummarize.length > 0) {
|
if (messagesToSummarize.length > 0) {
|
||||||
const historyResult = await generateSummary(
|
const historyResult = await generateSummaryWithUsage(
|
||||||
messagesToSummarize,
|
messagesToSummarize,
|
||||||
model,
|
model,
|
||||||
settings.reserveTokens,
|
settings.reserveTokens,
|
||||||
@@ -822,7 +853,7 @@ export async function compact(
|
|||||||
summaryUsage = historyUsage ? combineUsage(historyUsage, turnPrefixResult.usage) : turnPrefixResult.usage;
|
summaryUsage = historyUsage ? combineUsage(historyUsage, turnPrefixResult.usage) : turnPrefixResult.usage;
|
||||||
} else {
|
} else {
|
||||||
// Just generate history summary
|
// Just generate history summary
|
||||||
const result = await generateSummary(
|
const result = await generateSummaryWithUsage(
|
||||||
messagesToSummarize,
|
messagesToSummarize,
|
||||||
model,
|
model,
|
||||||
settings.reserveTokens,
|
settings.reserveTokens,
|
||||||
|
|||||||
@@ -1,4 +1,5 @@
|
|||||||
import type { Usage } from "@earendil-works/pi-ai/compat";
|
import type { Usage } from "@earendil-works/pi-ai/compat";
|
||||||
|
import type { SessionEntry } from "./session-manager.ts";
|
||||||
|
|
||||||
export interface UsageTotals {
|
export interface UsageTotals {
|
||||||
input: number;
|
input: number;
|
||||||
@@ -25,3 +26,45 @@ export function addUsageToTotals(totals: UsageTotals, usage: Usage): void {
|
|||||||
totals.cacheWrite += usage.cacheWrite;
|
totals.cacheWrite += usage.cacheWrite;
|
||||||
totals.cost += usage.cost.total;
|
totals.cost += usage.cost.total;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
export interface UsageCostBreakdownEntry {
|
||||||
|
key: string;
|
||||||
|
cost: number;
|
||||||
|
tokens: number;
|
||||||
|
}
|
||||||
|
|
||||||
|
/** Group attributable assistant usage by model and all other usage into a separate bucket. */
|
||||||
|
export function getUsageCostBreakdown(entries: SessionEntry[]): UsageCostBreakdownEntry[] {
|
||||||
|
const totalsByKey = new Map<string, UsageTotals>();
|
||||||
|
|
||||||
|
for (const entry of entries) {
|
||||||
|
let key: string | undefined;
|
||||||
|
let usage: Usage | undefined;
|
||||||
|
if (entry.type === "message" && entry.message.role === "assistant") {
|
||||||
|
key = `${entry.message.provider}/${entry.message.responseModel ?? entry.message.model}`;
|
||||||
|
usage = entry.message.usage;
|
||||||
|
} else if (entry.type === "message" && entry.message.role === "toolResult" && entry.message.usage) {
|
||||||
|
key = "Tools/summaries";
|
||||||
|
usage = entry.message.usage;
|
||||||
|
} else if ((entry.type === "branch_summary" || entry.type === "compaction") && entry.usage) {
|
||||||
|
key = "Tools/summaries";
|
||||||
|
usage = entry.usage;
|
||||||
|
}
|
||||||
|
if (!key || !usage) continue;
|
||||||
|
|
||||||
|
let totals = totalsByKey.get(key);
|
||||||
|
if (!totals) {
|
||||||
|
totals = createUsageTotals();
|
||||||
|
totalsByKey.set(key, totals);
|
||||||
|
}
|
||||||
|
addUsageToTotals(totals, usage);
|
||||||
|
}
|
||||||
|
|
||||||
|
return Array.from(totalsByKey, ([key, totals]) => ({
|
||||||
|
key,
|
||||||
|
cost: totals.cost,
|
||||||
|
tokens: totals.input + totals.output + totals.cacheRead + totals.cacheWrite,
|
||||||
|
}))
|
||||||
|
.filter((entry) => entry.cost > 0 || entry.tokens > 0)
|
||||||
|
.sort((a, b) => b.cost - a.cost);
|
||||||
|
}
|
||||||
|
|||||||
@@ -42,6 +42,7 @@ export {
|
|||||||
type GenerateBranchSummaryOptions,
|
type GenerateBranchSummaryOptions,
|
||||||
generateBranchSummary,
|
generateBranchSummary,
|
||||||
generateSummary,
|
generateSummary,
|
||||||
|
generateSummaryWithUsage,
|
||||||
getLastAssistantUsage,
|
getLastAssistantUsage,
|
||||||
prepareBranchEntries,
|
prepareBranchEntries,
|
||||||
serializeConversation,
|
serializeConversation,
|
||||||
|
|||||||
@@ -86,6 +86,7 @@ import type { SourceInfo } from "../../core/source-info.ts";
|
|||||||
import { isInstallTelemetryEnabled } from "../../core/telemetry.ts";
|
import { isInstallTelemetryEnabled } from "../../core/telemetry.ts";
|
||||||
import type { TruncationResult } from "../../core/tools/truncate.ts";
|
import type { TruncationResult } from "../../core/tools/truncate.ts";
|
||||||
import { hasTrustRequiringProjectResources, ProjectTrustStore } from "../../core/trust-manager.ts";
|
import { hasTrustRequiringProjectResources, ProjectTrustStore } from "../../core/trust-manager.ts";
|
||||||
|
import { getUsageCostBreakdown } from "../../core/usage-totals.ts";
|
||||||
import { getChangelogPath, getNewEntries, normalizeChangelogLinks, parseChangelog } from "../../utils/changelog.ts";
|
import { getChangelogPath, getNewEntries, normalizeChangelogLinks, parseChangelog } from "../../utils/changelog.ts";
|
||||||
import { copyToClipboard, readClipboardText } from "../../utils/clipboard.ts";
|
import { copyToClipboard, readClipboardText } from "../../utils/clipboard.ts";
|
||||||
import { extensionForImageMimeType, readClipboardImage } from "../../utils/clipboard-image.ts";
|
import { extensionForImageMimeType, readClipboardImage } from "../../utils/clipboard-image.ts";
|
||||||
@@ -5597,22 +5598,9 @@ export class InteractiveMode {
|
|||||||
const cacheWaste = computeCacheWaste(entries, this.session.modelRuntime);
|
const cacheWaste = computeCacheWaste(entries, this.session.modelRuntime);
|
||||||
|
|
||||||
// Cost/token totals per provider/model actually used (e.g. OpenRouter `auto`
|
// Cost/token totals per provider/model actually used (e.g. OpenRouter `auto`
|
||||||
// resolves to a concrete responseModel), sorted by cost descending.
|
// resolves to a concrete responseModel). Usage without model attribution is
|
||||||
const perModelMap = new Map<string, { key: string; cost: number; tokens: number }>();
|
// grouped separately so the breakdown reconciles with the session total.
|
||||||
for (const entry of entries) {
|
const usageBreakdown = getUsageCostBreakdown(entries);
|
||||||
if (entry.type !== "message" || entry.message.role !== "assistant") continue;
|
|
||||||
const message = entry.message;
|
|
||||||
const usage = message.usage;
|
|
||||||
const key = `${message.provider}/${message.responseModel ?? message.model}`;
|
|
||||||
let bucket = perModelMap.get(key);
|
|
||||||
if (!bucket) {
|
|
||||||
bucket = { key, cost: 0, tokens: 0 };
|
|
||||||
perModelMap.set(key, bucket);
|
|
||||||
}
|
|
||||||
bucket.cost += usage.cost.total;
|
|
||||||
bucket.tokens += usage.input + usage.output + usage.cacheRead + usage.cacheWrite;
|
|
||||||
}
|
|
||||||
const perModel = Array.from(perModelMap.values()).sort((a, b) => b.cost - a.cost);
|
|
||||||
|
|
||||||
let info = `${theme.bold("Session Info")}\n\n`;
|
let info = `${theme.bold("Session Info")}\n\n`;
|
||||||
if (sessionName) {
|
if (sessionName) {
|
||||||
@@ -5646,8 +5634,8 @@ export class InteractiveMode {
|
|||||||
if (stats.cost > 0 || cacheWaste.missedTokens > 0) {
|
if (stats.cost > 0 || cacheWaste.missedTokens > 0) {
|
||||||
info += `\n${theme.bold("Cost")}\n`;
|
info += `\n${theme.bold("Cost")}\n`;
|
||||||
info += `${theme.fg("dim", "Total:")} $${stats.cost.toFixed(3)}`;
|
info += `${theme.fg("dim", "Total:")} $${stats.cost.toFixed(3)}`;
|
||||||
if (perModel.length > 1) {
|
if (usageBreakdown.length > 1) {
|
||||||
for (const entry of perModel) {
|
for (const entry of usageBreakdown) {
|
||||||
info += `\n ${theme.fg("dim", `${entry.key}:`)} $${entry.cost.toFixed(3)} ${theme.fg("dim", `(${formatTokens(entry.tokens)} tokens)`)}`;
|
info += `\n ${theme.fg("dim", `${entry.key}:`)} $${entry.cost.toFixed(3)} ${theme.fg("dim", `(${formatTokens(entry.tokens)} tokens)`)}`;
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -5,6 +5,7 @@ import { AgentSession } from "../src/core/agent-session.ts";
|
|||||||
import { AuthStorage } from "../src/core/auth-storage.ts";
|
import { AuthStorage } from "../src/core/auth-storage.ts";
|
||||||
import { SessionManager } from "../src/core/session-manager.ts";
|
import { SessionManager } from "../src/core/session-manager.ts";
|
||||||
import { SettingsManager } from "../src/core/settings-manager.ts";
|
import { SettingsManager } from "../src/core/settings-manager.ts";
|
||||||
|
import { getUsageCostBreakdown } from "../src/core/usage-totals.ts";
|
||||||
import { createInMemoryModelRegistry, getModelRuntime } from "./model-runtime-test-utils.ts";
|
import { createInMemoryModelRegistry, getModelRuntime } from "./model-runtime-test-utils.ts";
|
||||||
import { createTestResourceLoader } from "./utilities.ts";
|
import { createTestResourceLoader } from "./utilities.ts";
|
||||||
|
|
||||||
@@ -224,6 +225,31 @@ describe("AgentSession.getSessionStats", () => {
|
|||||||
}
|
}
|
||||||
});
|
});
|
||||||
|
|
||||||
|
it("groups tool and summary usage separately from model-attributed usage", () => {
|
||||||
|
const sessionManager = SessionManager.inMemory();
|
||||||
|
const rootId = sessionManager.appendMessage(createUserMessage("hello", 1));
|
||||||
|
sessionManager.appendMessage({
|
||||||
|
...createAssistantMessage("response", 100, 2),
|
||||||
|
usage: { ...createUsage(100), cost: { ...createUsage(100).cost, total: 0.5 } },
|
||||||
|
});
|
||||||
|
sessionManager.appendMessage(
|
||||||
|
createToolResultMessage({ ...createUsage(100), cost: { ...createUsage(100).cost, total: 1 } }),
|
||||||
|
);
|
||||||
|
sessionManager.appendCompaction("summary", rootId, 100, undefined, false, {
|
||||||
|
...createUsage(100),
|
||||||
|
cost: { ...createUsage(100).cost, total: 2 },
|
||||||
|
});
|
||||||
|
sessionManager.branchWithSummary(null, "branch summary", undefined, false, {
|
||||||
|
...createUsage(100),
|
||||||
|
cost: { ...createUsage(100).cost, total: 3 },
|
||||||
|
});
|
||||||
|
|
||||||
|
expect(getUsageCostBreakdown(sessionManager.getEntries())).toEqual([
|
||||||
|
{ key: "Tools/summaries", cost: 6, tokens: 300 },
|
||||||
|
{ key: `${model.provider}/${model.id}`, cost: 0.5, tokens: 100 },
|
||||||
|
]);
|
||||||
|
});
|
||||||
|
|
||||||
it("ignores zero-usage messages when checking for post-compaction context usage", async () => {
|
it("ignores zero-usage messages when checking for post-compaction context usage", async () => {
|
||||||
const { session, sessionManager } = await createSession();
|
const { session, sessionManager } = await createSession();
|
||||||
|
|
||||||
|
|||||||
@@ -1,7 +1,12 @@
|
|||||||
import type { AgentMessage } from "@earendil-works/pi-agent-core";
|
import type { AgentMessage } from "@earendil-works/pi-agent-core";
|
||||||
import type { AssistantMessage, Model } from "@earendil-works/pi-ai";
|
import type { AssistantMessage, Model } from "@earendil-works/pi-ai";
|
||||||
import { beforeEach, describe, expect, it, vi } from "vitest";
|
import { beforeEach, describe, expect, it, vi } from "vitest";
|
||||||
import { type CompactionPreparation, compact, generateSummary } from "../src/core/compaction/index.ts";
|
import {
|
||||||
|
type CompactionPreparation,
|
||||||
|
compact,
|
||||||
|
generateSummary,
|
||||||
|
generateSummaryWithUsage,
|
||||||
|
} from "../src/core/compaction/index.ts";
|
||||||
|
|
||||||
const { completeSimpleMock } = vi.hoisted(() => ({
|
const { completeSimpleMock } = vi.hoisted(() => ({
|
||||||
completeSimpleMock: vi.fn(),
|
completeSimpleMock: vi.fn(),
|
||||||
@@ -57,7 +62,7 @@ describe("generateSummary reasoning options", () => {
|
|||||||
});
|
});
|
||||||
|
|
||||||
it("uses the provided thinking level for reasoning-capable models", async () => {
|
it("uses the provided thinking level for reasoning-capable models", async () => {
|
||||||
const result = await generateSummary(
|
const result = await generateSummaryWithUsage(
|
||||||
messages,
|
messages,
|
||||||
createModel(true),
|
createModel(true),
|
||||||
2000,
|
2000,
|
||||||
@@ -79,6 +84,12 @@ describe("generateSummary reasoning options", () => {
|
|||||||
});
|
});
|
||||||
});
|
});
|
||||||
|
|
||||||
|
it("preserves the string result from generateSummary", async () => {
|
||||||
|
await expect(generateSummary(messages, createModel(false), 2000, "test-key")).resolves.toBe(
|
||||||
|
"## Goal\nTest summary",
|
||||||
|
);
|
||||||
|
});
|
||||||
|
|
||||||
it("does not set reasoning when thinking is off", async () => {
|
it("does not set reasoning when thinking is off", async () => {
|
||||||
await generateSummary(
|
await generateSummary(
|
||||||
messages,
|
messages,
|
||||||
|
|||||||
@@ -523,6 +523,52 @@ describe("createBranchedSession", () => {
|
|||||||
}
|
}
|
||||||
});
|
});
|
||||||
|
|
||||||
|
it("preserves tool and summary usage across a file-backed reload", () => {
|
||||||
|
const tempDir = join(tmpdir(), `session-usage-roundtrip-${Date.now()}`);
|
||||||
|
mkdirSync(tempDir, { recursive: true });
|
||||||
|
|
||||||
|
try {
|
||||||
|
const session = SessionManager.create(tempDir, tempDir);
|
||||||
|
const rootId = session.appendMessage(userMsg("question"));
|
||||||
|
session.appendMessage(assistantMsg("answer"));
|
||||||
|
const usage = {
|
||||||
|
input: 10,
|
||||||
|
output: 20,
|
||||||
|
cacheRead: 30,
|
||||||
|
cacheWrite: 40,
|
||||||
|
totalTokens: 100,
|
||||||
|
cost: { input: 0.1, output: 0.2, cacheRead: 0.3, cacheWrite: 0.4, total: 1 },
|
||||||
|
};
|
||||||
|
session.appendMessage({
|
||||||
|
role: "toolResult",
|
||||||
|
toolCallId: "call-1",
|
||||||
|
toolName: "nested-model",
|
||||||
|
content: [{ type: "text", text: "result" }],
|
||||||
|
isError: false,
|
||||||
|
usage,
|
||||||
|
timestamp: Date.now(),
|
||||||
|
});
|
||||||
|
session.appendCompaction("summary", rootId, 100, undefined, false, usage);
|
||||||
|
session.branchWithSummary(rootId, "branch summary", undefined, false, usage);
|
||||||
|
|
||||||
|
const file = session.getSessionFile();
|
||||||
|
expect(file).toBeDefined();
|
||||||
|
const reopened = SessionManager.open(file!, tempDir);
|
||||||
|
expect(reopened.getEntries()).toEqual(
|
||||||
|
expect.arrayContaining([
|
||||||
|
expect.objectContaining({ type: "compaction", usage }),
|
||||||
|
expect.objectContaining({ type: "branch_summary", usage }),
|
||||||
|
expect.objectContaining({
|
||||||
|
type: "message",
|
||||||
|
message: expect.objectContaining({ role: "toolResult", usage }),
|
||||||
|
}),
|
||||||
|
]),
|
||||||
|
);
|
||||||
|
} finally {
|
||||||
|
rmSync(tempDir, { recursive: true, force: true });
|
||||||
|
}
|
||||||
|
});
|
||||||
|
|
||||||
it("writes file immediately when forking from a point with assistant messages", () => {
|
it("writes file immediately when forking from a point with assistant messages", () => {
|
||||||
const tempDir = join(tmpdir(), `session-fork-with-assistant-${Date.now()}`);
|
const tempDir = join(tmpdir(), `session-fork-with-assistant-${Date.now()}`);
|
||||||
mkdirSync(tempDir, { recursive: true });
|
mkdirSync(tempDir, { recursive: true });
|
||||||
|
|||||||
Reference in New Issue
Block a user