@@ -8,6 +8,7 @@
|
|||||||
- Fixed OpenAI Codex user-agent construction to synchronously load Node OS metadata, avoiding a startup race that could report `pi (browser)` in Node/Bun.
|
- Fixed OpenAI Codex user-agent construction to synchronously load Node OS metadata, avoiding a startup race that could report `pi (browser)` in Node/Bun.
|
||||||
- Fixed Fireworks GLM 5.2 Fast to use the OpenAI-compatible endpoint and `thinkingLevelMap`, aligning it with GLM 5.2 ([#6195](https://github.com/earendil-works/pi/issues/6195)).
|
- Fixed Fireworks GLM 5.2 Fast to use the OpenAI-compatible endpoint and `thinkingLevelMap`, aligning it with GLM 5.2 ([#6195](https://github.com/earendil-works/pi/issues/6195)).
|
||||||
- Fixed Amazon Bedrock prompt-cache points for Claude Fable 5 and Claude Sonnet 5 ([#6235](https://github.com/earendil-works/pi/issues/6235)).
|
- Fixed Amazon Bedrock prompt-cache points for Claude Fable 5 and Claude Sonnet 5 ([#6235](https://github.com/earendil-works/pi/issues/6235)).
|
||||||
|
- Fixed DS4 server context overflow detection for `Prompt has ... tokens, but the configured context size is ... tokens` errors ([#6262](https://github.com/earendil-works/pi/issues/6262)).
|
||||||
|
|
||||||
### Added
|
### Added
|
||||||
|
|
||||||
|
|||||||
@@ -24,6 +24,7 @@ import type { AssistantMessage } from "../types.ts";
|
|||||||
* - GitHub Copilot: "prompt token count of X exceeds the limit of Y"
|
* - GitHub Copilot: "prompt token count of X exceeds the limit of Y"
|
||||||
* - MiniMax: "invalid params, context window exceeds limit"
|
* - MiniMax: "invalid params, context window exceeds limit"
|
||||||
* - Kimi For Coding: "Your request exceeded model token limit: X (requested: Y)"
|
* - Kimi For Coding: "Your request exceeded model token limit: X (requested: Y)"
|
||||||
|
* - DS4: "Prompt has X tokens, but the configured context size is Y tokens"
|
||||||
* - Cerebras: "400/413 status code (no body)"
|
* - Cerebras: "400/413 status code (no body)"
|
||||||
* - Mistral: "Prompt contains X tokens ... too large for model with Y maximum context length"
|
* - Mistral: "Prompt contains X tokens ... too large for model with Y maximum context length"
|
||||||
* - z.ai: Does NOT error, accepts overflow silently - handled via usage.input > contextWindow
|
* - z.ai: Does NOT error, accepts overflow silently - handled via usage.input > contextWindow
|
||||||
@@ -50,6 +51,7 @@ const OVERFLOW_PATTERNS = [
|
|||||||
/context window exceeds limit/i, // MiniMax
|
/context window exceeds limit/i, // MiniMax
|
||||||
/exceeded model token limit/i, // Kimi For Coding
|
/exceeded model token limit/i, // Kimi For Coding
|
||||||
/too large for model with \d+ maximum context length/i, // Mistral
|
/too large for model with \d+ maximum context length/i, // Mistral
|
||||||
|
/prompt has [\d,]+ tokens?, but the configured context size is [\d,]+ tokens?/i, // DS4 server
|
||||||
/model_context_window_exceeded/i, // z.ai non-standard finish_reason surfaced as error text
|
/model_context_window_exceeded/i, // z.ai non-standard finish_reason surfaced as error text
|
||||||
/prompt too long; exceeded (?:max )?context length/i, // Ollama explicit overflow error
|
/prompt too long; exceeded (?:max )?context length/i, // Ollama explicit overflow error
|
||||||
/context[_ ]length[_ ]exceeded/i, // Generic fallback
|
/context[_ ]length[_ ]exceeded/i, // Generic fallback
|
||||||
@@ -98,6 +100,7 @@ const NON_OVERFLOW_PATTERNS = [
|
|||||||
* - llama.cpp: "exceeds the available context size"
|
* - llama.cpp: "exceeds the available context size"
|
||||||
* - LM Studio: "greater than the context length"
|
* - LM Studio: "greater than the context length"
|
||||||
* - Kimi For Coding: "exceeded model token limit: X (requested: Y)"
|
* - Kimi For Coding: "exceeded model token limit: X (requested: Y)"
|
||||||
|
* - DS4: "Prompt has X tokens, but the configured context size is Y tokens"
|
||||||
*
|
*
|
||||||
* **Unreliable detection:**
|
* **Unreliable detection:**
|
||||||
* - z.ai: Sometimes accepts overflow silently (detectable via usage.input > contextWindow),
|
* - z.ai: Sometimes accepts overflow silently (detectable via usage.input > contextWindow),
|
||||||
|
|||||||
@@ -63,6 +63,18 @@ describe("isContextOverflow", () => {
|
|||||||
expect(isContextOverflow(message, 131072)).toBe(true);
|
expect(isContextOverflow(message, 131072)).toBe(true);
|
||||||
});
|
});
|
||||||
|
|
||||||
|
it("detects DS4 configured context size errors", () => {
|
||||||
|
const message = createErrorMessage(
|
||||||
|
"400 Prompt has 256468 tokens, but the configured context size is 256000 tokens",
|
||||||
|
);
|
||||||
|
expect(isContextOverflow(message, 256000)).toBe(true);
|
||||||
|
|
||||||
|
const commaMessage = createErrorMessage(
|
||||||
|
"Prompt has 5,958,968 tokens, but the configured context size is 256,000 tokens",
|
||||||
|
);
|
||||||
|
expect(isContextOverflow(commaMessage, 256000)).toBe(true);
|
||||||
|
});
|
||||||
|
|
||||||
it("does not treat generic non-overflow Ollama errors as overflow", () => {
|
it("does not treat generic non-overflow Ollama errors as overflow", () => {
|
||||||
const message = createErrorMessage("500 `model runner crashed unexpectedly`");
|
const message = createErrorMessage("500 `model runner crashed unexpectedly`");
|
||||||
expect(isContextOverflow(message, 32768)).toBe(false);
|
expect(isContextOverflow(message, 32768)).toBe(false);
|
||||||
|
|||||||
Reference in New Issue
Block a user