feat(coding-agent): improved prompt caching for ephemeral side-channel requests

- Added `buildSideRequestContext` to the `Agent` class to generate prompt-cache-friendly provider contexts.
- Updated ephemeral side-channel turns to forward the full tool catalog to maintain prompt cache hit rates.
- Injected a `developer` role reminder into ephemeral turns to instruct the model to suppress tool calls.
- Implemented automatic post-processing to strip any tool calls from ephemeral turn responses.
- Exported message and dialect helper functions in `agent-loop.ts` to support context construction.
This commit is contained in:
can1357
2026-06-20 23:18:29 +02:00
parent b88d97cd25
commit b5437c2cad
8 changed files with 277 additions and 21 deletions
+8 -1
View File
@@ -1,6 +1,13 @@
# Changelog
## [Unreleased]
### Added
- Added `buildSideRequestContext` to the `Agent` class to build prompt-cache-friendly provider Contexts for side-channels or ephemeral requests.
### Changed
- Exported helper functions `normalizeMessagesForProvider` and `resolveOwnedDialectFromEnv` from `packages/agent/src/agent-loop.ts`.
## [16.1.5] - 2026-06-19
@@ -901,4 +908,4 @@ Initial release under @oh-my-pi scope. See previous releases at [badlogic/pi-mon
### Changed
- `Agent` constructor now has all options optional (empty options use defaults).
- `queueMessage()` is now synchronous (no longer returns a Promise).
- `queueMessage()` is now synchronous (no longer returns a Promise).
+2 -2
View File
@@ -121,7 +121,7 @@ class HarmonyLeakInterruption extends Error {
this.name = "HarmonyLeakInterruption";
}
}
function resolveOwnedDialectFromEnv(value: string | undefined): Dialect | undefined {
export function resolveOwnedDialectFromEnv(value: string | undefined): Dialect | undefined {
switch (value) {
case "1":
case "true":
@@ -502,7 +502,7 @@ function createDetailedCapture(config: AgentLoopConfig): {
};
}
function normalizeMessagesForProvider(
export function normalizeMessagesForProvider(
messages: Context["messages"],
model: AgentLoopConfig["model"],
): Context["messages"] {
+35 -1
View File
@@ -24,9 +24,17 @@ import {
} from "@oh-my-pi/pi-ai";
import type { Dialect } from "@oh-my-pi/pi-ai/dialect";
import type { HarmonyAuditEvent } from "@oh-my-pi/pi-ai/utils/harmony-leak";
import { preferredDialect } from "@oh-my-pi/pi-catalog/identity";
import { getBundledModel } from "@oh-my-pi/pi-catalog/models";
import { logger } from "@oh-my-pi/pi-utils";
import { abortReasonText, agentLoop, agentLoopContinue } from "./agent-loop";
import {
abortReasonText,
agentLoop,
agentLoopContinue,
normalizeMessagesForProvider,
normalizeTools,
resolveOwnedDialectFromEnv,
} from "./agent-loop";
import type { AppendOnlyContextManager } from "./append-only-context";
import type {
AgentContext,
@@ -662,6 +670,32 @@ export class Agent {
this.#appendOnlyContext = manager;
}
/**
* Assemble the provider Context for a side-channel (no-loop) request, mirroring
* the main loop's prefix (system + normalized tools) so it shares the prompt
* cache. Never touches the append-only log or the tool-choice queue. Owned/
* in-band dialect sessions stay tools-less (matching their no-native-tools wire
* shape and avoiding tool-markup leakage). `llmMessages` is already converted
* (and, in production, obfuscated) by the caller.
*/
buildSideRequestContext(llmMessages: Message[]): Context {
const model = this.#state.model;
if (!model) throw new Error("No active model on agent");
const ownedDialect = this.#dialect ?? resolveOwnedDialectFromEnv(Bun.env.PI_DIALECT);
const messages = normalizeMessagesForProvider(llmMessages, model);
const tools = ownedDialect
? []
: (normalizeTools(
this.#state.tools,
this.#intentTracing,
preferredDialect(model.id),
this.#pruneToolDescriptions,
) ?? []);
let context: Context = { systemPrompt: this.#state.systemPrompt, messages, tools };
if (this.#transformProviderContext) context = this.#transformProviderContext(context, model);
return context;
}
subscribe(fn: (e: AgentEvent) => void): () => void {
this.#listeners.add(fn);
return () => this.#listeners.delete(fn);
@@ -0,0 +1,78 @@
import { describe, expect, it, mock } from "bun:test";
import { type Context, z } from "@oh-my-pi/pi-ai";
import { createMockModel } from "@oh-my-pi/pi-ai/providers/mock";
import { Agent } from "../src/agent";
import type { AgentTool } from "../src/types";
describe("Agent — buildSideRequestContext", () => {
const model = createMockModel({ responses: [] });
const tool: AgentTool = {
name: "test_tool",
label: "Test Tool",
description: "a cool tool",
parameters: z.object({ arg: z.string() }) as unknown as AgentTool["parameters"],
execute: async () => ({ content: [{ type: "text", text: "success" }], details: { value: "success" } }),
};
it("forwards the tool catalog for native providers", () => {
const agent = new Agent({
initialState: {
model,
systemPrompt: ["system"],
tools: [tool],
},
});
const context = agent.buildSideRequestContext([
{ role: "user", content: [{ type: "text", text: "Q?" }], timestamp: Date.now() },
]);
expect(context.tools).toBeDefined();
expect(context.tools!.length).toBe(1);
expect(context.tools![0].name).toBe("test_tool");
expect(context.systemPrompt).toEqual(["system"]);
});
it("returns empty tools when owned dialect is active", () => {
const agent = new Agent({
initialState: {
model,
systemPrompt: ["system"],
tools: [tool],
},
dialect: "glm",
});
const context = agent.buildSideRequestContext([
{ role: "user", content: [{ type: "text", text: "Q?" }], timestamp: Date.now() },
]);
expect(context.tools).toEqual([]);
expect(context.systemPrompt).toEqual(["system"]);
});
it("invokes transformProviderContext filter if present", () => {
const transformSpy = mock((ctx: Context): Context => {
return {
...ctx,
systemPrompt: ["transformed-system"],
};
});
const agent = new Agent({
initialState: {
model,
systemPrompt: ["system"],
tools: [tool],
},
transformProviderContext: transformSpy,
});
const context = agent.buildSideRequestContext([
{ role: "user", content: [{ type: "text", text: "Q?" }], timestamp: Date.now() },
]);
expect(transformSpy).toHaveBeenCalledTimes(1);
expect(context.systemPrompt).toEqual(["transformed-system"]);
});
});
+1 -1
View File
@@ -1,7 +1,6 @@
# Changelog
## [Unreleased]
### Added
- Added "Prose Only Thinking" setting to opt-out of rendering code blocks within AI thinking traces
@@ -11,6 +10,7 @@
### Changed
- Changed side-channel turns (`/btw`, `/omfg`, and IRC auto-replies) to forward the main turn's tool catalog to preserve the prompt-cache layout, while injecting a reminder to suppress tool usage and discarding any generated tool calls.
- Changed `/btw`, `/tan`, `/omfg`, `/memory`, `/rename`, and `/move` to save the typed command text to TUI prompt history so they can be recalled with the up arrow.
- Changed the temporary model picker to label Alt+P selections as session-only and point users to Alt+M or `/model` for role model assignment. ([#2952](https://github.com/can1357/oh-my-pi/issues/2952))
- Replaced `new Promise((resolve, reject) => ...)` in `AsyncDrain` with `Promise.withResolvers()` per the repo's promise-construction convention
@@ -0,0 +1,3 @@
<system-reminder>
This is an ephemeral side-channel turn that reuses the current conversation's context. The tool catalog stays attached only to keep the prompt cache warm — tools are NOT available on this turn. Do NOT emit any tool call; reply with plain text only. Any tool call you produce is discarded without executing.
</system-reminder>
@@ -74,7 +74,6 @@ import {
import type { ProtectedToolMatcher } from "@oh-my-pi/pi-agent-core/compaction/tool-protection";
import type {
AssistantMessage,
Context,
ImageContent,
Message,
MessageAttribution,
@@ -229,6 +228,7 @@ import planModeReferencePrompt from "../prompts/system/plan-mode-reference.md" w
import planModeToolDecisionReminderPrompt from "../prompts/system/plan-mode-tool-decision-reminder.md" with {
type: "text",
};
import sideChannelNoToolsReminder from "../prompts/system/side-channel-no-tools.md" with { type: "text" };
import ttsrInterruptTemplate from "../prompts/system/ttsr-interrupt.md" with { type: "text" };
import ttsrToolReminderTemplate from "../prompts/system/ttsr-tool-reminder.md" with { type: "text" };
import unexpectedStopRetryTemplate from "../prompts/system/unexpected-stop-retry.md" with { type: "text" };
@@ -11245,12 +11245,14 @@ export class AgentSession {
/**
* Run a single ephemeral side-channel turn against this session's current
* model + system prompt + history. No tools are used; the side request
* does not block on, or interfere with, any in-flight main turn. The
* model + system prompt + history. The main turn's tool catalog is sent
* to preserve the prompt cache, but the model is reminded not to call
* tools and any tool calls are discarded. The side request
* does not block on, or interfere with, any in-flight main turn. The
* session's history and persisted state are NOT modified by this call.
*
* Used by `BtwController` (`/btw`) and `OmfgController` (`/omfg`) to share
* the snapshot + stream pipeline. The snapshot includes any in-flight
* the snapshot + stream pipeline. The snapshot includes any in-flight
* streaming assistant text so the model sees the half-finished response
* rather than missing context.
*/
@@ -11267,15 +11269,7 @@ export class AgentSession {
const cacheSessionId = this.sessionId;
const snapshot = this.#buildEphemeralSnapshot(args.promptText);
const llmMessages = await this.convertMessagesToLlm(snapshot, args.signal);
const context: Context = {
systemPrompt: this.systemPrompt,
messages: llmMessages,
// Empty tools array: with toolChoice="none" some encoders still serialize the
// recipient's tool catalog and the model leaks raw call markup
// (<function_calls>, DSML envelopes) into IRC replies. Stripping tools here
// removes the surface entirely.
tools: [],
};
const context = this.agent.buildSideRequestContext(llmMessages);
const options = this.prepareSimpleStreamOptions(
{
apiKey: this.#modelRegistry.resolver(model, cacheSessionId),
@@ -11291,7 +11285,6 @@ export class AgentSession {
hideThinkingSummary: this.agent.hideThinkingSummary,
serviceTier: this.#effectiveServiceTier(model),
signal: args.signal,
toolChoice: "none",
},
model.provider,
);
@@ -11331,9 +11324,13 @@ export class AgentSession {
if (args.onTextDelta && replyText.length > emittedReplyText.length) {
args.onTextDelta(replyText.slice(emittedReplyText.length));
}
const sanitizedMessage: AssistantMessage = {
...assistantMessage,
content: assistantMessage.content.filter(block => block.type !== "toolCall"),
};
return {
replyText: args.dedupeReply === false ? replyText.trim() : dedupeEphemeralReply(replyText.trim()),
assistantMessage,
assistantMessage: sanitizedMessage,
};
}
@@ -11374,6 +11371,12 @@ export class AgentSession {
}
}
}
messages.push({
role: "developer",
content: [{ type: "text", text: sideChannelNoToolsReminder }],
attribution: "agent",
timestamp: Date.now(),
});
messages.push({
role: "user",
content: [{ type: "text", text: promptText }],
@@ -1,5 +1,5 @@
import { afterEach, describe, expect, it, vi } from "bun:test";
import { Agent, type AgentMessage, AppendOnlyContextManager } from "@oh-my-pi/pi-agent-core";
import { Agent, type AgentMessage, type AgentTool, AppendOnlyContextManager } from "@oh-my-pi/pi-agent-core";
import {
type Api,
type Context,
@@ -874,4 +874,135 @@ describe("AgentSession message pipeline", () => {
const occurrences = forkedPrompt.split(injected).length - 1;
expect(occurrences).toBe(1);
});
it("ephemeral side-channel forwards native tools, injects developer reminder, leaves toolChoice auto", async () => {
const api = "test-ephemeral-tools-warm-cache";
let capturedContext: Context | undefined;
let capturedOptions: SimpleStreamOptions | undefined;
registerCustomApi(api, (_model, context, options) => {
capturedContext = context;
capturedOptions = options;
const stream = new AssistantMessageEventStream();
queueMicrotask(() => {
const message = createAssistantMessage("Not using tools");
stream.push({ type: "text_delta", contentIndex: 0, delta: "Not using tools", partial: message });
stream.push({ type: "done", reason: "stop", message });
});
return stream;
});
const model = buildModel({
id: "side-model-with-tools",
name: "Side Model with Tools",
api,
provider: "test-provider",
baseUrl: "",
reasoning: false,
input: ["text"],
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
contextWindow: 4096,
maxTokens: 1024,
} as ModelSpec<Api>) as Model<Api>;
const tool: AgentTool = {
name: "side_tool",
label: "Side Tool",
description: "A tool in side channel",
parameters: { type: "object", properties: {} },
execute: async () => ({ content: [], details: {} }),
};
const session = new AgentSession({
agent: new Agent({
initialState: {
model,
systemPrompt: ["system prompt"],
messages: [],
tools: [tool],
},
}),
sessionManager: SessionManager.inMemory(),
settings: Settings.isolated({ "compaction.enabled": false }),
modelRegistry: createModelRegistryStub() as never,
});
sessions.push(session);
const result = await session.runEphemeralTurn({ promptText: "Side Question?" });
expect(result.replyText).toBe("Not using tools");
expect(capturedContext).toBeDefined();
expect(capturedContext!.tools).toBeDefined();
expect(capturedContext!.tools!.length).toBe(1);
expect(capturedContext!.tools![0].name).toBe("side_tool");
// Developer reminder injected immediately before user prompt
const messages = capturedContext!.messages;
expect(messages.length).toBeGreaterThanOrEqual(2);
const lastMessage = messages.at(-1);
const secondToLast = messages.at(-2);
expect(lastMessage?.role).toBe("user");
expect(getConvertedUserText(lastMessage)).toBe("Side Question?");
expect(secondToLast?.role).toBe("developer");
expect(secondToLast?.content).toBeDefined();
const textContent = secondToLast?.content as { text?: string }[];
expect(textContent[0].text).toContain("tool catalog stays attached");
// Tool choice must be undefined (not "none") for cache hits
expect(capturedOptions?.toolChoice).toBeUndefined();
});
it("ephemeral side-channel discards any emitted tool calls", async () => {
const api = "test-ephemeral-tools-discard";
registerCustomApi(api, (_model, _context, _options) => {
const stream = new AssistantMessageEventStream();
queueMicrotask(() => {
const message = createAssistantMessage("Here is text");
message.content.push({
type: "toolCall",
id: "call_123",
name: "side_tool",
arguments: {},
});
stream.push({ type: "text_delta", contentIndex: 0, delta: "Here is text", partial: message });
stream.push({ type: "done", reason: "stop", message });
});
return stream;
});
const model = buildModel({
id: "side-model-discard",
name: "Side Model Discard",
api,
provider: "test-provider",
baseUrl: "",
reasoning: false,
input: ["text"],
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
contextWindow: 4096,
maxTokens: 1024,
} as ModelSpec<Api>) as Model<Api>;
const session = new AgentSession({
agent: new Agent({
initialState: {
model,
systemPrompt: ["system prompt"],
messages: [],
tools: [],
},
}),
sessionManager: SessionManager.inMemory(),
settings: Settings.isolated({ "compaction.enabled": false }),
modelRegistry: createModelRegistryStub() as never,
});
sessions.push(session);
const result = await session.runEphemeralTurn({ promptText: "Side Question?" });
expect(result.replyText).toBe("Here is text");
expect(result.assistantMessage.content.some(block => block.type === "toolCall")).toBe(false);
expect(result.assistantMessage.content.every(block => block.type !== "toolCall")).toBe(true);
});
});