fix(coding-agent): ensured side-channel turns obfuscate secrets and tools
- Verified that side-channel turns maintain stable prompt cache parity with main turns. - Applied secret and tool obfuscation to ephemeral assistant side-channel data to prevent leakage. - Updated `Agent` and `AgentSession` test suites to validate consistency under native dialect environments.
This commit is contained in:
@@ -1,9 +1,44 @@
|
||||
import { describe, expect, it, mock } from "bun:test";
|
||||
import { type Context, z } from "@oh-my-pi/pi-ai";
|
||||
import { type AssistantMessage, type Context, z } from "@oh-my-pi/pi-ai";
|
||||
import { createMockModel } from "@oh-my-pi/pi-ai/providers/mock";
|
||||
import { AssistantMessageEventStream } from "@oh-my-pi/pi-ai/utils/event-stream";
|
||||
import { Agent } from "../src/agent";
|
||||
import type { AgentTool } from "../src/types";
|
||||
|
||||
async function withNativeDialectEnv<T>(fn: () => T | Promise<T>): Promise<T> {
|
||||
const previous = Bun.env.PI_DIALECT;
|
||||
delete Bun.env.PI_DIALECT;
|
||||
try {
|
||||
return await fn();
|
||||
} finally {
|
||||
if (previous === undefined) {
|
||||
delete Bun.env.PI_DIALECT;
|
||||
} else {
|
||||
Bun.env.PI_DIALECT = previous;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
function testAssistantMessage(text: string): AssistantMessage {
|
||||
return {
|
||||
role: "assistant",
|
||||
content: [{ type: "text", text }],
|
||||
api: "mock",
|
||||
provider: "mock",
|
||||
model: "mock",
|
||||
usage: {
|
||||
input: 0,
|
||||
output: 0,
|
||||
cacheRead: 0,
|
||||
cacheWrite: 0,
|
||||
totalTokens: 0,
|
||||
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 },
|
||||
},
|
||||
stopReason: "stop",
|
||||
timestamp: Date.now(),
|
||||
};
|
||||
}
|
||||
|
||||
describe("Agent — buildSideRequestContext", () => {
|
||||
const model = createMockModel({ responses: [] });
|
||||
const tool: AgentTool = {
|
||||
@@ -14,23 +49,64 @@ describe("Agent — buildSideRequestContext", () => {
|
||||
execute: async () => ({ content: [{ type: "text", text: "success" }], details: { value: "success" } }),
|
||||
};
|
||||
|
||||
it("forwards the tool catalog for native providers", () => {
|
||||
const agent = new Agent({
|
||||
initialState: {
|
||||
model,
|
||||
systemPrompt: ["system"],
|
||||
tools: [tool],
|
||||
},
|
||||
it("forwards the tool catalog for native providers", async () => {
|
||||
await withNativeDialectEnv(() => {
|
||||
const agent = new Agent({
|
||||
initialState: {
|
||||
model,
|
||||
systemPrompt: ["system"],
|
||||
tools: [tool],
|
||||
},
|
||||
});
|
||||
|
||||
const context = agent.buildSideRequestContext([
|
||||
{ role: "user", content: [{ type: "text", text: "Q?" }], timestamp: Date.now() },
|
||||
]);
|
||||
|
||||
expect(context.tools).toBeDefined();
|
||||
expect(context.tools!.length).toBe(1);
|
||||
expect(context.tools![0].name).toBe("test_tool");
|
||||
expect(context.systemPrompt).toEqual(["system"]);
|
||||
});
|
||||
});
|
||||
|
||||
const context = agent.buildSideRequestContext([
|
||||
{ role: "user", content: [{ type: "text", text: "Q?" }], timestamp: Date.now() },
|
||||
]);
|
||||
it("matches the main loop's native stable prefix", async () => {
|
||||
await withNativeDialectEnv(async () => {
|
||||
let mainContext: Context | undefined;
|
||||
const agent = new Agent({
|
||||
initialState: {
|
||||
model,
|
||||
systemPrompt: ["system"],
|
||||
tools: [tool],
|
||||
},
|
||||
streamFn: (_model, context) => {
|
||||
mainContext = context;
|
||||
const stream = new AssistantMessageEventStream();
|
||||
queueMicrotask(() => {
|
||||
const message = testAssistantMessage("ok");
|
||||
stream.push({ type: "text_delta", contentIndex: 0, delta: "ok", partial: message });
|
||||
stream.push({ type: "done", reason: "stop", message });
|
||||
});
|
||||
return stream;
|
||||
},
|
||||
});
|
||||
|
||||
expect(context.tools).toBeDefined();
|
||||
expect(context.tools!.length).toBe(1);
|
||||
expect(context.tools![0].name).toBe("test_tool");
|
||||
expect(context.systemPrompt).toEqual(["system"]);
|
||||
await agent.prompt("Q?");
|
||||
|
||||
const sideAgent = new Agent({
|
||||
initialState: {
|
||||
model,
|
||||
systemPrompt: ["system"],
|
||||
tools: [tool],
|
||||
},
|
||||
});
|
||||
const sideContext = sideAgent.buildSideRequestContext([
|
||||
{ role: "user", content: [{ type: "text", text: "Q?" }], timestamp: Date.now() },
|
||||
]);
|
||||
|
||||
expect(JSON.stringify(sideContext.systemPrompt)).toBe(JSON.stringify(mainContext?.systemPrompt));
|
||||
expect(JSON.stringify(sideContext.tools)).toBe(JSON.stringify(mainContext?.tools));
|
||||
});
|
||||
});
|
||||
|
||||
it("returns empty tools when owned dialect is active", () => {
|
||||
|
||||
@@ -1,6 +1,7 @@
|
||||
# Changelog
|
||||
|
||||
## [Unreleased]
|
||||
|
||||
### Added
|
||||
|
||||
- Added "Prose Only Thinking" setting to opt-out of rendering code blocks within AI thinking traces
|
||||
@@ -17,6 +18,7 @@
|
||||
|
||||
### Fixed
|
||||
|
||||
- Fixed side-channel turns failing to correctly obfuscate secrets and tools when inheriting prompt cache layout
|
||||
- Fixed session history becoming desynchronized when using the `rewind` tool
|
||||
- Fixed `rewind` tool output and temporary assistant side-channel data polluting the prompt cache
|
||||
- Fixed `read` against a SQLite table with many columns (e.g. 33) rendering every cell as an ellipsis and chopping the right edge. The ASCII table shrinker bottomed out at `MIN_COLUMN_WIDTH=1` so every multi-char cell collapsed to `…`, and the final line truncation then cut off the right side. The renderer now bumps the per-column floor to 3 and falls back to a vertical `column: value` block layout per row when the column count exceeds the horizontal width budget. ([#3107](https://github.com/can1357/oh-my-pi/issues/3107))
|
||||
@@ -33,9 +35,6 @@
|
||||
- Fixed legacy Pi extensions importing `getModel`/`getModels` from the `@oh-my-pi/pi-ai` package root failing to load, by restoring them as compatibility aliases for `@oh-my-pi/pi-catalog`'s `getBundledModel`/`getBundledModels`. The root `StringEnum` compatibility shim also accepts TypeScript enum objects in addition to value arrays ([#2907](https://github.com/can1357/oh-my-pi/pull/2907)).
|
||||
- Fixed the Alt+M model-configuration menu so role assignment remains selectable for models whose context window is smaller than the current session; the context-size disabling still applies to the Alt+P temporary active-model switch ([#2861](https://github.com/can1357/oh-my-pi/issues/2861)).
|
||||
- Fixed the advisor raising false `blocker`s in plan mode (e.g. "don't write a plan file") because it only saw a 120-char truncation of the injected plan-mode rules, which cut off at `NEVER create, edit, or delete files — excep…` and hid the "except the single plan file" carve-out. The advisor delta now expands the primary agent's constraint context (`plan-mode-context`, `plan-mode-reference`) verbatim inside an XML-escaped `<primary-context>` wrapper instead of a one-liner, and `AdvisorRuntime` dedupes the re-injected prompts so an unchanged copy collapses to a marker rather than re-feeding the full rules every turn.
|
||||
|
||||
### Fixed
|
||||
|
||||
- Fixed auto-compaction being suppressed when a `before_provider_request` extension shrinks the outgoing request below the real stored conversation (e.g. a context-compression proxy such as Headroom, or an aggressive obfuscator). The provider then reports deflated prompt tokens, so the threshold check never fired and the stored history grew unbounded until it overflowed the context window and could no longer be compacted at all. The compaction decision (both the pre-prompt and post-response paths) now floors the provider-reported context tokens by the agent's own local estimate of the stored conversation, so on-wire compression can no longer hide a too-large history from the auto-compactor. Context display and cost accounting still use the exact provider usage; only the compaction trigger takes the floor.
|
||||
|
||||
## [16.1.7] - 2026-06-20
|
||||
|
||||
@@ -18,7 +18,7 @@ import { Settings } from "@oh-my-pi/pi-coding-agent/config/settings";
|
||||
import * as memoryBackend from "@oh-my-pi/pi-coding-agent/memory-backend";
|
||||
import type { MemoryBackend } from "@oh-my-pi/pi-coding-agent/memory-backend/types";
|
||||
import { type MnemopiSessionState, setMnemopiSessionState } from "@oh-my-pi/pi-coding-agent/mnemopi/state";
|
||||
import { SecretObfuscator } from "@oh-my-pi/pi-coding-agent/secrets";
|
||||
import { obfuscateProviderContext, SecretObfuscator } from "@oh-my-pi/pi-coding-agent/secrets";
|
||||
import { AgentSession, type AgentSessionEvent } from "@oh-my-pi/pi-coding-agent/session/agent-session";
|
||||
import { convertToLlm, wrapSteeringForModel } from "@oh-my-pi/pi-coding-agent/session/messages";
|
||||
import { SessionManager } from "@oh-my-pi/pi-coding-agent/session/session-manager";
|
||||
@@ -56,6 +56,20 @@ function getConvertedUserText(message: Message | undefined): string {
|
||||
return text.text;
|
||||
}
|
||||
|
||||
async function withNativeDialectEnv<T>(fn: () => Promise<T>): Promise<T> {
|
||||
const previous = Bun.env.PI_DIALECT;
|
||||
delete Bun.env.PI_DIALECT;
|
||||
try {
|
||||
return await fn();
|
||||
} finally {
|
||||
if (previous === undefined) {
|
||||
delete Bun.env.PI_DIALECT;
|
||||
} else {
|
||||
Bun.env.PI_DIALECT = previous;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
describe("AgentSession message pipeline", () => {
|
||||
const sessions: AgentSession[] = [];
|
||||
|
||||
@@ -432,6 +446,83 @@ describe("AgentSession message pipeline", () => {
|
||||
expect(JSON.stringify(capturedContext)).not.toContain(secret);
|
||||
});
|
||||
|
||||
it("keeps obfuscated side-channel stable prefix byte-identical to the main turn", async () => {
|
||||
await withNativeDialectEnv(async () => {
|
||||
const api = "test-ephemeral-obfuscated-prefix-parity";
|
||||
const secret = "PREFIX_SECRET_TOKEN_12345";
|
||||
let callCount = 0;
|
||||
let mainContext: Context | undefined;
|
||||
let sideContext: Context | undefined;
|
||||
registerCustomApi(api, (_model, context, _options) => {
|
||||
if (callCount === 0) {
|
||||
mainContext = context;
|
||||
} else {
|
||||
sideContext = context;
|
||||
}
|
||||
callCount += 1;
|
||||
const stream = new AssistantMessageEventStream();
|
||||
queueMicrotask(() => {
|
||||
const message = createAssistantMessage("Answer");
|
||||
stream.push({ type: "text_delta", contentIndex: 0, delta: "Answer", partial: message });
|
||||
stream.push({ type: "done", reason: "stop", message });
|
||||
});
|
||||
return stream;
|
||||
});
|
||||
|
||||
const model = buildModel({
|
||||
id: "side-model-prefix-parity",
|
||||
name: "Side Model Prefix Parity",
|
||||
api,
|
||||
provider: "test-provider",
|
||||
baseUrl: "",
|
||||
reasoning: false,
|
||||
input: ["text"],
|
||||
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
|
||||
contextWindow: 4096,
|
||||
maxTokens: 1024,
|
||||
} as ModelSpec<Api>) as Model<Api>;
|
||||
const obfuscator = new SecretObfuscator([{ type: "plain", content: secret }]);
|
||||
const tool: AgentTool = {
|
||||
name: "secret_probe",
|
||||
label: "Secret Probe",
|
||||
description: `Tool description ${secret}`,
|
||||
parameters: {
|
||||
type: "object",
|
||||
properties: {
|
||||
value: { type: "string", description: `Schema description ${secret}` },
|
||||
},
|
||||
required: ["value"],
|
||||
},
|
||||
execute: async () => ({ content: [], details: {} }),
|
||||
};
|
||||
const agent = new Agent({
|
||||
initialState: {
|
||||
model,
|
||||
systemPrompt: [`system prompt with ${secret}`],
|
||||
messages: [],
|
||||
tools: [tool],
|
||||
},
|
||||
transformProviderContext: context => obfuscateProviderContext(obfuscator, context),
|
||||
});
|
||||
const session = new AgentSession({
|
||||
agent,
|
||||
sessionManager: SessionManager.inMemory(),
|
||||
settings: Settings.isolated({ "compaction.enabled": false }),
|
||||
modelRegistry: createModelRegistryStub() as never,
|
||||
obfuscator,
|
||||
});
|
||||
sessions.push(session);
|
||||
|
||||
await agent.prompt("Main Question?");
|
||||
await session.runEphemeralTurn({ promptText: `Side Question ${secret}?` });
|
||||
|
||||
expect(JSON.stringify(mainContext?.systemPrompt)).toBe(JSON.stringify(sideContext?.systemPrompt));
|
||||
expect(JSON.stringify(mainContext?.tools)).toBe(JSON.stringify(sideContext?.tools));
|
||||
expect(JSON.stringify(sideContext?.systemPrompt)).not.toContain(secret);
|
||||
expect(JSON.stringify(sideContext?.tools)).not.toContain(secret);
|
||||
});
|
||||
});
|
||||
|
||||
it("records raw SSE diagnostics into the session buffer before request hooks", async () => {
|
||||
const requestOnSseEvent = vi.fn();
|
||||
const session = new AgentSession({
|
||||
@@ -876,81 +967,83 @@ describe("AgentSession message pipeline", () => {
|
||||
});
|
||||
|
||||
it("ephemeral side-channel forwards native tools, injects developer reminder, leaves toolChoice auto", async () => {
|
||||
const api = "test-ephemeral-tools-warm-cache";
|
||||
let capturedContext: Context | undefined;
|
||||
let capturedOptions: SimpleStreamOptions | undefined;
|
||||
registerCustomApi(api, (_model, context, options) => {
|
||||
capturedContext = context;
|
||||
capturedOptions = options;
|
||||
const stream = new AssistantMessageEventStream();
|
||||
queueMicrotask(() => {
|
||||
const message = createAssistantMessage("Not using tools");
|
||||
stream.push({ type: "text_delta", contentIndex: 0, delta: "Not using tools", partial: message });
|
||||
stream.push({ type: "done", reason: "stop", message });
|
||||
await withNativeDialectEnv(async () => {
|
||||
const api = "test-ephemeral-tools-warm-cache";
|
||||
let capturedContext: Context | undefined;
|
||||
let capturedOptions: SimpleStreamOptions | undefined;
|
||||
registerCustomApi(api, (_model, context, options) => {
|
||||
capturedContext = context;
|
||||
capturedOptions = options;
|
||||
const stream = new AssistantMessageEventStream();
|
||||
queueMicrotask(() => {
|
||||
const message = createAssistantMessage("Not using tools");
|
||||
stream.push({ type: "text_delta", contentIndex: 0, delta: "Not using tools", partial: message });
|
||||
stream.push({ type: "done", reason: "stop", message });
|
||||
});
|
||||
return stream;
|
||||
});
|
||||
return stream;
|
||||
|
||||
const model = buildModel({
|
||||
id: "side-model-with-tools",
|
||||
name: "Side Model with Tools",
|
||||
api,
|
||||
provider: "test-provider",
|
||||
baseUrl: "",
|
||||
reasoning: false,
|
||||
input: ["text"],
|
||||
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
|
||||
contextWindow: 4096,
|
||||
maxTokens: 1024,
|
||||
} as ModelSpec<Api>) as Model<Api>;
|
||||
|
||||
const tool: AgentTool = {
|
||||
name: "side_tool",
|
||||
label: "Side Tool",
|
||||
description: "A tool in side channel",
|
||||
parameters: { type: "object", properties: {} },
|
||||
execute: async () => ({ content: [], details: {} }),
|
||||
};
|
||||
|
||||
const session = new AgentSession({
|
||||
agent: new Agent({
|
||||
initialState: {
|
||||
model,
|
||||
systemPrompt: ["system prompt"],
|
||||
messages: [],
|
||||
tools: [tool],
|
||||
},
|
||||
}),
|
||||
sessionManager: SessionManager.inMemory(),
|
||||
settings: Settings.isolated({ "compaction.enabled": false }),
|
||||
modelRegistry: createModelRegistryStub() as never,
|
||||
});
|
||||
sessions.push(session);
|
||||
|
||||
const result = await session.runEphemeralTurn({ promptText: "Side Question?" });
|
||||
|
||||
expect(result.replyText).toBe("Not using tools");
|
||||
expect(capturedContext).toBeDefined();
|
||||
expect(capturedContext!.tools).toBeDefined();
|
||||
expect(capturedContext!.tools!.length).toBe(1);
|
||||
expect(capturedContext!.tools![0].name).toBe("side_tool");
|
||||
|
||||
// Developer reminder injected immediately before user prompt
|
||||
const messages = capturedContext!.messages;
|
||||
expect(messages.length).toBeGreaterThanOrEqual(2);
|
||||
const lastMessage = messages.at(-1);
|
||||
const secondToLast = messages.at(-2);
|
||||
|
||||
expect(lastMessage?.role).toBe("user");
|
||||
expect(getConvertedUserText(lastMessage)).toBe("Side Question?");
|
||||
|
||||
expect(secondToLast?.role).toBe("developer");
|
||||
expect(secondToLast?.content).toBeDefined();
|
||||
const textContent = secondToLast?.content as { text?: string }[];
|
||||
expect(textContent[0].text).toContain("tool catalog stays attached");
|
||||
|
||||
// Tool choice must be undefined (not "none") for cache hits
|
||||
expect(capturedOptions?.toolChoice).toBeUndefined();
|
||||
});
|
||||
|
||||
const model = buildModel({
|
||||
id: "side-model-with-tools",
|
||||
name: "Side Model with Tools",
|
||||
api,
|
||||
provider: "test-provider",
|
||||
baseUrl: "",
|
||||
reasoning: false,
|
||||
input: ["text"],
|
||||
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
|
||||
contextWindow: 4096,
|
||||
maxTokens: 1024,
|
||||
} as ModelSpec<Api>) as Model<Api>;
|
||||
|
||||
const tool: AgentTool = {
|
||||
name: "side_tool",
|
||||
label: "Side Tool",
|
||||
description: "A tool in side channel",
|
||||
parameters: { type: "object", properties: {} },
|
||||
execute: async () => ({ content: [], details: {} }),
|
||||
};
|
||||
|
||||
const session = new AgentSession({
|
||||
agent: new Agent({
|
||||
initialState: {
|
||||
model,
|
||||
systemPrompt: ["system prompt"],
|
||||
messages: [],
|
||||
tools: [tool],
|
||||
},
|
||||
}),
|
||||
sessionManager: SessionManager.inMemory(),
|
||||
settings: Settings.isolated({ "compaction.enabled": false }),
|
||||
modelRegistry: createModelRegistryStub() as never,
|
||||
});
|
||||
sessions.push(session);
|
||||
|
||||
const result = await session.runEphemeralTurn({ promptText: "Side Question?" });
|
||||
|
||||
expect(result.replyText).toBe("Not using tools");
|
||||
expect(capturedContext).toBeDefined();
|
||||
expect(capturedContext!.tools).toBeDefined();
|
||||
expect(capturedContext!.tools!.length).toBe(1);
|
||||
expect(capturedContext!.tools![0].name).toBe("side_tool");
|
||||
|
||||
// Developer reminder injected immediately before user prompt
|
||||
const messages = capturedContext!.messages;
|
||||
expect(messages.length).toBeGreaterThanOrEqual(2);
|
||||
const lastMessage = messages.at(-1);
|
||||
const secondToLast = messages.at(-2);
|
||||
|
||||
expect(lastMessage?.role).toBe("user");
|
||||
expect(getConvertedUserText(lastMessage)).toBe("Side Question?");
|
||||
|
||||
expect(secondToLast?.role).toBe("developer");
|
||||
expect(secondToLast?.content).toBeDefined();
|
||||
const textContent = secondToLast?.content as { text?: string }[];
|
||||
expect(textContent[0].text).toContain("tool catalog stays attached");
|
||||
|
||||
// Tool choice must be undefined (not "none") for cache hits
|
||||
expect(capturedOptions?.toolChoice).toBeUndefined();
|
||||
});
|
||||
|
||||
it("ephemeral side-channel discards any emitted tool calls", async () => {
|
||||
|
||||
Reference in New Issue
Block a user