test(coding-agent): align retry rollover, flag table, and python detach tests with merged semantics

- retry-cap rollover repro no longer pins which sibling credential the
  session hash starts from; it asserts all four are rolled through
- STRING_VALUE_FLAGS table test accepts value-validating flags that
  reject the consumed token with CliUsageError (--max-time)
- python dispose-detach test follows the eval abort-shield contract:
  runs aborted mid-flight report cancelled even when the kernel races
  to completion
This commit is contained in:
can1357
2026-07-14 19:06:31 +02:00
parent cfb5d71c0a
commit 22374ab217
3 changed files with 23 additions and 9 deletions
@@ -493,9 +493,12 @@ describe("AgentSession python cleanup", () => {
expect(startSpy).toHaveBeenCalledTimes(1);
blockedExecution.resolve(OK_EXECUTION);
// The dispose abort was requested while the execution was still running;
// the executor reports such runs as cancelled even when the kernel races
// to completion. Detachment is proven below: the kernel survives, is not
// restarted, and keeps serving the surviving session.
await expect(firstExecution).resolves.toMatchObject({
cancelled: false,
exitCode: 0,
cancelled: true,
stdinRequested: false,
});
await secondSession.executePython("print('owner-b after detach')");
@@ -239,7 +239,10 @@ describe("AgentSession retry delay cap", () => {
throw new Error("Expected streamSimple to pass a resolved string API key");
}
requestedKeys.push(apiKey);
return apiKey === "anthropic-key-D"
// Succeed only once the fourth distinct sibling is attempted; the
// session-hash start index is arbitrary, so the repro must not pin
// which credential comes first — only that all four are rolled through.
return new Set(requestedKeys).size >= 4
? { content: ["recovered on fourth credential"], stopReason: "stop" }
: { throw: rateLimitError };
},
@@ -291,7 +294,7 @@ describe("AgentSession retry delay cap", () => {
await session.waitForIdle();
expect(requestedModels).toEqual([`${model.provider}/${model.id}`]);
expect(requestedKeys).toEqual(["anthropic-key-A", "anthropic-key-B", "anthropic-key-C", "anthropic-key-D"]);
expect(requestedKeys).toHaveLength(4);
expect(new Set(requestedKeys).size).toBe(4);
expect(mock.calls).toHaveLength(4);
expect(retryStartEvents).toHaveLength(0);
+13 -5
View File
@@ -1,6 +1,7 @@
import { describe, expect, it } from "bun:test";
import { parseArgs } from "../src/cli/args";
import { OPTIONAL_VALUE_FLAGS, STRING_VALUE_FLAGS } from "../src/cli/flag-tables";
import { CliUsageError } from "../src/cli/usage-error";
/**
* Catches the set → args.ts direction of drift between
@@ -27,11 +28,18 @@ import { OPTIONAL_VALUE_FLAGS, STRING_VALUE_FLAGS } from "../src/cli/flag-tables
describe("STRING_VALUE_FLAGS table is honored by args.ts parseArgs", () => {
for (const flag of STRING_VALUE_FLAGS) {
it(`${flag} consumes the next token unconditionally`, () => {
const result = parseArgs([flag, "--profile", "work"]);
expect(
result.profile,
`parseArgs should treat --profile as the value of ${flag}, not as a profile activation`,
).toBeUndefined();
try {
const result = parseArgs([flag, "--profile", "work"]);
expect(
result.profile,
`parseArgs should treat --profile as the value of ${flag}, not as a profile activation`,
).toBeUndefined();
} catch (error) {
// Value-validating flags (e.g. --max-time) reject "--profile" as their
// value; consuming-and-rejecting still proves the flag swallowed the
// token instead of activating the profile.
expect(error).toBeInstanceOf(CliUsageError);
}
});
}
});