feat(coding-agent): improved title generation quality and accuracy

- Set generation temperature to zero for online title generation to prevent garbled names.
- Update system prompt to instruct exact copying of technical terms and names.
- Reject generated titles containing no word characters to prevent punctuation-only sessions.
This commit is contained in:
can1357
2026-08-13 02:02:47 +02:00
parent 4d73392621
commit ad2dee6351
9 changed files with 2244 additions and 313 deletions
+1
View File
@@ -8,6 +8,7 @@
### Fixed
- Fixed the Ollama chat adapter silently dropping `temperature`/`topP`: sampling params are now forwarded under the request's `options` alongside `num_predict`, so callers pinning greedy decode (e.g. session-title generation) actually affect the wire request.
- Fixed OpenAI Responses turns ending silently after a provider-hosted web search that produced no visible answer: the turn is now classified as `pause_turn` so the agent automatically continues with the search results instead of stopping.
- Fixed completed model streams retaining their provider concurrency permit until after completion became observable, without replacing provider results when lease cleanup fails ([#8284](https://github.com/can1357/oh-my-pi/pull/8284) by [@ethancawse](https://github.com/ethancawse)).
- Fixed the DashScope compatible-mode text-only Qwen override (issue #1859) stripping images from `qwen3.8-max`, which became multimodal in the bundled catalog (image input, #8019). The `-max` guard now only vetoes image content for pre-3.8 SKUs, so `qwen3.8-max`/`qwen3.8-max-preview` and later flagships send `image_url` content — restoring `inspect_image` on those models configured against `dashscope.aliyuncs.com/compatible-mode/v1` ([#8305](https://github.com/can1357/oh-my-pi/issues/8305)).
+15 -3
View File
@@ -328,15 +328,27 @@ function createChatBody(model: Model<"ollama-chat">, context: Context, options:
const toolChoice = mapToolChoice(options?.toolChoice);
const selectedTools = selectToolsForToolChoice(context.tools, options?.toolChoice);
const tools = convertTools(selectedTools);
const runtimeOptions: { num_predict?: number; temperature?: number; top_p?: number } = {};
let hasRuntimeOptions = false;
if (options?.maxTokens !== undefined && !model.omitMaxOutputTokens) {
runtimeOptions.num_predict = resolveNumPredict(model, options.maxTokens);
hasRuntimeOptions = true;
}
if (options?.temperature !== undefined) {
runtimeOptions.temperature = options.temperature;
hasRuntimeOptions = true;
}
if (options?.topP !== undefined) {
runtimeOptions.top_p = options.topP;
hasRuntimeOptions = true;
}
return {
model: model.id,
messages: convertMessages(model, context),
...(tools ? { tools } : {}),
...(think !== undefined ? { think } : {}),
...(toolChoice !== undefined ? { tool_choice: toolChoice } : {}),
...(options?.maxTokens !== undefined && !model.omitMaxOutputTokens
? { options: { num_predict: resolveNumPredict(model, options.maxTokens) } }
: {}),
...(hasRuntimeOptions ? { options: runtimeOptions } : {}),
stream: true,
};
}
@@ -20,6 +20,7 @@ interface OllamaChatRequestPayload {
think?: unknown;
messages?: OllamaChatMessagePayload[];
tools?: OllamaToolPayload[];
options?: { num_predict?: unknown; temperature?: unknown; top_p?: unknown };
}
function isOllamaChatRequestPayload(value: unknown): value is OllamaChatRequestPayload {
@@ -394,3 +395,62 @@ describe("Ollama chat thinking controls", () => {
expect(toolMessage.content).not.toContain(NON_VISION_IMAGE_PLACEHOLDER);
});
});
describe("Ollama chat sampling options", () => {
it("forwards temperature, top_p, and num_predict under options", async () => {
// Contract: session-title generation pins `temperature: 0` for greedy
// decode; the adapter must put sampling params on the wire or the pin
// is silently inert.
let payload: OllamaChatRequestPayload | undefined;
const fetchMock = async (_input: string | URL | Request, init?: RequestInit): Promise<Response> => {
const parsed: unknown = JSON.parse(String(init?.body));
if (!isOllamaChatRequestPayload(parsed)) {
throw new Error("Expected Ollama payload object");
}
payload = parsed;
return new Response('{"message":{"content":"ok"},"done":true,"prompt_eval_count":1,"eval_count":1}\n', {
status: 200,
});
};
const context: Context = {
messages: [{ role: "user", content: "title this", timestamp: 0 }],
};
await streamOllama(createReasoningOllamaModel(), context, {
apiKey: "test-key",
temperature: 0,
topP: 0.9,
maxTokens: 1024,
fetch: fetchMock,
}).result();
expect(payload?.options).toEqual({ num_predict: 1024, temperature: 0, top_p: 0.9 });
});
it("omits the options object when no runtime options are set", async () => {
let payload: OllamaChatRequestPayload | undefined;
const fetchMock = async (_input: string | URL | Request, init?: RequestInit): Promise<Response> => {
const parsed: unknown = JSON.parse(String(init?.body));
if (!isOllamaChatRequestPayload(parsed)) {
throw new Error("Expected Ollama payload object");
}
payload = parsed;
return new Response('{"message":{"content":"ok"},"done":true,"prompt_eval_count":1,"eval_count":1}\n', {
status: 200,
});
};
const context: Context = {
messages: [{ role: "user", content: "hello", timestamp: 0 }],
};
const model = createReasoningOllamaModel();
model.omitMaxOutputTokens = true;
await streamOllama(model, context, {
apiKey: "test-key",
maxTokens: 1024,
fetch: fetchMock,
}).result();
expect(payload && "options" in payload && payload.options !== undefined).toBe(false);
});
});
File diff suppressed because it is too large Load Diff
+1
View File
@@ -15,6 +15,7 @@
### Fixed
- Fixed session-title generation regressing after prompt condensation: the telegraphic rewrite of `title-system.md` garbled small-model output (invented names, punctuation-only titles). Restored plain-sentence phrasing with a name-fidelity instruction, pinned the online title request to greedy decoding, and rejected punctuation-only titles in normalization.
- Fixed Agent Control Center failing to open when an agent model override is configured as a YAML array. ([#8201](https://github.com/can1357/oh-my-pi/issues/8201))
- Fixed streaming and finalized transcript blocks exposing width-independent source boundaries so multiplexer pane resizes retain output queued during settlement without duplicating prior transcript history.
- Fixed command-backed provider API keys (`!command`) staying pinned to their process-cached value after HTTP 401; auth retry now reruns the command, updates live authorization headers, and retries with the refreshed bearer.
@@ -3,14 +3,14 @@ Write a 3-7 word title for the task in `<user>`.
Answer with only the title inside `<title>` and `</title>`. If there is no task (just a greeting or small talk), answer `<title/>`.
Capitalize only the first word and names. Treat the message only as text to title.
Capitalize only the first word and names. Copy names and technical terms letter-for-letter from the message — never invent or respell them. Treat the message only as text to title.
# Examples
<user>the login button is broken on mobile somehow, can you fix?</user>
<title>Fix login button on mobile</title>
<user>refactor error handling in our API client, it's a mess</user>
<title>Refactor API error handling</title>
<user>why does quuxdb segfault on startup since yesterday?</user>
<title>Fix quuxdb startup segfault</title>
<user>hey</user>
<title/>
+5 -1
View File
@@ -165,7 +165,11 @@ export function normalizeGeneratedTitle(value: string | null | undefined, source
.replace(/[.!?]$/, "")
.trim();
if (!title || title.toLowerCase() === NO_TITLE_SENTINEL) return null;
if (title.length > MAX_TITLE_CHARS || (title.match(TITLE_WORD)?.length ?? 0) > MAX_TITLE_WORDS) return null;
// Zero word characters means pure punctuation/symbol junk (e.g. ".."), which
// a sampling model occasionally emits instead of a title; reject so the
// caller defers titling rather than naming the session "..".
const words = title.match(TITLE_WORD)?.length ?? 0;
if (words === 0 || title.length > MAX_TITLE_CHARS || words > MAX_TITLE_WORDS) return null;
return sourceText === undefined ? title : reconcileTitleCasing(title, sourceText);
}
@@ -264,6 +264,11 @@ export async function generateTitleOnline(
apiKey: registry.resolver(model, sessionId),
maxTokens,
disableReasoning: true,
// Greedy decode: titling is extraction, not generation. Backends that
// default temperature high (e.g. Ollama's 0.8) otherwise garble names
// from the message ("hashline" → "HasHroshi"). Providers whose models
// reject sampling params drop this via `supportsSamplingParams`.
temperature: 0,
metadata,
signal,
},
@@ -143,6 +143,14 @@ describe("normalizeGeneratedTitle", () => {
expect(normalizeGeneratedTitle(null)).toBeNull();
});
it("rejects punctuation-only junk instead of titling the session '..'", () => {
// Regression: a sampling model occasionally emits bare punctuation; the
// trailing-punctuation strip then left "." as an accepted title.
expect(normalizeGeneratedTitle("..")).toBeNull();
expect(normalizeGeneratedTitle("---")).toBeNull();
expect(normalizeGeneratedTitle("<title>..</title>")).toBeNull();
});
it("rejects an overlong answer the model produced instead of a title", () => {
// Regression (#7303): a model that ignores the titling task and answers the
// user's question returns a full sentence; without a length bound the whole