Files
oh-my-pi/packages/coding-agent/test/tts/speakable.test.ts
T
can1357 94645752ff feat(coding-agent/tts): introduced natural speech synthesis pipeline
- Added an enhanced speech pipeline that utilizes small models to rewrite text for natural language synthesis.
- Implemented `SpeechEnhancer` and `BlockAccumulator` to manage fence-aware text streaming and paragraph splitting.
- Configured a new `speech.enhanced` setting to toggle between mechanical and enhanced vocalization modes.
- Resolved `EPIPE` rejections during speech playback by ensuring stream flushes and suppressing stop-related errors.
2026-07-02 08:59:13 +02:00

172 lines
6.2 KiB
TypeScript

import { describe, expect, it } from "bun:test";
import { SpeakableStream } from "@oh-my-pi/pi-coding-agent/tts/speakable";
/** Push each delta in order, then flush; returns per-push segments plus the flush tail. */
function speak(...deltas: string[]): { pushed: string[][]; all: string[] } {
const stream = new SpeakableStream();
const pushed = deltas.map(delta => stream.push(delta));
const flushed = stream.flush();
return { pushed, all: [...pushed.flat(), ...flushed] };
}
describe("SpeakableStream code fences", () => {
it("silences a fenced block while speaking the prose around it", () => {
const { all } = speak(
"Here is the code:\n```ts\nconst x = 1;\nconsole.log(x);\n```\nAnd that is all of it now.\n",
);
expect(all).toEqual(["Here is the code:", "And that is all of it now."]);
});
it("silences a block whose opening and closing fences are split across deltas", () => {
const { pushed, all } = speak(
"Intro line here.\n``",
"`ts\nconst hidden = 42;\n``",
"`\nOutro after code block.\n",
);
expect(pushed[0]).toEqual(["Intro line here."]);
expect(pushed[1]).toEqual([]);
expect(all).toEqual(["Intro line here.", "Outro after code block."]);
});
});
describe("SpeakableStream tables", () => {
it("silences | rows while speaking surrounding prose", () => {
const { all } = speak("Results below.\n| a | b |\n| --- | --- |\n| 1 | 2 |\nDone with the table now.\n");
expect(all).toEqual(["Results below.", "Done with the table now."]);
});
});
describe("SpeakableStream links and URLs", () => {
it("speaks only the label of a markdown link split across deltas, with no early mid-link cut", () => {
const { pushed, all } = speak("See [the do", "cs](https://exam", "ple.com/path) for details.\n");
expect(pushed[0]).toEqual([]);
expect(pushed[1]).toEqual([]);
expect(all).toEqual(["See the docs for details."]);
});
it.each([
[
"bare https URL speaks only the host",
"Repo lives at https://github.com/foo/bar?x for now.\n",
"Repo lives at github.com for now.",
],
[
"www URL speaks the host without the www prefix or path",
"Visit www.example.com/path when you can.\n",
"Visit example.com when you can.",
],
])("%s", (_name, input, spoken) => {
expect(speak(input).all).toEqual([spoken]);
});
});
describe("SpeakableStream inline markup and line markers", () => {
it.each([
[
"inline code speaks the identifier without ticks",
"Call `parseConfig` before use today.\n",
["Call parseConfig before use today."],
],
[
"bold, italic, and strikethrough markers are stripped",
"This **bold** and *ital* and ~~struck~~ text stays.\n",
["This bold and ital and struck text stays."],
],
[
"heading markers are stripped but the title speaks",
"## Release Notes\nBody text follows here.\n",
["Release Notes", "Body text follows here."],
],
[
"bullet markers are stripped",
"- item one is ready\n- item two is ready\n",
["item one is ready", "item two is ready"],
],
["numbered list markers speak as a numeric prefix", "1. First\n", ["1, First"]],
])("%s", (_name, input, spoken) => {
expect(speak(input).all).toEqual(spoken);
});
});
describe("SpeakableStream file paths", () => {
it("collapses a multi-directory path to its basename", () => {
const { all } = speak("Edit packages/coding-agent/src/tts/vocalizer.ts to fix it.\n");
expect(all).toEqual(["Edit vocalizer.ts to fix it."]);
});
it("leaves two-component tokens like and/or untouched", () => {
const { all } = speak("Use and/or as needed today.\n");
expect(all).toEqual(["Use and/or as needed today."]);
});
});
describe("SpeakableStream streaming latency", () => {
it("emits a completed sentence from push() itself, without waiting for the next sentence", () => {
const stream = new SpeakableStream();
expect(stream.push("First sentence is long enough here. ")).toEqual(["First sentence is long enough here."]);
expect(stream.flush()).toEqual([]);
});
});
describe("SpeakableStream silent replies", () => {
it("yields zero segments for a reply that is only markup, whitespace, and label-less link markup", () => {
const stream = new SpeakableStream();
expect(stream.push("---\n\n \n**\n![](https://x.com/a.png)\n```\nlet a = 1;\n```\n| a |\n")).toEqual([]);
expect(stream.flush()).toEqual([]);
});
});
describe("SpeakableStream flush and flushIdle", () => {
it("flush() drains a trailing partial sentence that push() held back", () => {
const stream = new SpeakableStream();
expect(stream.push("This is a trailing partial")).toEqual([]);
expect(stream.flush()).toEqual(["This is a trailing partial"]);
});
it("flushIdle() refuses a short mid-sentence fragment, which a later flush() still drains", () => {
const stream = new SpeakableStream();
expect(stream.push("The")).toEqual([]);
expect(stream.flushIdle()).toEqual([]);
expect(stream.flush()).toEqual(["The"]);
});
it("flushIdle() drains a short but complete thought", () => {
const stream = new SpeakableStream();
expect(stream.push("Done here now.")).toEqual([]);
expect(stream.flushIdle()).toEqual(["Done here now."]);
expect(stream.flush()).toEqual([]);
});
});
describe("SpeakableStream segment length cap", () => {
it("force-splits an unpunctuated 1000+ char run into <=280-char segments that preserve every word", () => {
const run = Array.from({ length: 200 }, (_, i) => `word${i}`).join(" ");
expect(run.length).toBeGreaterThan(1000);
const { all } = speak(run);
expect(all.length).toBeGreaterThan(1);
for (const segment of all) expect(segment.length).toBeLessThanOrEqual(280);
expect(all.join(" ")).toBe(run);
});
});
describe("SpeakableStream abbreviations", () => {
it.each([
[
'"e.g. " near the start does not end the first segment',
"See e.g. the docs for more. ",
"See e.g. the docs for more.",
],
// The abbreviation here sits past the first-segment minimum, so only the
// abbreviation guard (not the length floor) prevents a cut after "e.g. ".
[
'"e.g. " past the minimum cut length still does not split the sentence',
"See the docs e.g. the guide for more info. ",
"See the docs e.g. the guide for more info.",
],
])("%s", (_name, input, spoken) => {
const stream = new SpeakableStream();
expect(stream.push(input)).toEqual([spoken]);
expect(stream.flush()).toEqual([]);
});
});