fix(tts): cap replay retention so long utterances stay bounded in memory

Streamed PCM kept for the nonzero-exit replay is now dropped once the
utterance exceeds 60s (~5.8 MB at 24 kHz mono f32): the recovered failure
is a short clip that fits the pipe buffer before a broken backend dies,
while replaying a long already-played utterance would duplicate audio and
unbounded retention would defeat streaming for long input.
This commit is contained in:
can1357
2026-07-23 22:05:04 +02:00
parent 46ae99cea8
commit eaa3002b9d
2 changed files with 38 additions and 1 deletions
@@ -51,4 +51,22 @@ describe("StreamingAudioPlayer nonzero-exit fallback", () => {
await player.end();
expect(played.length).toBe(2);
});
it("drops the replay buffer once the utterance exceeds the retention cap", async () => {
// Long input must not accumulate unbounded PCM; past the cap the
// nonzero-exit replay is forfeited rather than duplicating audio the
// backend already played.
const played: string[] = [];
const player = new StreamingAudioPlayer({
commandsFor: (): PlayerCommand[] => [{ cmd: "sh", args: ["-c", "cat >/dev/null; exit 1"] }],
playAudio: async wavPath => {
played.push(wavPath);
},
replayRetentionSeconds: 0.25,
});
player.start(24_000);
player.write(clip());
await player.end();
expect(played.length).toBe(0);
});
});