Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
4 changes: 2 additions & 2 deletions skills-manifest.json
Original file line number Diff line number Diff line change
Expand Up @@ -50,8 +50,8 @@
"files": 12
},
"media-use": {
"hash": "08a544dfaca9e846",
"files": 158
"hash": "a7208c47707dfab0",
"files": 159
},
"motion-graphics": {
"hash": "32641ae2b94c4a8f",
Expand Down
42 changes: 40 additions & 2 deletions skills/media-use/audio/scripts/audio.mjs
Original file line number Diff line number Diff line change
Expand Up @@ -39,7 +39,7 @@
// the generate path it is spawned detached (bgm_pending:true) — run wait-bgm.mjs
// before assembling.

import { existsSync, mkdirSync, readFileSync } from "node:fs";
import { existsSync, mkdirSync, readdirSync, readFileSync } from "node:fs";
import { dirname, join, resolve } from "node:path";
import { fileURLToPath } from "node:url";
import { heygenAuthHeaders, heygenCredential, loadEnvFromDir } from "./lib/heygen.mjs";
Expand Down Expand Up @@ -68,6 +68,7 @@ const die = (m) => {
process.exit(1);
};
const r3 = (x) => Number(x.toFixed(3));
const lineText = (line) => String(line.text ?? "").trim();

// Two independent reports of an unbounded Promise.all over TTS lines
// overwhelming a machine: one OOM'd 12/13 concurrent Kokoro TTS +
Expand Down Expand Up @@ -140,7 +141,7 @@ if (only.has("tts") && lines.length) {
console.error(`· tts: ${ttsProvider} · voice ${voiceId} · ${lines.length} line(s)`);
const synthLine = async (line) => {
const id = String(line.id);
const text = String(line.text ?? "").trim();
const text = lineText(line);
if (!text) {
anomalies.push(`line ${id}: empty text — skipped`);
return null;
Expand Down Expand Up @@ -174,6 +175,43 @@ if (only.has("tts") && lines.length) {
for (const v of voices)
console.error(` voice ${v.id}: ${v.path} (${v.duration_s}s, ${v.words.length} words)`);
}

// ── reconcile externally-supplied narration ──────────────────────────────────
// A narration WAV can land in assets/voice/<id>.wav without ever going
// through the synthLine loop above — hand-placed, copied over from another
// run, or left behind by one that generated the file but didn't finish
// registering it. voices[] only ever grows inside that loop, so such a file
// is invisible to every consumer that trusts voices[] (assemble-index.mjs's
// <audio> emission, captions.mjs) with zero warning. Runs every invocation,
// independent of --only, so a workflow that reassembles after only bgm/sfx
// still picks up a line that TTS registered in an earlier run and a manual
// drop-in registers on the very next run. Only reconciles ids the current
// request still asks for AND still has real text for (`lines`, same
// empty-text exclusion synthLine uses above) — a stale file from a
// since-edited or since-cleared script line is left alone, not resurrected.
const voiceDir = join(hyperframesDir, "assets", "voice");
if (lines.length && existsSync(voiceDir)) {
const expectedIds = new Set(lines.filter((l) => lineText(l)).map((l) => String(l.id)));
const knownIds = new Set(voices.map((v) => String(v.id)));
for (const file of readdirSync(voiceDir)) {
if (!file.endsWith(".wav")) continue;
const id = file.slice(0, -".wav".length);
if (knownIds.has(id) || !expectedIds.has(id)) continue;
const rel = `assets/voice/${file}`;
const dur = ffprobeDuration(join(hyperframesDir, rel));
if (!isFinite(dur) || dur <= 0) {
anomalies.push(
`voice ${id}: found ${rel} on disk but couldn't read its duration — not reconciled`,
);
continue;
}
// No word timing is available for a file that never went through TTS —
// captions.mjs simply has nothing to render for this line, same as any
// other voice with an empty words[] today.
voices.push({ id, path: rel, duration_s: r3(dur), words: [] });
anomalies.push(`voice ${id}: found ${rel} on disk, missing from the ledger — reconciled`);
}
}
const hasVoice = voices.length > 0;
const totalDuration = r3(voices.reduce((a, v) => a + (v.duration_s || 0), 0));

Expand Down
152 changes: 152 additions & 0 deletions skills/media-use/audio/scripts/audio.reconcile.test.mjs
Original file line number Diff line number Diff line change
@@ -0,0 +1,152 @@
import assert from "node:assert/strict";
import { existsSync, mkdirSync, mkdtempSync, readFileSync, rmSync, writeFileSync } from "node:fs";
import { tmpdir } from "node:os";
import { join } from "node:path";
import { spawnSync } from "node:child_process";
import test from "node:test";

// Regression: a narration WAV that lands in assets/voice/<id>.wav without ever
// going through this engine's own TTS loop (hand-placed, copied from another
// run, or left over from an interrupted one) used to be invisible to every
// consumer that trusts audio_meta.json's voices[] — assemble-index.mjs never
// emits an <audio> element for it, with no warning at all. The fix
// reconciles assets/voice/ against voices[] on every engine invocation.

const engineScript = new URL("./audio.mjs", import.meta.url).pathname;
const HAS_FFMPEG =
spawnSync("ffmpeg", ["-version"], { stdio: "ignore" }).status === 0 &&
spawnSync("ffprobe", ["-version"], { stdio: "ignore" }).status === 0;

function fixture({ lines, existingVoices, voiceFiles }) {
const dir = mkdtempSync(join(tmpdir(), "mu-audio-reconcile-"));
writeFileSync(join(dir, "audio_request.json"), JSON.stringify({ lines }));
if (existingVoices) {
writeFileSync(
join(dir, "audio_meta.json"),
JSON.stringify({ bgm: null, voices: existingVoices, sfx: [] }),
);
}
if (voiceFiles?.length) {
const voiceDir = join(dir, "assets", "voice");
mkdirSync(voiceDir, { recursive: true });
for (const file of voiceFiles) {
// 1s silent tone — small, deterministic, real enough for ffprobe to read a duration.
const gen = spawnSync(
"ffmpeg",
["-y", "-f", "lavfi", "-i", "anullsrc=r=8000:cl=mono", "-t", "1", join(voiceDir, file)],
{ encoding: "utf8" },
);
assert.equal(gen.status, 0, gen.stderr);
}
}
return { dir, cleanup: () => rmSync(dir, { recursive: true, force: true }) };
}

function runEngine(dir) {
const outPath = join(dir, "audio_meta.json");
// --only "" disables the tts/bgm/sfx stages entirely (none of their names
// appear in the empty split), so only the unconditional reconciliation
// pass runs — no TTS provider or network access needed for this test.
const r = spawnSync(
process.execPath,
[
engineScript,
"--request",
join(dir, "audio_request.json"),
"--hyperframes",
dir,
"--out",
outPath,
"--only",
"",
],
{ encoding: "utf8" },
);
assert.equal(r.status, 0, r.stderr);
return JSON.parse(readFileSync(outPath, "utf8"));
}

test(
"a narration WAV present on disk but missing from the ledger is backfilled",
{ skip: !HAS_FFMPEG },
(t) => {
const { dir, cleanup } = fixture({
lines: [{ id: "01", text: "Hello" }],
existingVoices: [],
voiceFiles: ["01.wav"],
});
t.after(cleanup);

const meta = runEngine(dir);

assert.equal(meta.voices.length, 1, JSON.stringify(meta));
assert.equal(meta.voices[0].id, "01");
assert.equal(meta.voices[0].path, "assets/voice/01.wav");
assert.ok(meta.voices[0].duration_s > 0, "expected a real, positive duration");
assert.deepEqual(meta.voices[0].words, []);
},
);

test("a voice already present in the ledger is not duplicated", { skip: !HAS_FFMPEG }, (t) => {
const { dir, cleanup } = fixture({
lines: [{ id: "01", text: "Hello" }],
existingVoices: [{ id: "01", path: "assets/voice/01.wav", duration_s: 2.5, words: [] }],
voiceFiles: ["01.wav"],
});
t.after(cleanup);

const meta = runEngine(dir);

assert.equal(meta.voices.length, 1);
// The pre-existing ledger entry wins verbatim — reconciliation only fills gaps.
assert.equal(meta.voices[0].duration_s, 2.5);
});

test(
"a stale WAV for a line the current script no longer asks for is left alone",
{ skip: !HAS_FFMPEG },
(t) => {
const { dir, cleanup } = fixture({
lines: [{ id: "01", text: "Hello" }],
existingVoices: [],
voiceFiles: ["01.wav", "99.wav"],
});
t.after(cleanup);

const meta = runEngine(dir);

assert.equal(meta.voices.length, 1);
assert.equal(meta.voices[0].id, "01");
assert.ok(
!meta.voices.some((v) => v.id === "99"),
"a file with no matching request line must not be resurrected",
);
},
);

test(
"a WAV for a line whose text was since cleared is left alone, not resurrected",
{ skip: !HAS_FFMPEG },
(t) => {
const { dir, cleanup } = fixture({
lines: [{ id: "01", text: " " }],
existingVoices: [],
voiceFiles: ["01.wav"],
});
t.after(cleanup);

const meta = runEngine(dir);

assert.deepEqual(meta.voices, [], "empty-text line must not resurrect a leftover WAV");
},
);

test("no assets/voice directory at all is a no-op, not a crash", { skip: !HAS_FFMPEG }, (t) => {
const { dir, cleanup } = fixture({ lines: [{ id: "01", text: "Hello" }], existingVoices: [] });
t.after(cleanup);
assert.equal(existsSync(join(dir, "assets", "voice")), false);

const meta = runEngine(dir);

assert.deepEqual(meta.voices, []);
});
Loading