Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
Show all changes
40 commits
Select commit Hold shift + click to select a range
db2e824
fix(media-use): record agent-made voice, music and sound effects in t…
miguel-heygen Sep 28, 2026
c5c0796
fix(media-use): record in place only when --source says the agent mad…
miguel-heygen Sep 28, 2026
f2cf728
test(media-use): a rewritten agent file keeps one manifest record
miguel-heygen Sep 28, 2026
662159e
fix(media-use): leave a person's own sound effect unlabelled and mark…
miguel-heygen Sep 28, 2026
18377c6
Merge remote-tracking branch 'origin/main' into fix/media-use-audio-r…
miguel-heygen Sep 28, 2026
d02da28
refactor(media-use): record agent audio in-process through the shared…
miguel-heygen Sep 28, 2026
48f3bc0
style(media-use): format the in-process recording and refresh the ski…
miguel-heygen Sep 28, 2026
d7c456f
chore(skills): refresh the skills manifest for the media-use changes
miguel-heygen Sep 28, 2026
5ea4c34
fix(media-use): report a broken ffprobe setting plainly and document …
miguel-heygen Sep 28, 2026
db8b2d8
style(skills): format the pipeline manifest assertion
miguel-heygen Sep 28, 2026
04eb199
chore(skills): refresh the skills manifest
miguel-heygen Sep 28, 2026
8b37a1d
fix(media-use): never replace a voice, music or sound file the person…
miguel-heygen Sep 28, 2026
6a2db3e
style(media-use): format the keep-person-files change
miguel-heygen Sep 28, 2026
fc23d5f
chore(skills): refresh the skills manifest
miguel-heygen Sep 28, 2026
af3c484
fix(media-use): a regenerated file gets a fresh record, and lookups s…
miguel-heygen Sep 28, 2026
03271d0
test(media-use): a replaced record no longer answers a prompt lookup …
miguel-heygen Sep 28, 2026
f44bdad
chore(skills): refresh the skills manifest
miguel-heygen Sep 28, 2026
75ade58
chore(skills): refresh the skills manifest from a clean tree
miguel-heygen Sep 28, 2026
2dbd1b9
Merge remote-tracking branch 'origin/fix/media-use-audio-records' int…
miguel-heygen Sep 28, 2026
0fe891a
fix(media-use): never give a recorded file the id of a download in fl…
miguel-heygen Sep 28, 2026
bf463a0
chore(skills): refresh the media-use manifest hash
miguel-heygen Sep 28, 2026
33aa4e3
Merge remote-tracking branch 'origin/fix/media-use-audio-records' int…
miguel-heygen Sep 28, 2026
e896082
fix(media-use): say when audio lands on a new name, and never let two…
miguel-heygen Sep 28, 2026
1c162a8
style(media-use): format the keep-files change
miguel-heygen Sep 28, 2026
e40129d
chore(skills): refresh the media-use manifest hash
miguel-heygen Sep 28, 2026
c5609e4
fix(media-use): keep two sound effects off one file when a name is taken
miguel-heygen Sep 28, 2026
c281f52
refactor(media-use): drop a set the one-at-a-time downloads never needed
miguel-heygen Sep 28, 2026
ea2c2ee
chore(skills): refresh the media-use manifest hash
miguel-heygen Sep 28, 2026
d4d4f05
fix(media-use): keep two sound effects off one recorded file on a rerun
miguel-heygen Sep 28, 2026
0fa6986
style(media-use): format the rerun test
miguel-heygen Sep 28, 2026
de003bc
chore(skills): refresh the media-use manifest hash
miguel-heygen Sep 28, 2026
9978597
fix(media-use): list only a file's newest record in candidates and th…
miguel-heygen Sep 28, 2026
739a3e2
revert(media-use): leave promote as it was, nothing calls it
miguel-heygen Sep 28, 2026
fdcabb8
test(media-use): give the index test's two assets their own files
miguel-heygen Sep 28, 2026
d66ebfa
chore(skills): refresh the media-use manifest hash
miguel-heygen Sep 28, 2026
7cdc4ae
Merge remote-tracking branch 'origin/fix/media-use-audio-records' int…
miguel-heygen Sep 28, 2026
c4209f2
chore(skills): regenerate the manifest after merging the base branch
miguel-heygen Sep 28, 2026
dd78781
fix(core): a figma import lists each media file once, like media-use
miguel-heygen Sep 28, 2026
ac0edee
Merge remote-tracking branch 'origin/fix/media-use-audio-records' int…
miguel-heygen Sep 28, 2026
c512878
Merge remote-tracking branch 'origin/main' into fix/media-use-keep-pe…
miguel-heygen Sep 29, 2026
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
3 changes: 3 additions & 0 deletions packages/cli/src/media-use/lib/manifest.mjs
Original file line number Diff line number Diff line change
Expand Up @@ -79,6 +79,9 @@ export function appendRecord(projectDir, record) {
appendFileSync(p, line);
}

/** Sources that mean the agent made or fetched the file; any other file is the person's own. */
export const AGENT_SOURCES = ["generated", "search", "bundled"];

/** The record a path has now: the manifest only appends, so the last one for a path wins. */
export function latestRecordFor(projectDir, path) {
return readManifest(projectDir).findLast((record) => record.path === path);
Expand Down
8 changes: 4 additions & 4 deletions packages/cli/src/media-use/resolve.mjs
Original file line number Diff line number Diff line change
Expand Up @@ -5,6 +5,7 @@ import { existsSync, statSync, writeFileSync, renameSync, rmSync, realpathSync }
import { resolve, join, extname, basename, relative, isAbsolute, sep } from "node:path";
import { parseArgs } from "node:util";
import {
AGENT_SOURCES,
appendRecord,
latestRecordFor,
recordInPlace,
Expand Down Expand Up @@ -60,7 +61,6 @@ import {
} from "./lib/local-media-search.mjs";

const INGEST_TYPES = listTypes();
const RECORDED_SOURCES = ["generated", "search", "bundled"];
const DEFAULT_EXT = {
bgm: ".wav",
sfx: ".mp3",
Expand Down Expand Up @@ -130,7 +130,7 @@ Options:
--reuse <sha> Import a specific global-cache asset (by content sha/prefix,
from --candidates) into this project
--from <file> Freeze a local file or direct public URL (ingest)
--source <how> With --from: how the file was made (${RECORDED_SOURCES.join(" | ")}).
--source <how> With --from: how the file was made (${AGENT_SOURCES.join(" | ")}).
A file already inside the project is then recorded where it is
--params <json> Build an explicit parametric LUT (lut/grade only)
--for <media> Analyze a local image/video and add measured grade adjust
Expand Down Expand Up @@ -918,8 +918,8 @@ async function ingest(src) {
console.error(`error: refusing to ingest a 0-byte file: ${src}`);
process.exit(2);
}
if (args.source && !RECORDED_SOURCES.includes(args.source)) {
console.error(`error: --source takes one of: ${RECORDED_SOURCES.join(", ")}`);
if (args.source && !AGENT_SOURCES.includes(args.source)) {
console.error(`error: --source takes one of: ${AGENT_SOURCES.join(", ")}`);
process.exit(2);
}
if (args.source && (type === "lut" || type === "grade")) {
Expand Down
2 changes: 1 addition & 1 deletion skills-manifest.json
Original file line number Diff line number Diff line change
Expand Up @@ -54,7 +54,7 @@
"files": 1
},
"media-use": {
"hash": "1871ad36628ac54b",
"hash": "0744050e3114e60a",
"files": 104
},
"motion-graphics": {
Expand Down
2 changes: 2 additions & 0 deletions skills/media-use/audio/references/bgm.md
Original file line number Diff line number Diff line change
Expand Up @@ -15,6 +15,8 @@ One music bed per composition, produced by the shared audio engine (`scripts/aud
- **`query`** — the mood, used for retrieval and as a fallback prompt seed (e.g. a storyboard's `music:` field, falling back to `message` → `arc` → `"calm cinematic underscore"`).
- **`prompt`** — an explicit full prompt for generation; omit and the engine infers one (see Mood inference). Optional `blob` / `archetype` / `arc` feed that inference.

Both routes keep a file of yours already at the output name: the engine writes the next free name (`track-2.mp3`), reports it as an anomaly, and `bgm.path` carries the real path.

## HeyGen retrieval (default)

`searchSounds(query, "music", { limit: 5 })` → `GET /audio/sounds?query=<mood>&type=music&limit=5`. Take the top result (ranked by `score`), download its presigned `audio_url` → `assets/bgm/track.mp3`. Synchronous. No match → skip (BGM is optional; never fail the render over it). Cue written to `audio_meta.json`:
Expand Down
5 changes: 2 additions & 3 deletions skills/media-use/audio/references/sfx.md
Original file line number Diff line number Diff line change
Expand Up @@ -16,15 +16,14 @@ Each line names the effects it wants: `lines[].sfx: ["whoosh", "ui click"]`. The
"id": "3", // joins the cue to the caller's model (frame / scene / segment)
"name": "whoosh",
"file": "assets/sfx/whoosh.mp3", // downloaded or copied, relative to project root
"source": "heygen" | "local" | "project", // which route resolved it; "project": a file already
// in assets/sfx under that name that is not the library's copy
"source": "heygen" | "local", // which route resolved it
"offset_s": 0, // delay from the line's start
"duration_s": 0.57,
"volume": 0.35 // SFX sit UNDER voice + BGM
}
```

A cue that matches nothing is **skipped** (recorded as an anomaly); SFX never blocks a render.
A cue that matches nothing is **skipped** (recorded as an anomaly); SFX never blocks a render. Neither route replaces a file of yours already at the output name: the cue gets the next free name (`whoosh-2.mp3`), reported as an anomaly, and `file` carries the real path.

## HeyGen retrieval (credentialed)

Expand Down
2 changes: 1 addition & 1 deletion skills/media-use/audio/references/tts.md
Original file line number Diff line number Diff line change
Expand Up @@ -135,7 +135,7 @@ node <SKILL_DIR>/audio/scripts/audio.mjs \
--request ./audio_request.json --hyperframes . --out ./audio_meta.json --only tts
```

The engine saves `assets/voice/intro.wav`, measures its duration, and transcribes
The engine saves `assets/voice/intro.wav` (or `intro-2.wav` when a file of yours already has that name; `voices[].path` says which), measures its duration, and transcribes
it into `voices[].words` in `audio_meta.json`. Check that every requested line
has audio and nonempty word timings before building a captioned video. Review
the timings against the actual audio; transcription is estimated alignment,
Expand Down
7 changes: 5 additions & 2 deletions skills/media-use/audio/scripts/audio.mjs
Original file line number Diff line number Diff line change
Expand Up @@ -55,7 +55,7 @@ import { generateBgmDetached, inferBgmPrompt, retrieveBgm } from "./lib/bgm.mjs"
import { resolveSfx } from "./lib/sfx.mjs";
import { mapWithConcurrency } from "./lib/concurrency.mjs";
import { openAudioMeta } from "./lib/audio-meta.mjs";
import { recordInManifest, writtenAssets } from "./lib/media-record.mjs";
import { recordInManifest, voicePaths, writtenAssets } from "./lib/media-record.mjs";

const HERE = dirname(fileURLToPath(import.meta.url));
const argv = process.argv.slice(2);
Expand Down Expand Up @@ -138,14 +138,15 @@ if (only.has("tts") && lines.length) {
lang,
});
console.error(`· tts: ${ttsProvider} · voice ${voiceId} · ${lines.length} line(s)`);
const paths = voicePaths(hyperframesDir, lines, anomalies);
const synthLine = async (line) => {
const id = String(line.id);
const text = String(line.text ?? "").trim();
if (!text) {
anomalies.push(`line ${id}: empty text — skipped`);
return null;
}
const rel = `assets/voice/${id}.wav`;
const rel = paths.get(id);
const abs = join(hyperframesDir, rel);
const { ok, words, error } = await synthesizeOne({
provider: ttsProvider,
Expand Down Expand Up @@ -217,6 +218,7 @@ if (only.has("bgm")) {
headers: heygenAuthHeaders(),
hyperframesDir,
hasVoice,
anomalies,
});
if (bgm) {
bgmFields.bgm_provider = "heygen";
Expand All @@ -243,6 +245,7 @@ if (only.has("bgm")) {
lyriaRecipe: existsSync(lyriaRecipe) ? lyriaRecipe : null,
seedSeconds,
hasVoice,
anomalies,
});
if (gen.disabled) {
anomalies.push(`bgm: ${gen.reason}`);
Expand Down
69 changes: 61 additions & 8 deletions skills/media-use/audio/scripts/audio.test.mjs
Original file line number Diff line number Diff line change
@@ -1,9 +1,10 @@
import { strict as assert } from "node:assert";
import { test } from "node:test";
import { mkdtempSync, rmSync, existsSync, writeFileSync } from "node:fs";
import { mkdtempSync, mkdirSync, readFileSync, rmSync, existsSync, writeFileSync } from "node:fs";
import { join, dirname } from "node:path";
import { tmpdir } from "node:os";
import { fileURLToPath } from "node:url";
import { recordInManifest } from "./lib/media-record.mjs";
import { resolveSfx } from "./lib/sfx.mjs";

// Proves the relocated engine (skills/media-use/audio/) still resolves its
Expand Down Expand Up @@ -48,26 +49,78 @@ test("an unknown cue is reported, not fatal", async () => {
}
});

test("a person's own file under a bundled name is not labelled as the library's", async () => {
test("a person's own file under a bundled name survives, and the cue gets the next free name", async () => {
const dir = mkdtempSync(join(tmpdir(), "mu-audio-"));
try {
const own = join(dir, "assets", "sfx", "whoosh.mp3");
mkdirSync(dirname(own), { recursive: true });
writeFileSync(own, "the person's own whoosh");
const resolve = () =>
resolveSfx({
cues: [{ id: "1", name: "whoosh" }],
heygenOK: false,
hyperframesDir: dir,
sfxLibDir,
});
const copied = await resolve();
const reused = await resolve();
writeFileSync(join(dir, copied.sfx[0].file), "the person's own whoosh");
const own = await resolve();

const first = await resolve();
const second = await resolve();

assert.equal(readFileSync(own, "utf8"), "the person's own whoosh");
assert.deepEqual(
[first, second].map(({ sfx }) => [sfx[0].file, sfx[0].source]),
[
["assets/sfx/whoosh-2.mp3", "local"],
["assets/sfx/whoosh-2.mp3", "local"],
],
);
assert.deepEqual(
[copied, reused, own].map(({ sfx }) => sfx[0].source),
["local", "local", "project"],
readFileSync(join(dir, "assets/sfx/whoosh-2.mp3")),
readFileSync(join(sfxLibDir, "whoosh.mp3")),
);
} finally {
rmSync(dir, { recursive: true, force: true });
}
});

test("two effects never share a file when one's name is taken by the person, run after run", async () => {
const dir = mkdtempSync(join(tmpdir(), "mu-audio-"));
try {
mkdirSync(join(dir, "assets", "sfx"), { recursive: true });
writeFileSync(join(dir, "assets", "sfx", "glitch.mp3"), "the person's own glitch");
const realFetch = globalThis.fetch;
globalThis.fetch = async (url) => {
const query = new URL(url).searchParams.get("query");
if (query)
return Response.json({ data: [{ audio_url: `https://sound.test/${query}`, score: 0.6 }] });
return new Response(`bytes of ${new URL(url).pathname}`);
};
const run = async (names) => {
const { sfx } = await resolveSfx({
cues: names.map((name, index) => ({ id: String(index), name })),
heygenOK: true,
headers: {},
hyperframesDir: dir,
sfxLibDir,
});
const files = sfx.map(({ file }) => file);
recordInManifest(
dir,
[...new Set(files)].map((path) => ({ path, type: "sfx", source: "search" })),
);
return files;
};
try {
assert.deepEqual(await run(["glitch"]), ["assets/sfx/glitch-2.mp3"]);
assert.deepEqual(await run(["glitch", "glitch 2", "glitch"]), [
"assets/sfx/glitch-2.mp3",
"assets/sfx/glitch-2-2.mp3",
"assets/sfx/glitch-2.mp3",
]);
} finally {
globalThis.fetch = realFetch;
}
} finally {
rmSync(dir, { recursive: true, force: true });
}
});
17 changes: 14 additions & 3 deletions skills/media-use/audio/scripts/gemini-pipeline.test.mjs
Original file line number Diff line number Diff line change
Expand Up @@ -9,9 +9,15 @@ import { spawnSync } from "node:child_process";
// external audio tools are fixtures. This is not a live synthesis/render test.
for (const expired of [false, true]) {
for (const only of ["tts", "tts,bgm,sfx"]) {
// With expired credentials the person already has an assets/voice/intro.wav, which the engine must keep.
const voicePath = expired ? "assets/voice/intro-2.wav" : "assets/voice/intro.wav";
test(`Gemini engine returns caption metadata with ${expired ? "expired" : "absent"} HeyGen credentials (${only})`, (t) => {
const dir = mkdtempSync(join(tmpdir(), "hf-gemini-pipeline-"));
t.after(() => rmSync(dir, { recursive: true, force: true }));
if (expired) {
mkdirSync(join(dir, "assets/voice"), { recursive: true });
writeFileSync(join(dir, "assets/voice/intro.wav"), "the person's own intro");
}
const config = join(dir, "heygen");
mkdirSync(config);
if (expired)
Expand Down Expand Up @@ -48,7 +54,7 @@ const assert = require('node:assert/strict');
const fs = require('node:fs');
const path = require('node:path');
const args = process.argv.slice(2);
assert.deepEqual(args.slice(0, 3), ['hyperframes', 'transcribe', 'assets/voice/intro.wav']);
assert.deepEqual(args.slice(0, 3), ['hyperframes', 'transcribe', '${voicePath}']);
assert.ok(fs.existsSync(args[2]));
assert.equal(args[args.indexOf('--model')+1], 'small.en');
fs.writeFileSync(path.join(args[args.indexOf('--dir')+1], 'transcript.json'), JSON.stringify([
Expand Down Expand Up @@ -99,7 +105,7 @@ fs.writeFileSync(path.join(args[args.indexOf('--dir')+1], 'transcript.json'), JS
assert.deepEqual(meta.voices, [
{
id: "intro",
path: "assets/voice/intro.wav",
path: voicePath,
duration_s: 1.25,
words: [
{ id: "w0", text: "Hello", start: 0.1, end: 0.4 },
Expand All @@ -113,8 +119,13 @@ fs.writeFileSync(path.join(args[args.indexOf('--dir')+1], 'transcript.json'), JS
manifest
.map((line) => JSON.parse(line))
.map(({ path, type, source }) => [path, type, source]),
[["assets/voice/intro.wav", "voice", "generated"]],
[[voicePath, "voice", "generated"]],
);
if (expired)
assert.equal(
readFileSync(join(dir, "assets/voice/intro.wav"), "utf8"),
"the person's own intro",
);
});
}
}
12 changes: 8 additions & 4 deletions skills/media-use/audio/scripts/lib/bgm.mjs
Original file line number Diff line number Diff line change
Expand Up @@ -12,9 +12,10 @@
// Missing/failed BGM never blocks a render.

import { spawn, spawnSync } from "node:child_process";
import { existsSync, mkdirSync, openSync, closeSync } from "node:fs";
import { existsSync, mkdirSync, openSync, closeSync, rmSync } from "node:fs";
import { join } from "node:path";
import { downloadTo, searchSounds } from "./heygen.mjs";
import { agentWritePath } from "./media-record.mjs";
import { pythonInvocation } from "./python.mjs";

const r3 = (x) => Number(x.toFixed(3));
Expand Down Expand Up @@ -45,12 +46,12 @@ function pipInstall(deps) {
}

// ── retrieval (HeyGen music library) ──────────────────────────────────────────
export async function retrieveBgm({ query, headers, hyperframesDir, hasVoice }) {
export async function retrieveBgm({ query, headers, hyperframesDir, hasVoice, anomalies }) {
const q = query || "calm cinematic underscore";
const results = await searchSounds(q, "music", headers, { limit: 5 });
if (!results.length) return null;
const top = results[0];
const rel = "assets/bgm/track.mp3";
const rel = agentWritePath(hyperframesDir, "assets/bgm/track.mp3", { anomalies });
await downloadTo(top.audio_url, join(hyperframesDir, rel));
return {
path: rel,
Expand Down Expand Up @@ -113,8 +114,9 @@ export function generateBgmDetached({
lyriaRecipe,
seedSeconds = 28,
hasVoice,
anomalies,
}) {
const rel = "assets/bgm/track.wav";
const rel = agentWritePath(hyperframesDir, "assets/bgm/track.wav", { anomalies });
const abs = join(hyperframesDir, rel);
mkdirSync(join(hyperframesDir, "assets", "bgm"), { recursive: true });
const log = join(hyperframesDir, "assets", "bgm", `bgm-${Date.now()}.log`);
Expand All @@ -141,6 +143,7 @@ export function generateBgmDetached({
"--prompt",
prompt,
]);
rmSync(abs, { force: true }); // wait-bgm takes any file here as the finished track
const proc = spawn(cmd, args, { detached: true, stdio: ["ignore", fd, fd] });
proc.unref();
closeSync(fd);
Expand All @@ -159,6 +162,7 @@ export function generateBgmDetached({
const loops = targetS > seedS ? Math.ceil(targetS / seedS) : 1;
const script = musicgenScript({ prompt, abs, targetS, seedS });
const { cmd, args } = pythonInvocation(["-c", script]);
rmSync(abs, { force: true });
const proc = spawn(cmd, args, { detached: true, stdio: ["ignore", fd, fd] });
proc.unref();
closeSync(fd);
Expand Down
46 changes: 45 additions & 1 deletion skills/media-use/audio/scripts/lib/bgm.test.mjs
Original file line number Diff line number Diff line change
@@ -1,6 +1,15 @@
import { test } from "node:test";
import assert from "node:assert/strict";
import { BGM_BED_VOLUME, BGM_SILENT_VOLUME, bgmDefaultVolume } from "./bgm.mjs";
import { existsSync, mkdirSync, mkdtempSync, rmSync, writeFileSync } from "node:fs";
import { join } from "node:path";
import { tmpdir } from "node:os";
import { appendRecord } from "../../../scripts/lib/manifest.mjs";
import {
BGM_BED_VOLUME,
BGM_SILENT_VOLUME,
bgmDefaultVolume,
generateBgmDetached,
} from "./bgm.mjs";

// Regression: narrated pipelines used to ship BGM at 0.8 (≈ -2 dB), ~16 dB
// hotter than a music bed under a voice should be. The default under narration
Expand Down Expand Up @@ -28,3 +37,38 @@ test("the narrated default is well below the voice (≈ 0 dBFS)", () => {
`bed should sit ≥16 dB under the voice, got ${separation.toFixed(1)} dB`,
);
});

test(
"generating music again clears the engine's old track, so waiting cannot mistake it for the new one",
{ skip: process.platform === "win32" && "the fake python is a shell script" },
(t) => {
const dir = mkdtempSync(join(tmpdir(), "hf-bgm-"));
const path = process.env.PATH;
t.after(() => {
process.env.PATH = path;
rmSync(dir, { recursive: true, force: true });
});
mkdirSync(join(dir, "bin"));
for (const name of ["python3", "python"])
writeFileSync(join(dir, "bin", name), "#!/bin/sh\nexit 0\n", { mode: 0o755 });
process.env.PATH = `${join(dir, "bin")}:${path}`;
mkdirSync(join(dir, "assets/bgm"), { recursive: true });
writeFileSync(join(dir, "assets/bgm/track.wav"), "last run's track");
appendRecord(dir, {
id: "bgm_001",
type: "bgm",
path: "assets/bgm/track.wav",
source: "generated",
});

const gen = generateBgmDetached({
prompt: "calm",
durationS: 5,
hyperframesDir: dir,
anomalies: [],
});

assert.equal(gen.path, "assets/bgm/track.wav");
assert.equal(existsSync(join(dir, "assets/bgm/track.wav")), false);
},
);
Loading
Loading