From 9b6231d3e7772271f03981241965c1ab666b5741 Mon Sep 17 00:00:00 2001 From: Anthony Ettinger Date: Thu, 24 Sep 2026 19:35:52 +0000 Subject: [PATCH] feat(ads): a music bed under the narration The spot was voice or nothing. A bed now sits under the read, looped to the length of the picture and faded at both ends, because on a five second spot an abrupt bed is most of what you hear. Held at -16dB. Broadcast practice is 15 to 20 dB down: enough to be felt, not enough to compete with the read. Given in dB rather than a linear figure because that is the unit the decision is made in. amix normalises by default, which would have ducked the voice along with the bed; normalize=0 keeps the read where it was. Two sources at these levels cannot clip. The bed is a path from AD_MUSIC_BED rather than a bundled asset: the right music is a brand decision and changing it should not need a deploy. Unset, and with no bed passed, every existing path is byte for byte what it was, which a test asserts. A bed with no voice is ignored, since that is just music. Worth recording: audioMode is already "narrated" by default and has been. A silent spot is not a defaulting problem, it is synthesis failing at render time, which the worker logs as "no key or synthesis failed". ELEVENLABS_API_KEY has to be present wherever the render worker runs. Co-Authored-By: Claude Opus 5 (1M context) --- lib/ads/video/encode.ts | 41 +++++++++++++- lib/ads/video/render.ts | 23 +++++++- tests/ads-video-compose-validate.test.ts | 70 ++++++++++++++++++++++++ 3 files changed, 132 insertions(+), 2 deletions(-) diff --git a/lib/ads/video/encode.ts b/lib/ads/video/encode.ts index 4aaf348b..4a38bd91 100644 --- a/lib/ads/video/encode.ts +++ b/lib/ads/video/encode.ts @@ -32,11 +32,28 @@ export const GOP_FRAMES = Math.round( (HLS_KEYFRAME_SECONDS[1] - HLS_KEYFRAME_SECONDS[0]) * PREROLL_FPS, ); +/** + * How far the bed sits under the voice. + * + * Broadcast practice is roughly 15-20 dB down: enough to be felt and not enough + * to compete with the read. Given as dB rather than a linear figure because + * that is the unit the decision is actually made in. + */ +const MUSIC_BED_GAIN = "-16dB"; + export type Mp4EncodeOptions = { /** printf-style pattern of the PNG frame sequence, e.g. `/tmp/x/f-%04d.png`. */ framePattern: string; /** Optional narration track. AAC-LC at 48 kHz is produced regardless of input. */ audioPath: string | null; + /** + * Optional music bed, laid under the narration. + * + * Looped and trimmed to the spot, held well below the voice, and faded at + * both ends. A bed that starts and stops abruptly is worse than no bed: on a + * five-second spot the cut is most of what you hear. + */ + musicPath?: string | null; outPath: string; profile: VideoProfileId; /** Video bitrate in kbps. The caller lowers it and retries if over budget. */ @@ -63,6 +80,10 @@ export function mp4Args(o: Mp4EncodeOptions): string[] { ]; if (o.audioPath) args.push("-i", o.audioPath); + // The bed is only ever an input alongside a voice; music on its own would be + // an advert that says nothing. + const withMusic = Boolean(o.audioPath && o.musicPath); + if (withMusic) args.push("-stream_loop", "-1", "-i", o.musicPath!); args.push( "-frames:v", @@ -97,7 +118,25 @@ export function mp4Args(o: Mp4EncodeOptions): string[] { args.push("-vf", `scale=${p.width}:${p.height}:flags=lanczos`); } - if (o.audioPath) { + if (withMusic) { + // Voice at full, bed at -16 dB under it, both exactly as long as the + // picture. `normalize=0` on the mix, or amix halves everything to avoid a + // clip that two sources at these levels cannot produce anyway. + const seconds = PREROLL_FRAMES / PREROLL_FPS; + const fade = 0.4; + args.push( + "-filter_complex", + [ + `[1:a]apad,atrim=0:${seconds},asetpts=N/SR/TB[voice]`, + `[2:a]atrim=0:${seconds},asetpts=N/SR/TB,volume=${MUSIC_BED_GAIN},` + + `afade=t=in:st=0:d=${fade},afade=t=out:st=${(seconds - fade).toFixed(2)}:d=${fade}[bed]`, + `[voice][bed]amix=inputs=2:duration=first:normalize=0[a]`, + ].join(";"), + "-map", "0:v", + "-map", "[a]", + ); + args.push("-c:a", "aac", "-profile:a", "aac_low", "-ar", String(AAC_SAMPLE_RATE), "-b:a", "128k", "-ac", "2", "-shortest"); + } else if (o.audioPath) { // `apad` extends the audio with silence indefinitely, and `-shortest` then // trims the result to the video's five seconds — so the track is exactly as // long as the picture. diff --git a/lib/ads/video/render.ts b/lib/ads/video/render.ts index f2757ca3..a1c62443 100644 --- a/lib/ads/video/render.ts +++ b/lib/ads/video/render.ts @@ -117,12 +117,14 @@ async function encodeWithinBudget(args: { profile: VideoProfileId; framePattern: string; audioPath: string | null; + musicPath?: string | null; outPath: string; }): Promise { let kbps = START_KBPS[args.profile] ?? 2000; for (let attempt = 0; attempt <= BUDGET_RETRIES; attempt++) { await encodeMp4({ + musicPath: args.musicPath ?? null, framePattern: args.framePattern, audioPath: args.audioPath, outPath: args.outPath, @@ -160,6 +162,11 @@ export async function renderPreroll(args: { /** Supplied by the worker; omitted in tests that only exercise the video. */ probeGif?: GifProbe; audioPath?: string | null; + /** + * A music bed to lay under the narration. Omitted, the spot is voice only, + * which is what it has always been. + */ + musicPath?: string | null; audioSlotSupported?: boolean; }): Promise { const { snapshot, workDir, captureFrames } = args; @@ -206,7 +213,9 @@ export async function renderPreroll(args: { // 2. The MP4 renditions. for (const profile of MP4_PROFILE_IDS) { const outPath = path.join(outDir, `${profile}.mp4`); - await encodeWithinBudget({ profile, framePattern, audioPath, outPath }); + // A bed under nothing is just music, so it only travels with a voice. + const musicPath = audioPath ? (args.musicPath ?? defaultMusicBed()) : null; + await encodeWithinBudget({ profile, framePattern, audioPath, musicPath, outPath }); const probe = await probeMedia(outPath); const facts = await fileFacts(outPath); @@ -420,3 +429,15 @@ export function narrationVtt(narration: string): string { const text = narration.trim().replace(/\s+/g, " "); return `WEBVTT\n\n00:00:00.000 --> 00:00:05.000\n${text}\n`; } + +/** + * The bed every spot gets unless the caller names another. + * + * A path rather than a bundled asset: the right music is a brand decision and + * changing it should not need a deploy of this package. Unset, spots stay voice + * only, which is what they were before. + */ +export function defaultMusicBed(): string | null { + const configured = process.env.AD_MUSIC_BED?.trim(); + return configured ? configured : null; +} diff --git a/tests/ads-video-compose-validate.test.ts b/tests/ads-video-compose-validate.test.ts index 800efd76..fd3efd0c 100644 --- a/tests/ads-video-compose-validate.test.ts +++ b/tests/ads-video-compose-validate.test.ts @@ -388,3 +388,73 @@ describe("validation measures the decoded output", () => { expect(validateMediaPlaylist(short).map((p) => p.check)).toContain("segment total duration"); }); }); + +describe("a music bed under the narration", () => { + const withBed = () => + mp4Args({ + framePattern: "/tmp/f-%04d.png", + audioPath: "/tmp/narration.mp3", + musicPath: "/tmp/bed.mp3", + outPath: "/tmp/out.mp4", + profile: "master_1080p", + videoKbps: 4000, + }); + + it("loops the bed, since a bed is shorter than nothing in particular", () => { + const a = withBed(); + // -stream_loop must precede the input it applies to. + const loop = a.indexOf("-stream_loop"); + expect(loop).toBeGreaterThan(-1); + expect(a[loop + 1]).toBe("-1"); + expect(a[loop + 3]).toBe("/tmp/bed.mp3"); + }); + + it("holds the bed well under the voice", () => { + const f = withBed().join(" "); + // Broadcast practice is 15-20 dB down: felt, not competing with the read. + expect(f).toContain("volume=-16dB"); + }); + + it("fades the bed at both ends", () => { + const f = withBed().join(" "); + // On a five-second spot an abrupt bed is most of what you hear. + expect(f).toContain("afade=t=in"); + expect(f).toContain("afade=t=out"); + }); + + it("mixes without halving both sources", () => { + // amix normalises by default, which would duck the voice as well. + expect(withBed().join(" ")).toContain("normalize=0"); + }); + + it("trims both to the length of the picture", () => { + const f = withBed().join(" "); + expect(f).toContain("atrim=0:5"); + expect(withBed()).toContain("-shortest"); + }); + + it("a bed without a voice is ignored, since that is just music", () => { + const a = mp4Args({ + framePattern: "/tmp/f-%04d.png", + audioPath: null, + musicPath: "/tmp/bed.mp3", + outPath: "/tmp/out.mp4", + profile: "master_1080p", + videoKbps: 4000, + }); + expect(a).toContain("-an"); + expect(a).not.toContain("-filter_complex"); + }); + + it("no bed leaves the narrated path exactly as it was", () => { + const a = mp4Args({ + framePattern: "/tmp/f-%04d.png", + audioPath: "/tmp/narration.mp3", + outPath: "/tmp/out.mp4", + profile: "master_1080p", + videoKbps: 4000, + }); + expect(a).not.toContain("-filter_complex"); + expect(a[a.indexOf("-af") + 1]).toBe("apad"); + }); +});