From d130ef99fda879bc110abb2abab1f4e5a87a80c4 Mon Sep 17 00:00:00 2001 From: weijian Date: Mon, 21 Sep 2026 16:00:15 +0800 Subject: [PATCH 1/3] feat(guizang-product-video-skill): confirm motion up front and add per-line voiceover - ask once for style, UI motion + camera zoom in/out, and whether to narrate - record the answers in plan.motion and voiceoverRequired with exception reasons - add scripts/make_voiceover.py: per-line EasyRouter Gemini TTS, 48 kHz assembly, evidence - mix the voice as a third layer, duck music under speech, -14 LUFS when narrated - validate narration windows and motion consistency in check_delivery.py, extend regressions - document the motion language and the measured EasyRouter TTS model list --- skills/guizang-product-video-skill/README.md | 7 +- skills/guizang-product-video-skill/SKILL.md | 9 +- .../agents/openai.yaml | 2 +- .../marketplace.json | 10 +- .../references/audio-and-qa.md | 14 + .../references/audio-sourcing.md | 29 ++ .../references/onboarding.md | 4 + .../references/story-and-copy.md | 10 + .../scripts/check_delivery.py | 57 +++- .../scripts/init_project.py | 9 +- .../scripts/make_voiceover.py | 293 ++++++++++++++++++ .../scripts/mix_audio.py | 88 +++++- .../tests/test_regressions.py | 31 ++ 13 files changed, 537 insertions(+), 26 deletions(-) create mode 100755 skills/guizang-product-video-skill/scripts/make_voiceover.py diff --git a/skills/guizang-product-video-skill/README.md b/skills/guizang-product-video-skill/README.md index fa9348f..142efd5 100644 --- a/skills/guizang-product-video-skill/README.md +++ b/skills/guizang-product-video-skill/README.md @@ -44,6 +44,7 @@ npx skills add https://github.com/op7418/guizang-product-video-skill --skill gui | 暂时没有成熟的宣传视觉 | 用内置的 CodePilot 暖白/炭黑样式组织外层画面 | | 几句很技术的更新描述 | 写成完整的白话句子,让人知道改了什么、用起来有什么不同 | | 一支只有画面的片子 | 为本片用代码写配乐,给操作配独立音效,并在关键音效出现时压低音乐 | +| 只有画面、没有旁白 | 需要时可以按镜头逐句生成中文旁白,改一句只重录那一句 | 画面由代码渲染,能继续修改。标题、功能特写、完整工作区和细节镜头交替出现,既有大字文案,也留出看清操作的时间。 @@ -67,6 +68,8 @@ npx skills add https://github.com/op7418/guizang-product-video-skill --skill gui 不需要写一份很长的 brief,把**介绍什么、给谁看、发在哪里**说清楚就够了。 +第一次制作时,它会先确认三件事:风格、动效(界面操作动效 + 镜头 zoom in/out)、要不要旁白。旁白走 EasyRouter 的 Gemini TTS,需要你自己的 key;不想加旁白直接说。 + **做一支版本更新片** > 介绍 v2.1 到 v2.4 的主要更新,优先讲多模型切换、文件预览和浏览器操作。发 B 站,横版 50 秒左右。沿用我们的产品设计,文案口语化一点,不要把提交记录逐条念出来。 @@ -91,7 +94,7 @@ npx skills add https://github.com/op7418/guizang-product-video-skill --skill gui 2. **找真实内容。** 查版本与代码,核对功能状态,接通一个真实组件镜头。 3. **把故事讲顺。** 写分镜和白话文案,渲染关键画面,检查排版与阅读时间。 4. **让画面动起来。** 用可定位时间的主时间轴控制状态、入场、转场和细节。 -5. **把声音配好。** 原创配乐,查找合适音效,对齐操作节点,混音并处理音乐让位。 +5. **把声音配好。** 原创配乐,查找合适音效,需要时按句生成旁白,对齐操作节点,混音并处理音乐让位。 6. **检查,再导出。** 核对画面、字体、图片、Logo、音画同步与最终文件。 默认起点是 **45–60 秒、横版、中文**。你可以调整时长和画幅;竖版需要重新安排取景和文字,通常不会直接裁掉左右两边。 @@ -104,6 +107,8 @@ npx skills add https://github.com/op7418/guizang-product-video-skill --skill gui **音乐和音效分开混。** 点击、弹出、切换、完成提醒等动作有独立的音效事件。关键反馈出现时音乐会短暂降低,让叮咚、确认和转场真正听得见。 +**旁白按句生成。** 需要旁白时,用 EasyRouter 上的 Gemini TTS 按镜头逐句生成,再把每句对齐到镜头和卡点;哪一句不满意就重录哪一句。人声是锚,音乐会在这期间让位。不想要旁白就直说,只靠字幕、配乐和音效,片子也是完整的。 + ![配乐、动作音效与音乐让位](assets/readme/audio.jpg) ## 最后会拿到什么? diff --git a/skills/guizang-product-video-skill/SKILL.md b/skills/guizang-product-video-skill/SKILL.md index 2b64354..58703fd 100644 --- a/skills/guizang-product-video-skill/SKILL.md +++ b/skills/guizang-product-video-skill/SKILL.md @@ -1,7 +1,7 @@ --- name: guizang-product-video-skill category: 视频创作 -description: 制作代码驱动的软件版本更新宣传片(release notes video、changelog promo)。从真实更新提炼卖点,复用产品组件和设计语言,完成分镜、代码原创配乐、动作音效、渲染与验收。 +description: 制作代码驱动的软件版本更新宣传片(release notes video、changelog promo)。默认先确认风格、界面动效与镜头推拉、是否要 Gemini 画外音,再从真实更新提炼卖点,复用产品组件和设计语言,完成分镜、代码原创配乐、动作音效、逐句旁白、渲染与验收。 upstream: op7418/guizang-product-video-skill upstreamPath: . upstreamSha: c45aab4599bc6e883a6f6ea9d82f49ecdde1c8d0 @@ -15,11 +15,11 @@ license: "AGPL-3.0; CodePilot fallback assets: BUSL-1.1 (see README.md)" ## 流程 -1. **确认范围与风格。** 已有工程沿用 brief 和用户决定,只处理本次修改。新片优先补齐产品/仓库、版本范围、发布平台、时长与画幅、语言和链接要求。风格尚未确定时问一次:“沿用代码库自己的设计风格(推荐),还是默认的暖白/炭黑风格?” `repo` 使用产品设计;`default` 使用默认样式包装;`hybrid` 保留产品识别并调整外层排版。用户已经指定时直接执行。普通场景可建议横版、45–60 秒、中文。 +1. **确认范围与三个选择。** 已有工程沿用 brief 和用户决定,只处理本次修改。新片先补齐产品/仓库、版本范围、发布平台、时长与画幅、语言和链接要求,再用一次 `ask_user_question` 问清三件会改变工作量的事:**风格**(`repo` 沿用产品设计/`default` 默认暖白炭黑/`hybrid` 保留产品识别并调整外层排版)、**动效**(界面操作动效 + 镜头 zoom in/out/只做界面动效/只做镜头推拉/全静态)、**旁白**(要 Gemini 画外音/只要字幕与音效/稍后再定)。用户已经指定、或已有工程里记过答案时不要重复问。答案写进 `plan.json` 的 `motion.uiEffects`、`motion.cameraMove` 和 `voiceoverRequired`;用户否决的一项用 `motion.exceptionReason` 或 `voiceoverExceptionReason` 留下依据。普通场景可建议横版、45–60 秒、中文。 2. **初始化后检查环境。** 风格确定后,按下方工具入口初始化独立视频目录,再运行环境检查。仅对报告的缺项加载 [依赖安装](references/onboarding.md) 并补装,随后复查;`ready:true` 继续制作。已有工程直接检查,无需再次初始化。Python 本身缺失时先按依赖安装文档补齐 Python。检查每次执行,安装文档按缺项加载;browser 每次重查环境并实际启动浏览器,`cached:true` 仅表示环境指纹匹配上次成功记录;HyperFrames 指纹匹配时可跳过 doctor。 3. **调查更新并接通组件。** 按 [仓库与风格审计](references/repo-and-style.md) 确定日期/版本、发布状态和 3–5 组核心变化。找到对应业务组件、完整样式和所需状态,先接通一个功能镜头。React 项目的依赖解析、CSS/Tailwind 接入见 [起步工程](references/starter.md),其他挂载路径见 [组件接入](references/component-pipeline.md)。 4. **编排画面与文案。** 按 [分镜与文案](references/story-and-copy.md) 写解释、标题、动作及阅读时间,交替安排字卡、组件特写、工作区和细节。用真实渲染路径输出 3–6 张关键静帧自检;用户要求先看方向时等反馈,否则继续。默认样式可先看 [标题预览](assets/fallback/title-preview.png) 与 [组件预览](assets/fallback/preview.png)。 -5. **完成动效与声音。** 用主时间轴控制组件状态和镜头,支持前后 seek。按 [配乐与音效来源](references/audio-sourcing.md) 为当前影片代码原创配乐,先查找适合产品和动作的音效,缺项才用内置 WAV。按 [混音与验收](references/audio-and-qa.md) 对齐 `audio.cues`、压低关键音效期间的音乐并完成混音。 +5. **完成动效与声音。** 用主时间轴控制组件状态和镜头,支持前后 seek;界面动效与镜头推拉按 [分镜与文案](references/story-and-copy.md) 的镜头语言执行,动效对应真实状态变化,缩放后重新核对变换后的边界。按 [配乐与音效来源](references/audio-sourcing.md) 为当前影片代码原创配乐,先查找适合产品和动作的音效,缺项才用内置 WAV;用户要旁白时用 `scripts/make_voiceover.py` 按镜头逐句生成并量出实际时长。按 [混音与验收](references/audio-and-qa.md) 对齐 `audio.cues` 与 `audio.voiceover`、压低关键音效与人声期间的音乐并完成混音。 6. **验证并交付。** 检查最终 MP4 的裁切、字体、图片、Logo、阅读时间、声音和用户指定的链接处理。修改后重新导出并复查相关镜头。交付 MP4、可复现工程和少量预览;区分自动检查、实际观看/试听及仍受限部分。遇到历史同类问题可查 [案例复盘](references/case-study.md)。 ## 硬约束 @@ -28,6 +28,8 @@ license: "AGPL-3.0; CodePilot fallback assets: BUSL-1.1 (see README.md)" - **原组件。** 功能镜头优先接入实际业务组件、原样式及状态;抽象化用于取景、布局和外层动画。逐镜头核对导入图、来源与静帧。平台确实无法接入时记录阻碍和替代方式,遵从已有授权。 - **清楚排版。** 宣传标题默认有意义的英文与中文各占一个 span、分别指定字体,中文无衬线;中文说明交代对象、动作和结果。原产品内部字体保持其设计;用户指定的语言/字体优先。 - **完整声音。** 默认代码原创音乐与独立动作音效均入轨,关键反馈可闻、音画同步。用户要求静音或省略音效时,记录 `audioExceptionReason`;该字段保存用户依据。 +- **旁白服务于画面。** 用户要旁白时按镜头逐句生成,人声清楚、音乐在其间让位;每句的 `at` 落在所属镜头内,`duration` 用实测值;每句只讲一个有来源的信息点,不照抄标题。用户不要旁白时记录 `voiceoverExceptionReason`,不静默跳过。 +- **动效有依据。** 界面动效对应选中、展开、切换、完成等真实状态变化,镜头推拉服务“要让观众看哪里”;用户选全静态时不要为了“看起来在动”加无意义晃动,并记录 `motion.exceptionReason`。 - **隔离工程。** 源码适配、展示依赖和构建配置放视频工程;原产品代码和依赖保持不动。读取产品已安装依赖,必要时在视频工程固定版本补装适配所需包。 - **授权与真实验收。** 使用素材时保留来源和适用许可。默认样式仍受 [BSL 授权](assets/fallback/SOURCE.md) 约束。自动检查证明结构与文件一致性,视觉、语义和听感由实际审阅补充。 @@ -46,6 +48,7 @@ HyperFrames 工程使用 `--engine hyperframes`。新工程附 10 秒技术样 - [起步工程](references/starter.md):依赖接入、CSS、编译、静帧和导出耗时。 - [配乐源码](assets/audio/codepilot-score-example.py):48 秒、120 BPM、48 kHz 示例,复制进工程按本片改编。 - [内置音效](assets/audio/SOURCE.md):11 个 WAV;需改音色时运行 `scripts/make_sfx.py --output `。 +- `scripts/make_voiceover.py --plan plan.json --output `:按 `audio.voiceover.lines` 逐句生成旁白;key 只从 `EASYROUTER_API_KEY` 或工程 `.env` 读取,`--list-models` 可查实际 TTS 模型 id。 - `scripts/mix_audio.py `:混音、音乐让位、独立音轨和证据报告。 - `scripts/check_delivery.py [--video ] [--mix-report ]`:结构、文件、时间线与媒体验证。 - `python3 -m unittest discover -s /tests`:维护此 skill 时运行回归测试;日常制片无需读取测试源码。 diff --git a/skills/guizang-product-video-skill/agents/openai.yaml b/skills/guizang-product-video-skill/agents/openai.yaml index 36ea491..d848a1b 100644 --- a/skills/guizang-product-video-skill/agents/openai.yaml +++ b/skills/guizang-product-video-skill/agents/openai.yaml @@ -1,4 +1,4 @@ interface: display_name: "归藏 product video skill" short_description: "复用真实产品组件,用代码制作有配乐、有节奏、讲得清楚的软件更新宣传片" - default_prompt: "请使用 $guizang-product-video-skill,为这个产品最近几个版本制作一支宣传片,先确认用代码库风格还是默认风格。" + default_prompt: "请使用 $guizang-product-video-skill,为这个产品最近几个版本制作一支宣传片。先确认三件事:用代码库风格还是默认风格、要不要界面操作动效和镜头推拉、以及要不要画外音。" diff --git a/skills/guizang-product-video-skill/marketplace.json b/skills/guizang-product-video-skill/marketplace.json index 49d8309..00abad0 100644 --- a/skills/guizang-product-video-skill/marketplace.json +++ b/skills/guizang-product-video-skill/marketplace.json @@ -6,15 +6,15 @@ "zh": "产品宣传视频", "en": "Guizang Product Video" }, - "version": "1.0.1", - "usageExample": "我们刚发布了 v2.3.0,更新内容在 CHANGELOG.md 里。帮我把这次更新做成一支 45 秒左右的横版宣传片:复用产品自己的界面组件和配色,配上原创音乐和动作音效,最后交付 MP4 和可复现的工程目录。", + "version": "1.1.0", + "usageExample": "我们刚发布了 v2.3.0,更新内容在 CHANGELOG.md 里。帮我把这次更新做成一支 45 秒左右的横版宣传片:复用产品自己的界面组件和配色,界面操作动效保留、镜头适当推拉,再加一段按镜头逐句生成的旁白,配上原创音乐和动作音效,最后交付 MP4 和可复现的工程目录。", "description": { - "zh": "制作代码驱动的软件版本更新宣传片(release notes video、changelog promo)。从真实更新提炼卖点,复用产品组件和设计语言,完成分镜、代码原创配乐、动作音效、渲染与验收。", - "en": "Produce code-driven software release promo videos (release notes videos, changelog promos). Distill selling points from real updates, reuse the product's own components and design language, then handle storyboarding, original code-generated soundtrack, motion sound effects, rendering and QA." + "zh": "制作代码驱动的软件版本更新宣传片(release notes video、changelog promo)。默认先确认风格、界面操作动效与镜头推拉、是否要 Gemini 画外音,再从真实更新提炼卖点,复用产品组件和设计语言,完成分镜、代码原创配乐、动作音效、逐句旁白、渲染与验收。", + "en": "Produce code-driven software release promo videos (release notes videos, changelog promos). It first confirms style, UI motion and camera zoom, and whether you want a Gemini voiceover, then distills selling points from real updates, reuses the product components and design language, and handles storyboarding, an original code-generated soundtrack, motion sound effects, per-line narration, rendering and QA." } }, "storage": { - "packageKey": "skills/guizang-product-video-skill/guizang-product-video-skill_1.0.1.zip" + "packageKey": "skills/guizang-product-video-skill/guizang-product-video-skill_1.1.0.zip" }, "media": { "icon": { diff --git a/skills/guizang-product-video-skill/references/audio-and-qa.md b/skills/guizang-product-video-skill/references/audio-and-qa.md index 6c6646d..2ed823a 100644 --- a/skills/guizang-product-video-skill/references/audio-and-qa.md +++ b/skills/guizang-product-video-skill/references/audio-and-qa.md @@ -72,6 +72,19 @@ python3 /scripts/mix_audio.py plan.json 5. 连续点击可用 click/click-alt 做轻微音色变化,输入声是有节奏的短簇,转场音只留给画面结构变化,叮咚保留给值得注意的状态。不要所有字出现都“叮”一下。 6. 最后分别听音乐轨、音效轨、完整混音及最终 MP4;检查提示声的第一下是否被盖住、拖尾是否被切断、节奏是否拥挤。 +## 有旁白时人声是锚 + +有旁白时先给人声留位置,再决定音乐和音效。旁白是观众听懂内容的依据,不要让它和配乐互相抢。 + +- 音乐默认在语音期间压低约 6 dB:起音前 150ms 开始压、句尾后约 400ms 恢复,保持时间等于该句实测时长。别给所有句用一个固定 hold。 +- 音效仍要听得见。关键动作的反馈不能被人声吃掉;人声段里正好有关键音效时,先确认它还清楚,必要时错开动作或压短旁白。 +- 句与句之间不要忽大忽小;同一音色、同一增益,不额外做句内响度起伏。 +- 旁白不跨镜头硬切,也不要和字幕抢读的时间。 + +有配音时最终混音的响度目标按 **-14 LUFS**(无配音时用 -16),TP 不高于 -1.5 dBTP、LRA 约 8 作起点,再按平台和听感调整。人声在场时整体响度天然更高,沿用无配音的 -16 会把句子压住。 + +有旁白时 `mix_audio.py` 额外输出 `assets/voice-stem.wav` 供单独听人声轨,混音报告记录人声文件、每句时间与哈希,并把人声窗口计入让位记录。试听顺序:单独听人声轨 → 单独听音效轨 → 听 master → 听编码后的 MP4。 + ## 混音 音乐应听得见但不疲劳;无配音短片可把最终混音约 -16 LUFS、true peak 不高于 -1 至 -1.5 dBTP 当起点,并依据平台要求和听感调整。不同平台/题材不强制同一数值。有配音时给人声留空间,根据听感做 ducking。 @@ -105,6 +118,7 @@ ffmpeg -i renders/final.mp4 -af loudnorm=I=-16:TP=-1.5:LRA=8:print_format=json - | 事实 | 功能状态有证据;示例不冒充客户成果或跑分 | | 组件 | 原业务组件及样式实际上镜;构建图、状态驱动、来源清单与静帧一致;没有以基础控件或 tokens 复刻冒充整项功能复用 | | 音频 | BGM 与关键动作 SFX 都可闻;完成/通知反馈清楚;动作对齐;不刺耳、不爆音、结尾完整 | +| 旁白 | 人声清楚、与字幕和画面一致、不压过关键音效、结尾不被截断;句长与镜头匹配 | | 链接 | 按用户要求隐藏 URL/CTA,遮挡在缩放移动中不泄漏 | | 一致性 | 预览和 MP4 都检查,最终修改确实进入导出文件 | diff --git a/skills/guizang-product-video-skill/references/audio-sourcing.md b/skills/guizang-product-video-skill/references/audio-sourcing.md index c9509bf..3711329 100644 --- a/skills/guizang-product-video-skill/references/audio-sourcing.md +++ b/skills/guizang-product-video-skill/references/audio-sourcing.md @@ -66,6 +66,35 @@ PY 需要调整内置音色时才修改/运行 `scripts/make_sfx.py --output `,不要覆盖已经选定的外部素材。whoosh/sweep 的响亮落点位于文件中间,应试听/检查波形后设置 `syncOffset`;文件开始不等于声音落点。 +## 需要旁白时按句生成 + +需求确认阶段已经问过用户要不要旁白。只有用户要的时候才做,不要默认给每支片子加旁白。 + +旁白用 EasyRouter 上的 Gemini TTS 生成。在 [EasyRouter](https://ezr.sh/) 申请 key 后只放进环境变量 `EASYROUTER_API_KEY` 或视频工程的 `.env`(`.env` 已在忽略规则里)。key 不写进 `plan.json`、证据文件或提交记录,`evidence/voiceover.json` 里只记 `apiKeySource`。没有 key 时如实说明并等用户提供,不假装生成成功,也不要静默换成别的音色或服务。 + +按镜头逐句生成,而不是整段一次生成:每句的起点要贴镜头和卡点,改一句只重录那一句,总长度也不会失控。 + +实测记录(2026-09-21):`https://llm-endpoint.net/v1` 当时返回 102 个模型,其中可用的 TTS 是 `gemini-3.1-flash-tts-preview` 和 `gpt-4o-mini-tts`;不带 `-preview` 的 `gemini-3.1-flash-tts` 会返回 `model_not_found`。脚本走 OpenAI 兼容的 `/audio/speech`,把返回的音频转成 48 kHz 单声道 WAV。模型列表和可用性会变,制作前仍要用 `--list-models` 核对一次。 + +```sh +# 先查网关实际提供的 TTS 模型 id,不要凭记忆写死。 +python3 /scripts/make_voiceover.py --list-models +# 按 plan.json 的 audio.voiceover.lines 逐句生成、拼装并写证据。 +python3 /scripts/make_voiceover.py --plan plan.json --output +``` + +脚本按 `lines[].text` 生成 `assets/voice/line-NN.wav`,按每句的 `at` 拼成 `assets/voice/voiceover.wav`,并把实测时长、文件哈希和可粘回 plan 的 `planSnippet` 写进 `evidence/voiceover.json`。 + +旁白的写法: + +- 是写给人听的句子,不是标题的复述。一句放一到两个信息点,写完念一遍再改。 +- 句长跟着镜头时长走:中文约每秒 6–9 字是预警线,读不完就删词或加长镜头,不要靠加速。 +- `at` 落在所属镜头内,句尾留一点余量;旁白不跨镜头硬切。 +- 只讲有来源的事实,旁白里的数字和效果主张同样要能追溯到变更记录。 +- 音色按产品和受众选(Gemini 预置音色,如 Aoede、Kore、Puck、Charon),一部片子只用一个音色。 + +用户不要旁白时,在 plan 里记 `voiceoverRequired: false` 和 `voiceoverExceptionReason`(写用户的依据),不要一声不响地跳过。 + ## 4. 留下来源,再进入混音 工程 `evidence/audio-selection.json` 简要记录: diff --git a/skills/guizang-product-video-skill/references/onboarding.md b/skills/guizang-product-video-skill/references/onboarding.md index 0dacfdf..7d91b30 100644 --- a/skills/guizang-product-video-skill/references/onboarding.md +++ b/skills/guizang-product-video-skill/references/onboarding.md @@ -86,6 +86,10 @@ npx hyperframes browser ensure 锁定实际安装版本。doctor 按实际缺项解释:本地渲染需要 Node、FFmpeg、FFprobe、Chrome;未选择的 TTS、Whisper、MusicGen、Docker 不构成必装依赖。doctor schema 改变时更新检查适配器,而不是反复安装已存在的软件。只有使用 Docker 渲染时才处理其依赖。框架 CLI 与其他 skill 分开安装,按任务实际需要选择。 +## 旁白(可选) + +画外音不引入新的系统依赖,仍是 Python 与 FFmpeg,只多一个 EasyRouter key(在 https://ezr.sh/ 申请)。key 放环境变量 `EASYROUTER_API_KEY` 或视频工程的 `.env`,不要提交。缺 key 时不要把旁白写成已完成,说明情况并等用户提供;申请 key 不是必装步骤,用户不要旁白时整段可跳过。 + ## 复查 ```sh diff --git a/skills/guizang-product-video-skill/references/story-and-copy.md b/skills/guizang-product-video-skill/references/story-and-copy.md index ea274b7..5d6ec4e 100644 --- a/skills/guizang-product-video-skill/references/story-and-copy.md +++ b/skills/guizang-product-video-skill/references/story-and-copy.md @@ -83,6 +83,16 @@ - 空白或完全静止并非一概错误:短暂停顿用于强调/阅读。警惕无信息的长停顿;不要为了消除静止检测加入无意义晃动。 - 结尾要有完整落版与声音收束。无网站链接时,可以只留下产品名/Logo 与一句结束语。 +## 镜头语言与动效 + +动效范围在需求确认时已经问过,按答案执行,不重复问。整体选择记在 `plan.motion`(`uiEffects`、`cameraMove`),逐镜头的安排记在 `shots[].motion = {ui, camera}`。 + +- **界面操作动效对应真实状态变化。** 选中、展开、切换、输入、完成才给动效,入场一般 0.25–0.6 秒、有明确起止。不要给静态元素套浮动或呼吸动画来制造“在动”的感觉。 +- **镜头推拉服务“让观众看哪里”。** 要看清细节时推近(zoom in),要交代工作区和上下文时拉远(zoom out)。用外层 transform 缩放,幅度克制(约 1.0 → 1.04–1.08),一个镜头一到两次,不要全程持续缩放。 +- **缩放后核对变换后的边界。** 被放大的是组件、文字、遮罩还是整个页面,就按变换后的实际位置检查裁切和遮挡;父容器留了高度不代表 zoom 后的子元素没被裁。 +- **静止不是错误。** 用户选全静态时,用字卡层级、切换时机、停留和声音建立节奏,不要靠晃动补“动感”;全静态同样要记 `motion.exceptionReason`。 +- 交付前按首帧、中间、末帧和每个转场前后检查动效结果;静止检测只用来定位长停留,不作为加动画的理由。 + ## 静帧审阅 从实际代码截取代表镜头,按播放顺序排成一张联系表。核对是否有画面尺度变化、是否一眼知道主次、缩小到手机宽度能否读懂解释。静帧通过只能证明布局方向;节奏、动作连贯和声音仍需看成片。 diff --git a/skills/guizang-product-video-skill/scripts/check_delivery.py b/skills/guizang-product-video-skill/scripts/check_delivery.py index 5cbd59f..f5e49c5 100644 --- a/skills/guizang-product-video-skill/scripts/check_delivery.py +++ b/skills/guizang-product-video-skill/scripts/check_delivery.py @@ -118,6 +118,54 @@ def creative_checks(plan, errors, warnings, project_dir, mix_report, final_video if plan['duration']>=30 and len(kinds)<3:warnings.append('Long promo has fewer than three SFX roles; review sound variety rather than repeating one chime') if required-linked:issue.append('Key actions without SFX: '+', '.join(sorted(required-linked))) if production and final_video and mix_report is None:errors.append('Final video needs --mix-report evidence of BGM + SFX assembly; audio stream existence is insufficient') + motion=plan.get('motion') + if not isinstance(motion,dict): + issue.append('Record the user decision on UI motion and camera zoom in plan.motion');motion={} + elif not all(isinstance(motion.get(key),bool) for key in ['uiEffects','cameraMove']): + issue.append('plan.motion needs boolean uiEffects and cameraMove') + if (motion.get('uiEffects') is False or motion.get('cameraMove') is False) and not nonempty(motion.get('exceptionReason')): + issue.append('Declining UI motion or camera zoom needs motion.exceptionReason with the user basis') + camera_moves=0 + for shot in plan['shots']: + if not isinstance(shot,dict):continue + label=str(shot.get('id','shot')) + declared=shot.get('motion') + if declared is None:continue + if not isinstance(declared,dict):issue.append(label+' motion must be an object');continue + if declared.get('camera') in ['push-in','pull-out']: + camera_moves+=1 + if motion.get('cameraMove') is False:issue.append(label+' declares a camera move the user declined') + if declared.get('ui')=='state-change' and motion.get('uiEffects') is False: + issue.append(label+' declares a UI state animation the user declined') + if motion.get('cameraMove') is True and plan['duration']>=30 and not camera_moves: + warnings.append('Camera zoom in/out was confirmed but no shot records one; review the camera plan or record the change') + if not isinstance(plan.get('voiceoverRequired'),bool): + issue.append('voiceoverRequired must explicitly record the user decision on narration') + elif plan['voiceoverRequired'] is False: + if not nonempty(plan.get('voiceoverExceptionReason')): + issue.append('Declining voiceover needs voiceoverExceptionReason documenting the user request') + else: + voice=audio.get('voiceover') if isinstance(audio.get('voiceover'),dict) else {} + if not nonempty(voice.get('file')):issue.append('Narrated films need audio.voiceover.file') + lines=voice.get('lines') + if not isinstance(lines,list) or not lines: + issue.append('Narrated films need audio.voiceover.lines so every spoken line has evidence') + else: + spans={s.get('id'):(s.get('start'),s.get('end')) for s in plan['shots'] if isinstance(s,dict)} + for index,line in enumerate(lines,start=1): + label='voiceover line '+str(index) + if not isinstance(line,dict):issue.append(label+' must be an object');continue + if not nonempty(line.get('text')):issue.append(label+' needs its spoken text') + if not nonempty(line.get('file')):issue.append(label+' needs its generated file') + if not number(line.get('at')) or not number(line.get('duration')) or line['duration']<=0: + issue.append(label+' needs measured at and duration');continue + span=spans.get(line.get('shot')) + if span is None:issue.append(label+' names an unknown shot');continue + if not all(number(value) for value in span):continue + if line['at']span[1]+.05: + warnings.append(label+' runs outside its shot; align the narration with the picture') + if line['duration']>span[1]-span[0]: + warnings.append(label+' is longer than the shot it belongs to; shorten the copy or lengthen the shot') if mix_report is not None: try: report=json.loads(Path(mix_report).read_text()) @@ -129,8 +177,13 @@ def digest(p):return hashlib.sha256(p.read_bytes()).hexdigest() if sfx_required and not report.get('ducking') and not audio.get('ducking',{}).get('reason'):errors.append('Missing music ducking evidence') for item in report.get('timing',[]): if abs(item.get('beatErrorFrames',0))>2:warnings.append(item['actionId']+' is off its requested beat; adjust picture and audio together') - for entry in [report['master'],report['sfxStem'],report['music'],*([report['musicStem']] if 'musicStem' in report else []),*report['cues']]: - if digest(base/entry['file'])!=entry['sha256']:errors.append('Mix asset hash mismatch: '+entry['file']) + entries=[report['master'],report['sfxStem'],report['music'],*([report['musicStem']] if 'musicStem' in report else []),*([report['voiceStem']] if report.get('voiceStem') else []),*report['cues']] + if plan.get('voiceoverRequired'): + if not report.get('voiceover'):errors.append('Narrated film needs voiceover evidence in the mix report') + else:entries.extend([report['voiceover'],*report['voiceover'].get('lines',[])]) + for entry in entries: + if not isinstance(entry.get('sha256'),str):errors.append('Mix asset is missing its hash: '+str(entry.get('file'))) + elif digest(base/entry['file'])!=entry['sha256']:errors.append('Mix asset hash mismatch: '+entry['file']) warnings.append('Mix inputs/stem/master verified; listen to SFX audibility and verify this master was used in the final export') except (OSError,ValueError,KeyError,TypeError) as exc:errors.append('Invalid mix evidence: '+str(exc)) diff --git a/skills/guizang-product-video-skill/scripts/init_project.py b/skills/guizang-product-video-skill/scripts/init_project.py index 74f387b..6b38c12 100644 --- a/skills/guizang-product-video-skill/scripts/init_project.py +++ b/skills/guizang-product-video-skill/scripts/init_project.py @@ -52,10 +52,15 @@ def main(): shot.update({'descriptionAt':0,'claim':False,'source':[], 'plainExplanation':shot['description'], 'actions':[{'id':shot['id']+'-enter','at':0,'action':'copy enters','soundRequired':False}]}) shot.setdefault('component', None) + shots[0]['motion']={'ui':'none','camera':'hold'} + shots[1]['motion']={'ui':'state-change','camera':'hold'} + shots[2]['motion']={'ui':'none','camera':'hold'} shots[1]['actions'].append({'id':'component-appear','at':0.45,'action':'controls appear','soundRequired':True}) shots[2]['actions'][0]['soundRequired']=True plan = {'demo':True,'product':'软件更新 · 技术样片','style':args.style,'width':1920,'height':1080,'fps':30,'duration':10, 'repo':str(repo) if repo else None,'audioRequired':True,'sfxRequired':True, + 'voiceoverRequired':False,'voiceoverExceptionReason':'技术样片只验证渲染与混音链路,未安排旁白', + 'motion':{'uiEffects':True,'cameraMove':False,'exceptionReason':'技术样片只演示字卡与控件入场,不安排镜头推拉'}, 'typography':{'mode':'bilingual','zhFont':'PingFang SC / Noto Sans CJK SC','enFont':'Georgia','zhStyle':'sans-serif'}, 'audio':{'ducking':{'enabled':True},'music':{'file':'assets/music.wav','gain':0.65},'cues':[ {'at':3.45,'actionId':'component-appear','file':'assets/sfx/click.wav','gain':0.8,'role':'sfx','kind':'click'}, @@ -65,13 +70,13 @@ def main(): (target/'BRIEF.md').write_text(f"""# 视频 brief - 状态:技术起步,尚未完成产品调研与分镜。 -- 风格选择:{args.style}(应来自用户已确认的选择) +- 风格选择:{args.style};动效与旁白:见 plan.json 的 `motion` 与 `voiceoverRequired`(应来自用户已确认的选择) - 仓库:{repo or '未指定,制作真实产品内容前补充'} - 产品 / 更新范围 / 发布状态:待从用户输入和仓库确定。 - 平台 / 画幅 / 时长 / 语言:待记录;plan.json 当前仅为 10 秒技术样片。 - 品牌资源 / 字体:待审计。 - 是否允许链接 / CTA:待记录用户要求。 -- 声音:音乐默认用代码原创;音效先找适合本片的素材,缺项才用 skill 内置 WAV。技术样片尚未配音轨;plan 已分别列出配乐和关键动作音效,须实际准备并混入。 +- 声音:音乐默认用代码原创;音效先找适合本片的素材,缺项才用 skill 内置 WAV;需要旁白时按 `audio.voiceover.lines` 逐句生成并混入。技术样片尚未配音轨;plan 已分别列出配乐和关键动作音效,须实际准备并混入。 - 卖点证据、风格审计与素材来源:记录在 evidence/。 保留原有用户决定;没有确认的字段不要伪装成已确认。正式成片完成后同步 plan.json 与实际时间线。 diff --git a/skills/guizang-product-video-skill/scripts/make_voiceover.py b/skills/guizang-product-video-skill/scripts/make_voiceover.py new file mode 100755 index 0000000..1517833 --- /dev/null +++ b/skills/guizang-product-video-skill/scripts/make_voiceover.py @@ -0,0 +1,293 @@ +#!/usr/bin/env python3 +"""Per-line voiceover through the EasyRouter gateway, assembled into one track. + +Reads audio.voiceover.lines[] from plan.json (text and shot are authored with the copy, 'at' with +the storyboard), synthesises each line, measures the real duration, and assembles +assets/voice/voiceover.wav. Writes evidence/voiceover.json with a planSnippet to paste back into +the plan. The API key is read from the environment or a .env file; it is never written to output. +""" +import argparse +import base64 +import hashlib +import json +import os +import socket +import subprocess +import sys +import tempfile +import time +from pathlib import Path +from urllib.error import HTTPError, URLError +from urllib.request import ProxyHandler, Request, build_opener + +DEFAULT_BASE_URL = 'https://llm-endpoint.net/v1' +DEFAULT_MODEL = 'gemini-3.1-flash-tts-preview' +DEFAULT_VOICE = 'Aoede' +SAMPLE_RATE = 48000 +TRANSPORTS = ['speech', 'chat', 'gemini'] +VOICES = ('Zephyr Puck Charon Kore Fenrir Leda Orus Aoede Callirrhoe Autonoe Enceladus Iapetus ' + 'Umbriel Algieba Despina Erinome Algenib Rasalgethi Laomedeia Achernar Alnilam Schedar ' + 'Gacrux Pulcherrima Achird Zubenelgenubi Vindemiatrix Sadachbia Sadaltager Sulafat').split() + + +def finite(value): + return isinstance(value, (int, float)) and not isinstance(value, bool) + + +def sha256_bytes(data): + return hashlib.sha256(data).hexdigest() + + +def sha256_file(path): + return hashlib.sha256(path.read_bytes()).hexdigest() + + +def open_network(): + for name in ['https_proxy', 'HTTPS_PROXY', 'http_proxy', 'HTTP_PROXY', 'all_proxy', 'ALL_PROXY']: + value = os.environ.get(name) + if value: + return build_opener(ProxyHandler({'http': value, 'https': value})) + for port in (7890, 7897, 1087): + try: + connection = socket.create_connection(('127.0.0.1', port), timeout=1) + except OSError: + continue + connection.close() + proxy = 'http://127.0.0.1:%d' % port + return build_opener(ProxyHandler({'http': proxy, 'https': proxy})) + return build_opener() + + +def resolve_key(project): + value = os.environ.get('EASYROUTER_API_KEY') + if value: + return value, 'env:EASYROUTER_API_KEY' + candidates = [project / '.env', Path(__file__).resolve().parents[1] / '.env'] + for path in candidates: + if not path.is_file(): + continue + for line in path.read_text(encoding='utf-8').splitlines(): + line = line.strip() + if line.startswith('EASYROUTER_API_KEY='): + found = line.split('=', 1)[1].strip().strip('"').strip("'") + if found: + label = path.name if path.parent == project else 'skill-dir/.env' + return found, 'file:' + label + raise ValueError('No EASYROUTER_API_KEY. Put it in the environment or in /.env; ' + 'request one at https://ezr.sh/ and never commit it.') + + +def call(opener, url, key, payload=None, timeout=120): + data = json.dumps(payload).encode('utf-8') if payload is not None else None + headers = {'Authorization': 'Bearer ' + key, 'x-api-key': key, 'Content-Type': 'application/json'} + request = Request(url, data=data, headers=headers, method='POST' if data else 'GET') + try: + with opener.open(request, timeout=timeout) as response: + return response.status, response.read() + except HTTPError as error: + return error.code, error.read() + except URLError as error: + raise RuntimeError('Network error reaching %s: %s' % (url, error.reason)) + + +def excerpt(body): + return body[:300].decode('utf-8', 'replace') + + +def synthesize(opener, base, key, model, voice, text, transport): + """Return (payload, kind) where kind is 'container' for encoded audio or 'pcm' for raw samples.""" + if transport == 'speech': + status, body = call(opener, base + '/audio/speech', key, + {'model': model, 'input': text, 'voice': voice, 'response_format': 'wav'}) + if status != 200 or body[:1] == b'{': + raise RuntimeError('audio/speech returned %s: %s' % (status, excerpt(body))) + return body, 'container' + if transport == 'chat': + status, body = call(opener, base + '/chat/completions', key, + {'model': model, 'messages': [{'role': 'user', 'content': text}], + 'modalities': ['audio'], 'audio': {'voice': voice, 'format': 'wav'}}) + if status != 200: + raise RuntimeError('chat/completions returned %s: %s' % (status, excerpt(body))) + payload = json.loads(body) + choices = payload.get('choices') or [] + audio = (choices[0].get('message', {}).get('audio') or {}) if choices else {} + if not audio.get('data'): + raise RuntimeError('chat/completions returned no audio: ' + excerpt(body)) + return base64.b64decode(audio['data']), 'container' + status, body = call(opener, '%s/v1beta/models/%s:generateContent' % (base, model), key, + {'contents': [{'role': 'user', 'parts': [{'text': text}]}], + 'generationConfig': {'responseModalities': ['AUDIO'], + 'speechConfig': {'voiceConfig': {'prebuiltVoiceConfig': {'voiceName': voice}}}}}) + if status != 200: + raise RuntimeError('generateContent returned %s: %s' % (status, excerpt(body))) + payload = json.loads(body) + candidates = payload.get('candidates') or [] + parts = (candidates[0].get('content', {}).get('parts') or []) if candidates else [] + inline = (parts[0].get('inlineData') or parts[0].get('inline_data') or {}) if parts else {} + if not inline.get('data'): + raise RuntimeError('generateContent returned no audio: ' + excerpt(body)) + return base64.b64decode(inline['data']), 'pcm' + + +def write_wav(payload, kind, target): + with tempfile.TemporaryDirectory(prefix='voiceover-') as tmp: + source = Path(tmp) / 'raw.bin' + source.write_bytes(payload) + for label in ([kind] if kind == 'container' else []) + ['pcm']: + prefix = ['-f', 's16le', '-ar', '24000', '-ac', '1'] if label == 'pcm' else [] + result = subprocess.run(['ffmpeg', '-v', 'error', '-y', *prefix, '-i', str(source), + '-ar', str(SAMPLE_RATE), '-ac', '1', '-c:a', 'pcm_s16le', str(target)], + capture_output=True, text=True) + if result.returncode == 0 and target.is_file() and target.stat().st_size > 1024: + return + raise RuntimeError('Could not decode the returned audio: ' + result.stderr.strip()[:300]) + + +def duration_of(path): + measured = subprocess.run(['ffprobe', '-v', 'error', '-show_entries', 'format=duration', + '-of', 'default=nw=1:nk=1', str(path)], capture_output=True, text=True, check=True) + return float(measured.stdout.strip()) + + +def assemble(lines, duration, target): + args = ['ffmpeg', '-v', 'error', '-y'] + for line in lines: + args += ['-i', str(line['path'])] + filters = ['[%d:a]aresample=%d,adelay=%d:all=1[l%d]' % (index, SAMPLE_RATE, round(line['at'] * 1000), index) + for index, line in enumerate(lines)] + voices = ''.join('[l%d]' % index for index in range(len(lines))) + filters.append(voices + 'amix=inputs=%d:normalize=0,apad,atrim=duration=%s[voice]' % (len(lines), duration)) + args += ['-filter_complex', ';'.join(filters), '-map', '[voice]', '-ar', str(SAMPLE_RATE), '-ac', '1', + '-c:a', 'pcm_s16le', str(target)] + subprocess.run(args, check=True, capture_output=True) + + +def plan_lines(plan): + audio = plan.get('audio') if isinstance(plan.get('audio'), dict) else {} + voiceover = audio.get('voiceover') if isinstance(audio.get('voiceover'), dict) else {} + lines = voiceover.get('lines') + if not isinstance(lines, list) or not lines: + raise ValueError('plan.json has no audio.voiceover.lines; author one entry per spoken line ' + 'with shot, at and text before generating audio') + return voiceover, lines + + +def check_authored_lines(lines, duration): + for position, line in enumerate(lines, start=1): + label = 'line %d' % position + if not isinstance(line, dict): + raise ValueError(label + ' must be an object') + if not isinstance(line.get('text'), str) or not line['text'].strip(): + raise ValueError(label + ' needs the spoken text') + if not isinstance(line.get('shot'), str) or not line['shot'].strip(): + raise ValueError(label + ' needs the shot it belongs to') + if not finite(line.get('at')) or line['at'] < 0: + raise ValueError(label + ' needs an authored start time in "at" seconds') + if finite(duration) and line['at'] >= duration: + raise ValueError(label + ' starts after the film ends') + + +def main(): + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument('--plan', type=Path, default=Path('plan.json')) + parser.add_argument('--output', type=Path, help='Video project directory; defaults to the plan folder') + parser.add_argument('--voice', help='Gemini prebuilt voice name, e.g. Aoede or Kore') + parser.add_argument('--model', help='TTS model id; verify with --list-models instead of guessing') + parser.add_argument('--base-url', help='EasyRouter OpenAI-compatible base URL') + parser.add_argument('--transport', choices=['auto'] + TRANSPORTS, default='auto') + parser.add_argument('--list-models', action='store_true', help='Print the gateway TTS model ids and exit') + parser.add_argument('--force', action='store_true', help='Overwrite existing generated audio') + args = parser.parse_args() + + project = (args.output or args.plan.resolve().parent).expanduser().resolve() + try: + key, key_source = resolve_key(project) + opener = open_network() + if args.list_models: + base = (args.base_url or DEFAULT_BASE_URL).rstrip('/') + status, body = call(opener, base + '/models', key) + if status != 200: + raise RuntimeError('/models returned %s: %s' % (status, excerpt(body))) + ids = [item.get('id', '') for item in json.loads(body).get('data', []) if isinstance(item, dict)] + matches = [model for model in ids if any(token in model.lower() for token in ['tts', 'speech', 'audio'])] + print(json.dumps({'baseUrl': base, 'modelCount': len(ids), 'ttsCandidates': matches}, ensure_ascii=False, indent=2)) + return 0 + plan = json.loads(args.plan.read_text(encoding='utf-8')) + duration = plan.get('duration') + voiceover, lines = plan_lines(plan) + check_authored_lines(lines, duration) + base = (args.base_url or voiceover.get('endpoint') or DEFAULT_BASE_URL).rstrip('/') + model = args.model or voiceover.get('model') or DEFAULT_MODEL + voice = args.voice or voiceover.get('voice') or DEFAULT_VOICE + if voice not in VOICES: + print('Warning: %s is not a known Gemini prebuilt voice; the gateway will accept or reject it.' % voice, + file=sys.stderr) + voice_dir = project / 'assets/voice' + evidence_dir = project / 'evidence' + voice_dir.mkdir(parents=True, exist_ok=True) + evidence_dir.mkdir(parents=True, exist_ok=True) + assembled = voice_dir / 'voiceover.wav' + if not args.force and assembled.is_file(): + raise ValueError('%s exists; pass --force to regenerate, or delete it first' % assembled) + transports = TRANSPORTS if args.transport == 'auto' else [args.transport] + used_transport = None + records = [] + for position, line in enumerate(lines, start=1): + target = voice_dir / ('line-%02d.wav' % position) + if target.is_file() and not args.force: + raise ValueError('%s exists; pass --force to regenerate, or delete it first' % target) + failures = [] + for transport in transports: + try: + payload, kind = synthesize(opener, base, key, model, voice, line['text'], transport) + except (RuntimeError, ValueError) as error: + failures.append('%s: %s' % (transport, error)) + continue + write_wav(payload, kind, target) + used_transport = transport + break + else: + raise RuntimeError('All transports failed for line %d:\n %s' % (position, '\n '.join(failures))) + line_duration = duration_of(target) + records.append({'index': position, 'shot': line['shot'], 'at': line['at'], 'duration': round(line_duration, 3), + 'text': line['text'], 'textSha256': sha256_bytes(line['text'].encode('utf-8')), + 'file': str(target.relative_to(project)), 'sha256': sha256_file(target), + 'sampleRate': SAMPLE_RATE, 'path': target}) + warnings = [] + ordered = sorted(records, key=lambda item: item['at']) + for previous, current in zip(ordered, ordered[1:]): + if previous['at'] + previous['duration'] > current['at'] + 0.01: + raise ValueError('Lines %d and %d overlap; move one "at" later' % (previous['index'], current['index'])) + for record in ordered: + end = record['at'] + record['duration'] + if finite(duration) and end > duration + 0.05: + warnings.append('Line %d ends at %.2fs, past the %.2fs film; shorten it or move it earlier' + % (record['index'], end, duration)) + print('Warning: ' + warnings[-1], file=sys.stderr) + assemble(ordered, duration, assembled) + for record in records: + record.pop('path', None) + snippet = {'file': str(assembled.relative_to(project)), 'gain': voiceover.get('gain', 1.0), + 'provider': 'easyrouter', 'endpoint': base, 'model': model, 'voice': voice, + 'duck': voiceover.get('duck', {'db': 6, 'attack': 0.15, 'release': 0.4}), + 'lines': [{key: record[key] for key in ['shot', 'at', 'duration', 'text', 'file']} for record in records]} + report = {'schema': 1, 'provider': 'easyrouter', 'endpoint': base, 'model': model, 'voice': voice, + 'transport': used_transport, 'apiKeySource': key_source, + 'generatedAt': time.strftime('%Y-%m-%dT%H:%M:%S%z'), 'lines': records, + 'assembled': {'file': str(assembled.relative_to(project)), 'sha256': sha256_file(assembled), + 'duration': round(duration_of(assembled), 3)}, + 'warnings': warnings, 'planSnippet': snippet, + 'listeningStatus': 'Not auditioned by script; listen to each line and the assembled track.'} + (evidence_dir / 'voiceover.json').write_text(json.dumps(report, ensure_ascii=False, indent=2) + '\n') + print(json.dumps({'project': str(project), 'lines': len(records), 'transport': used_transport, + 'assembled': report['assembled'], 'planSnippet': snippet, + 'next': 'Paste planSnippet into plan.json audio.voiceover, then mix and audition.'}, + ensure_ascii=False, indent=2)) + return 0 + except (ValueError, RuntimeError, KeyError, OSError, subprocess.CalledProcessError) as error: + print(str(error), file=sys.stderr) + return 1 + + +if __name__ == '__main__': + sys.exit(main()) diff --git a/skills/guizang-product-video-skill/scripts/mix_audio.py b/skills/guizang-product-video-skill/scripts/mix_audio.py index 9e7302e..2612791 100644 --- a/skills/guizang-product-video-skill/scripts/mix_audio.py +++ b/skills/guizang-product-video-skill/scripts/mix_audio.py @@ -18,7 +18,7 @@ def finite(x):return isinstance(x,(int,float)) and not isinstance(x,bool) and ma 'click': (3,.10,.20), 'click-alt': (3,.10,.20), 'pop': (3.5,.12,.24), 'toggle': (3,.12,.24), 'typing': (2.5,.35,.25), 'whoosh': (4,.16,.30), 'sweep': (4,.20,.35), 'ding-dong': (6,.45,.40), 'success': (5,.35,.35), - 'error': (5,.35,.35), 'resolve': (5,.55,.45), + 'error': (5,.35,.35), 'resolve': (5,.55,.45), 'voice': (6,.40,.40), } def duck_windows(audio, cues, duration): @@ -50,6 +50,53 @@ def duck_windows(audio, cues, duration): 'end':min(duration,center+settings['hold']+settings['release']),'db':settings['db']}) return windows +def voice_lines(base, audio, duration): + """Validate the assembled voiceover and its per-line evidence; return (voiceover, lines).""" + voiceover=audio.get('voiceover') + if voiceover is None:return None,[] + if not isinstance(voiceover,dict):raise ValueError('audio.voiceover must be an object') + if not isinstance(voiceover.get('file'),str) or not (base/voiceover['file']).is_file(): + raise ValueError('Missing assembled voiceover: '+str(voiceover.get('file'))) + gain=voiceover.get('gain',1) + if not finite(gain) or not 0duration+.05:raise ValueError('Voiceover line '+str(index)+' ends after the film') + previous_end=line['at']+line['duration'] + if not isinstance(line.get('file'),str) or not (base/line['file']).is_file(): + raise ValueError('Missing voiceover line file: '+str(line.get('file'))) + return voiceover,lines + +def voice_windows(audio, lines, duration): + """Music gives way to each spoken line; the voice itself is never attenuated.""" + config=audio.get('ducking',{}) + if isinstance(config,dict) and config.get('enabled') is False:return [] + voiceover=audio.get('voiceover',{});override=voiceover.get('duck',{}) if isinstance(voiceover,dict) else {} + if not isinstance(override,dict):raise ValueError('audio.voiceover.duck must be an object') + settings={'db':DUCK_PRESETS['voice'][0],'attack':.15,'release':DUCK_PRESETS['voice'][2]} + settings.update({k:config[k] for k in settings if isinstance(config,dict) and k in config}) + settings.update(override) + for key in ['db','attack','release']: + if not finite(settings.get(key)):raise ValueError('Voice duck parameters must be finite') + if not 0<=settings['db']<=12 or not .005<=settings['attack']<=.5 or not .02<=settings['release']<=2: + raise ValueError('Voice duck envelope outside useful bounds') + if settings['db']==0: + if not settings.get('reason'):raise ValueError('A zero voice duck depth needs a reason') + return [] + windows=[] + for line in lines: + windows.append({'actionId':'voiceover:'+str(line.get('shot','line')),'kind':'voice', + 'start':max(0,line['at']-settings['attack']),'attackEnd':line['at'], + 'holdEnd':min(duration,line['at']+line['duration']), + 'end':min(duration,line['at']+line['duration']+settings['release']),'db':settings['db']}) + return windows + def duck_expression(windows): expressions=[] for w in windows: @@ -69,7 +116,8 @@ def mix(plan_path): duration=plan['duration'];audio=plan.get('audio',{});music=audio.get('music',{});cues=audio.get('cues',[]) if not finite(duration) or duration<=0:raise ValueError('Invalid duration') if not isinstance(cues,list) or not cues:raise ValueError('No SFX cues. A BGM-only master is not a completed sound design.') - sources=[music,*cues] + voiceover,voiceLines=voice_lines(base,audio,duration) + sources=[music,*cues]+([voiceover] if voiceover else []) for item in sources: if not isinstance(item,dict) or not isinstance(item.get('file'),str):raise ValueError('Every music/cue item needs a file') file=(base/item['file']).resolve() @@ -101,8 +149,8 @@ def mix(plan_path): if music_duration+.05 Date: Mon, 21 Sep 2026 16:34:10 +0800 Subject: [PATCH 2/3] fix(guizang-product-video-skill): keep legacy plans valid and make the voiceover gain real - missing motion/voiceoverRequired now warns instead of failing, so 1.0.1-era plans keep passing - mix the assembled audio.voiceover.file with its own gain; per-line files stay evidence and timing - apply lines[].gain while assembling and drop the per-line file requirement from the gate - align the narration wording in SKILL.md with the warning-level gate - add a real narrated-mix regression: gain reaches the stem, -14 LUFS, voice duck window --- skills/guizang-product-video-skill/SKILL.md | 4 +- .../references/audio-and-qa.md | 2 +- .../references/audio-sourcing.md | 3 + .../scripts/check_delivery.py | 17 ++++-- .../scripts/make_voiceover.py | 3 +- .../scripts/mix_audio.py | 13 ++--- .../tests/test_regressions.py | 58 ++++++++++++++++++- 7 files changed, 80 insertions(+), 20 deletions(-) diff --git a/skills/guizang-product-video-skill/SKILL.md b/skills/guizang-product-video-skill/SKILL.md index 58703fd..c6210d0 100644 --- a/skills/guizang-product-video-skill/SKILL.md +++ b/skills/guizang-product-video-skill/SKILL.md @@ -15,7 +15,7 @@ license: "AGPL-3.0; CodePilot fallback assets: BUSL-1.1 (see README.md)" ## 流程 -1. **确认范围与三个选择。** 已有工程沿用 brief 和用户决定,只处理本次修改。新片先补齐产品/仓库、版本范围、发布平台、时长与画幅、语言和链接要求,再用一次 `ask_user_question` 问清三件会改变工作量的事:**风格**(`repo` 沿用产品设计/`default` 默认暖白炭黑/`hybrid` 保留产品识别并调整外层排版)、**动效**(界面操作动效 + 镜头 zoom in/out/只做界面动效/只做镜头推拉/全静态)、**旁白**(要 Gemini 画外音/只要字幕与音效/稍后再定)。用户已经指定、或已有工程里记过答案时不要重复问。答案写进 `plan.json` 的 `motion.uiEffects`、`motion.cameraMove` 和 `voiceoverRequired`;用户否决的一项用 `motion.exceptionReason` 或 `voiceoverExceptionReason` 留下依据。普通场景可建议横版、45–60 秒、中文。 +1. **确认范围与三个选择。** 已有工程沿用 brief 和用户决定,只处理本次修改。新片先补齐产品/仓库、版本范围、发布平台、时长与画幅、语言和链接要求,再用一次 `ask_user_question` 问清三件会改变工作量的事:**风格**(`repo` 沿用产品设计/`default` 默认暖白炭黑/`hybrid` 保留产品识别并调整外层排版)、**动效**(界面操作动效 + 镜头 zoom in/out/只做界面动效/只做镜头推拉/全静态)、**旁白**(要 Gemini 画外音/只要字幕与音效/稍后再定)。用户已经指定、或已有工程里记过答案时不要重复问;旧工程缺 `motion` 或 `voiceoverRequired` 时补问一次写进 plan 再继续(脚本对缺失只告警,对自相矛盾才报错)。答案写进 `plan.json` 的 `motion.uiEffects`、`motion.cameraMove` 和 `voiceoverRequired`;用户否决的一项用 `motion.exceptionReason` 或 `voiceoverExceptionReason` 留下依据。普通场景可建议横版、45–60 秒、中文。 2. **初始化后检查环境。** 风格确定后,按下方工具入口初始化独立视频目录,再运行环境检查。仅对报告的缺项加载 [依赖安装](references/onboarding.md) 并补装,随后复查;`ready:true` 继续制作。已有工程直接检查,无需再次初始化。Python 本身缺失时先按依赖安装文档补齐 Python。检查每次执行,安装文档按缺项加载;browser 每次重查环境并实际启动浏览器,`cached:true` 仅表示环境指纹匹配上次成功记录;HyperFrames 指纹匹配时可跳过 doctor。 3. **调查更新并接通组件。** 按 [仓库与风格审计](references/repo-and-style.md) 确定日期/版本、发布状态和 3–5 组核心变化。找到对应业务组件、完整样式和所需状态,先接通一个功能镜头。React 项目的依赖解析、CSS/Tailwind 接入见 [起步工程](references/starter.md),其他挂载路径见 [组件接入](references/component-pipeline.md)。 4. **编排画面与文案。** 按 [分镜与文案](references/story-and-copy.md) 写解释、标题、动作及阅读时间,交替安排字卡、组件特写、工作区和细节。用真实渲染路径输出 3–6 张关键静帧自检;用户要求先看方向时等反馈,否则继续。默认样式可先看 [标题预览](assets/fallback/title-preview.png) 与 [组件预览](assets/fallback/preview.png)。 @@ -28,7 +28,7 @@ license: "AGPL-3.0; CodePilot fallback assets: BUSL-1.1 (see README.md)" - **原组件。** 功能镜头优先接入实际业务组件、原样式及状态;抽象化用于取景、布局和外层动画。逐镜头核对导入图、来源与静帧。平台确实无法接入时记录阻碍和替代方式,遵从已有授权。 - **清楚排版。** 宣传标题默认有意义的英文与中文各占一个 span、分别指定字体,中文无衬线;中文说明交代对象、动作和结果。原产品内部字体保持其设计;用户指定的语言/字体优先。 - **完整声音。** 默认代码原创音乐与独立动作音效均入轨,关键反馈可闻、音画同步。用户要求静音或省略音效时,记录 `audioExceptionReason`;该字段保存用户依据。 -- **旁白服务于画面。** 用户要旁白时按镜头逐句生成,人声清楚、音乐在其间让位;每句的 `at` 落在所属镜头内,`duration` 用实测值;每句只讲一个有来源的信息点,不照抄标题。用户不要旁白时记录 `voiceoverExceptionReason`,不静默跳过。 +- **旁白服务于画面。** 用户要旁白时按镜头逐句生成,人声清楚、音乐在其间让位;每句的 `at` 默认落在所属镜头内、`duration` 用实测值,越界或超长时脚本给出告警、需要人确认是否有意为之;每句只讲一个有来源的信息点,不照抄标题。用户不要旁白时记录 `voiceoverExceptionReason`,不静默跳过。 - **动效有依据。** 界面动效对应选中、展开、切换、完成等真实状态变化,镜头推拉服务“要让观众看哪里”;用户选全静态时不要为了“看起来在动”加无意义晃动,并记录 `motion.exceptionReason`。 - **隔离工程。** 源码适配、展示依赖和构建配置放视频工程;原产品代码和依赖保持不动。读取产品已安装依赖,必要时在视频工程固定版本补装适配所需包。 - **授权与真实验收。** 使用素材时保留来源和适用许可。默认样式仍受 [BSL 授权](assets/fallback/SOURCE.md) 约束。自动检查证明结构与文件一致性,视觉、语义和听感由实际审阅补充。 diff --git a/skills/guizang-product-video-skill/references/audio-and-qa.md b/skills/guizang-product-video-skill/references/audio-and-qa.md index 2ed823a..75d470b 100644 --- a/skills/guizang-product-video-skill/references/audio-and-qa.md +++ b/skills/guizang-product-video-skill/references/audio-and-qa.md @@ -78,7 +78,7 @@ python3 /scripts/mix_audio.py plan.json - 音乐默认在语音期间压低约 6 dB:起音前 150ms 开始压、句尾后约 400ms 恢复,保持时间等于该句实测时长。别给所有句用一个固定 hold。 - 音效仍要听得见。关键动作的反馈不能被人声吃掉;人声段里正好有关键音效时,先确认它还清楚,必要时错开动作或压短旁白。 -- 句与句之间不要忽大忽小;同一音色、同一增益,不额外做句内响度起伏。 +- 句与句之间不要忽大忽小;同一音色、同一增益,不额外做句内响度起伏。旁白整体音量在 `audio.voiceover.gain`(混音时生效);单句的 `lines[].gain` 在拼装时就已经烘进音轨,混音不会再动它。 - 旁白不跨镜头硬切,也不要和字幕抢读的时间。 有配音时最终混音的响度目标按 **-14 LUFS**(无配音时用 -16),TP 不高于 -1.5 dBTP、LRA 约 8 作起点,再按平台和听感调整。人声在场时整体响度天然更高,沿用无配音的 -16 会把句子压住。 diff --git a/skills/guizang-product-video-skill/references/audio-sourcing.md b/skills/guizang-product-video-skill/references/audio-sourcing.md index 3711329..1eef0aa 100644 --- a/skills/guizang-product-video-skill/references/audio-sourcing.md +++ b/skills/guizang-product-video-skill/references/audio-sourcing.md @@ -85,6 +85,9 @@ python3 /scripts/make_voiceover.py --plan plan.json --output =30 and not camera_moves: warnings.append('Camera zoom in/out was confirmed but no shot records one; review the camera plan or record the change') - if not isinstance(plan.get('voiceoverRequired'),bool): - issue.append('voiceoverRequired must explicitly record the user decision on narration') + if plan.get('voiceoverRequired') is None: + warnings.append('voiceoverRequired is not recorded; ask for the narration decision once and rerun the check') + elif not isinstance(plan['voiceoverRequired'],bool): + issue.append('voiceoverRequired must be a boolean user decision') elif plan['voiceoverRequired'] is False: if not nonempty(plan.get('voiceoverExceptionReason')): issue.append('Declining voiceover needs voiceoverExceptionReason documenting the user request') @@ -156,7 +162,6 @@ def creative_checks(plan, errors, warnings, project_dir, mix_report, final_video label='voiceover line '+str(index) if not isinstance(line,dict):issue.append(label+' must be an object');continue if not nonempty(line.get('text')):issue.append(label+' needs its spoken text') - if not nonempty(line.get('file')):issue.append(label+' needs its generated file') if not number(line.get('at')) or not number(line.get('duration')) or line['duration']<=0: issue.append(label+' needs measured at and duration');continue span=spans.get(line.get('shot')) @@ -180,7 +185,7 @@ def digest(p):return hashlib.sha256(p.read_bytes()).hexdigest() entries=[report['master'],report['sfxStem'],report['music'],*([report['musicStem']] if 'musicStem' in report else []),*([report['voiceStem']] if report.get('voiceStem') else []),*report['cues']] if plan.get('voiceoverRequired'): if not report.get('voiceover'):errors.append('Narrated film needs voiceover evidence in the mix report') - else:entries.extend([report['voiceover'],*report['voiceover'].get('lines',[])]) + else:entries.extend([report['voiceover'],*[line for line in report['voiceover'].get('lines',[]) if line.get('sha256')]]) for entry in entries: if not isinstance(entry.get('sha256'),str):errors.append('Mix asset is missing its hash: '+str(entry.get('file'))) elif digest(base/entry['file'])!=entry['sha256']:errors.append('Mix asset hash mismatch: '+entry['file']) diff --git a/skills/guizang-product-video-skill/scripts/make_voiceover.py b/skills/guizang-product-video-skill/scripts/make_voiceover.py index 1517833..71f5a70 100755 --- a/skills/guizang-product-video-skill/scripts/make_voiceover.py +++ b/skills/guizang-product-video-skill/scripts/make_voiceover.py @@ -153,7 +153,7 @@ def assemble(lines, duration, target): args = ['ffmpeg', '-v', 'error', '-y'] for line in lines: args += ['-i', str(line['path'])] - filters = ['[%d:a]aresample=%d,adelay=%d:all=1[l%d]' % (index, SAMPLE_RATE, round(line['at'] * 1000), index) + filters = ['[%d:a]aresample=%d,volume=%s,adelay=%d:all=1[l%d]' % (index, SAMPLE_RATE, line.get('gain', 1), round(line['at'] * 1000), index) for index, line in enumerate(lines)] voices = ''.join('[l%d]' % index for index in range(len(lines))) filters.append(voices + 'amix=inputs=%d:normalize=0,apad,atrim=duration=%s[voice]' % (len(lines), duration)) @@ -250,6 +250,7 @@ def main(): raise RuntimeError('All transports failed for line %d:\n %s' % (position, '\n '.join(failures))) line_duration = duration_of(target) records.append({'index': position, 'shot': line['shot'], 'at': line['at'], 'duration': round(line_duration, 3), + 'gain': line.get('gain', 1), 'text': line['text'], 'textSha256': sha256_bytes(line['text'].encode('utf-8')), 'file': str(target.relative_to(project)), 'sha256': sha256_file(target), 'sampleRate': SAMPLE_RATE, 'path': target}) diff --git a/skills/guizang-product-video-skill/scripts/mix_audio.py b/skills/guizang-product-video-skill/scripts/mix_audio.py index 2612791..b9c1578 100644 --- a/skills/guizang-product-video-skill/scripts/mix_audio.py +++ b/skills/guizang-product-video-skill/scripts/mix_audio.py @@ -159,14 +159,11 @@ def mix(plan_path): filters.append(''.join(f'[c{i}]' for i in range(len(cues)))+f'amix=inputs={len(cues)}:normalize=0,apad,atrim=duration={duration}[sfx]') # Float stem preserves summed transients until mastering; no early hard clipping. run([*inputs,'-filter_complex',';'.join(filters),'-map','[sfx]','-c:a','pcm_f32le','-ar','48000',str(stem)]) - voice_stem=voice_stem_path if voiceLines else None + voice_stem=voice_stem_path if voiceover else None if voice_stem: - voice_inputs=[];voice_filters=[] - for i,line in enumerate(voiceLines): - voice_inputs += ['-i',str((base/line['file']).resolve())] - voice_filters.append(f"[{i}:a]aresample=48000,aformat=channel_layouts=stereo,volume={line.get('gain',1)},adelay={round(line['at']*1000)}:all=1[v{i}]") - voice_filters.append(''.join(f'[v{i}]' for i in range(len(voiceLines)))+f'amix=inputs={len(voiceLines)}:normalize=0,aformat=channel_layouts=stereo,apad,atrim=duration={duration}[voice]') - run([*voice_inputs,'-filter_complex',';'.join(voice_filters),'-map','[voice]','-c:a','pcm_f32le','-ar','48000',str(voice_stem)]) + # The assembled track is the voice source that enters the mix; lines carry the timing map and the per-line evidence. + voice_filters=f"aresample=48000,aformat=channel_layouts=stereo,volume={voiceover.get('gain',1)},apad,atrim=duration={duration}" + run(['-i',str((base/voiceover['file']).resolve()),'-af',voice_filters,'-c:a','pcm_f32le','-ar','48000',str(voice_stem)]) windows=duck_windows(audio,cues,duration)+voice_windows(audio,voiceLines,duration) envelope=duck_expression(windows) bg_filters=f"aresample=48000,asetnsamples=n=240:p=0,volume='{music.get('gain',1)}*({envelope})':eval=frame,afade=t=in:d=0.025,afade=t=out:st={max(0,duration-.5)}:d=0.5,atrim=duration={duration}" @@ -198,7 +195,7 @@ def mix(plan_path): voiceover_entry=None if voiceover: voiceover_entry={**{k:v for k,v in voiceover.items() if k!='lines'},'sha256':sha((base/voiceover['file']).resolve()), - 'lines':[{**line,'sha256':sha((base/line['file']).resolve())} for line in voiceLines]} + 'lines':[{**line,**({'sha256':sha((base/line['file']).resolve())} if isinstance(line.get('file'),str) and (base/line['file']).is_file() else {})} for line in voiceLines]} report={'planSha256':sha(plan_path),'music':{**music,'sha256':sha(music_path)},'cues':[{**c,'sha256':sha((base/c['file']).resolve())} for c in cues], 'voiceover':voiceover_entry, 'master':{'file':'assets/master.wav','sha256':sha(master)},'sfxStem':{'file':'assets/sfx-stem.wav','sha256':sha(stem)}, diff --git a/skills/guizang-product-video-skill/tests/test_regressions.py b/skills/guizang-product-video-skill/tests/test_regressions.py index df33a8c..c512ca8 100644 --- a/skills/guizang-product-video-skill/tests/test_regressions.py +++ b/skills/guizang-product-video-skill/tests/test_regressions.py @@ -91,9 +91,63 @@ def test_shot_camera_move_conflicts_with_user_choice(self): self.plan['motion']={'uiEffects':True,'cameraMove':False,'exceptionReason':'User asked for a static camera'} self.plan['shots'][0]['motion']={'camera':'push-in'} self.assertTrue(any('camera move the user declined' in x for x in self.errors())) - def test_motion_must_be_recorded(self): + def test_missing_motion_and_narration_only_warn(self): + # Plans made before these questions existed must keep passing; the reminder is a warning. del self.plan['motion'] - self.assertTrue(any('plan.motion' in x for x in self.errors())) + del self.plan['voiceoverRequired'] + del self.plan['voiceoverExceptionReason'] + result=delivery.check(self.plan,project_dir=self.root) + self.assertEqual(result['errors'],[]) + self.assertTrue(any('plan.motion' in item for item in result['warnings'])) + self.assertTrue(any('voiceoverRequired' in item for item in result['warnings'])) + + +def mean_volume(path): + output=subprocess.run(['ffmpeg','-hide_banner','-i',str(path),'-af','volumedetect','-f','null','-'],capture_output=True,text=True).stderr + for line in output.splitlines(): + if 'mean_volume' in line:return float(line.split('mean_volume:')[1].strip().split()[0]) + raise AssertionError('volumedetect reported no mean_volume') + + +class VoiceMix(unittest.TestCase): + """Exercises the real narrated mix path, which the plan-only checks cannot cover.""" + @unittest.skipUnless(shutil.which('ffmpeg') and shutil.which('ffprobe'),'FFmpeg needed for the narrated mix regression') + def test_voice_layer_uses_the_assembled_track_and_its_gain(self): + with tempfile.TemporaryDirectory() as d: + root=Path(d);(root/'assets/sfx').mkdir(parents=True);(root/'assets/voice').mkdir(parents=True) + def tone(path,frequency,duration,channels=1): + subprocess.run(['ffmpeg','-v','error','-y','-f','lavfi','-i','sine=frequency=%d:duration=%s:sample_rate=48000'%(frequency,duration),'-ac',str(channels),str(path)],check=True) + tone(root/'assets/music.wav',300,2,2) + tone(root/'assets/voice/voiceover.wav',700,2) + tone(root/'assets/voice/line-01.wav',700,1) + shutil.copy(ROOT/'assets/audio/sfx/click.wav',root/'assets/sfx/click.wav') + plan={'demo':True,'style':'repo','duration':2,'fps':30,'width':1920,'height':1080,'audioRequired':True,'sfxRequired':True,'voiceoverRequired':True, + 'motion':{'uiEffects':True,'cameraMove':True}, + 'shots':[{'id':'intro','start':0,'end':2,'type':'title','headline':'旁白进混音','claim':False,'source':[], + 'description':'旁白进混音。','descriptionAt':0,'plainExplanation':'旁白进混音。', + 'actions':[{'id':'line-appear','at':0.8,'action':'重点行出现','soundRequired':True}]}], + 'audio':{'ducking':{'enabled':True},'music':{'file':'assets/music.wav','gain':0.6}, + 'voiceover':{'file':'assets/voice/voiceover.wav','gain':1.0, + 'lines':[{'shot':'intro','at':0.2,'duration':1.0,'text':'旁白进混音。','file':'assets/voice/line-01.wav'}]}, + 'cues':[{'at':0.8,'actionId':'line-appear','file':'assets/sfx/click.wav','gain':0.8,'role':'sfx','kind':'click'}]}} + plan_path=root/'plan.json' + def mix_with(gain): + plan['audio']['voiceover']['gain']=gain + plan_path.write_text(json.dumps(plan,ensure_ascii=False)) + subprocess.run([sys.executable,str(ROOT/'scripts/mix_audio.py'),str(plan_path)],capture_output=True,check=True) + report=json.loads((root/'evidence/audio-mix.json').read_text()) + stem=root/('stem-%s.wav'%gain);shutil.copy(root/'assets/voice-stem.wav',stem) + return report,stem + loud,loud_stem=mix_with(1.0) + quiet,quiet_stem=mix_with(0.4) + self.assertTrue(loud.get('voiceStem'),'a narrated mix keeps an isolated voice stem') + self.assertEqual(loud['normalization']['requested']['integratedLufs'],-14) + self.assertIn('voice',{window['kind'] for window in loud['ducking']['windows']}) + self.assertEqual(loud['voiceover']['file'],'assets/voice/voiceover.wav') + self.assertIn('sha256',loud['voiceover']) + self.assertTrue(any(line.get('sha256') for line in loud['voiceover']['lines']),'per-line evidence keeps its hash') + # 1.0 versus 0.4 is about 8 dB, so the overall gain has to reach the mixed voice track. + self.assertLess(mean_volume(quiet_stem),mean_volume(loud_stem)-5) class Preflight(unittest.TestCase): From 16da6a1e46735f8fe81ee15e91bf3f5fc3faa34e Mon Sep 17 00:00:00 2001 From: weijian Date: Mon, 21 Sep 2026 16:51:04 +0800 Subject: [PATCH 3/3] fix(guizang-product-video-skill): keep per-line voice evidence optional --- skills/guizang-product-video-skill/references/audio-and-qa.md | 2 +- skills/guizang-product-video-skill/references/audio-sourcing.md | 2 +- skills/guizang-product-video-skill/scripts/make_voiceover.py | 2 +- skills/guizang-product-video-skill/scripts/mix_audio.py | 2 -- skills/guizang-product-video-skill/tests/test_regressions.py | 2 ++ 5 files changed, 5 insertions(+), 5 deletions(-) diff --git a/skills/guizang-product-video-skill/references/audio-and-qa.md b/skills/guizang-product-video-skill/references/audio-and-qa.md index 75d470b..680896f 100644 --- a/skills/guizang-product-video-skill/references/audio-and-qa.md +++ b/skills/guizang-product-video-skill/references/audio-and-qa.md @@ -79,7 +79,7 @@ python3 /scripts/mix_audio.py plan.json - 音乐默认在语音期间压低约 6 dB:起音前 150ms 开始压、句尾后约 400ms 恢复,保持时间等于该句实测时长。别给所有句用一个固定 hold。 - 音效仍要听得见。关键动作的反馈不能被人声吃掉;人声段里正好有关键音效时,先确认它还清楚,必要时错开动作或压短旁白。 - 句与句之间不要忽大忽小;同一音色、同一增益,不额外做句内响度起伏。旁白整体音量在 `audio.voiceover.gain`(混音时生效);单句的 `lines[].gain` 在拼装时就已经烘进音轨,混音不会再动它。 -- 旁白不跨镜头硬切,也不要和字幕抢读的时间。 +- 旁白默认不跨镜头硬切,也不要和字幕抢读的时间;如果有意跨切点,需要结合画面、字幕和听感人工确认。 有配音时最终混音的响度目标按 **-14 LUFS**(无配音时用 -16),TP 不高于 -1.5 dBTP、LRA 约 8 作起点,再按平台和听感调整。人声在场时整体响度天然更高,沿用无配音的 -16 会把句子压住。 diff --git a/skills/guizang-product-video-skill/references/audio-sourcing.md b/skills/guizang-product-video-skill/references/audio-sourcing.md index 1eef0aa..bf66e9e 100644 --- a/skills/guizang-product-video-skill/references/audio-sourcing.md +++ b/skills/guizang-product-video-skill/references/audio-sourcing.md @@ -92,7 +92,7 @@ python3 /scripts/make_voiceover.py --plan plan.json --output duration+.05:raise ValueError('Voiceover line '+str(index)+' ends after the film') previous_end=line['at']+line['duration'] - if not isinstance(line.get('file'),str) or not (base/line['file']).is_file(): - raise ValueError('Missing voiceover line file: '+str(line.get('file'))) return voiceover,lines def voice_windows(audio, lines, duration): diff --git a/skills/guizang-product-video-skill/tests/test_regressions.py b/skills/guizang-product-video-skill/tests/test_regressions.py index c512ca8..aa92488 100644 --- a/skills/guizang-product-video-skill/tests/test_regressions.py +++ b/skills/guizang-product-video-skill/tests/test_regressions.py @@ -139,6 +139,7 @@ def mix_with(gain): stem=root/('stem-%s.wav'%gain);shutil.copy(root/'assets/voice-stem.wav',stem) return report,stem loud,loud_stem=mix_with(1.0) + (root/'assets/voice/line-01.wav').unlink() quiet,quiet_stem=mix_with(0.4) self.assertTrue(loud.get('voiceStem'),'a narrated mix keeps an isolated voice stem') self.assertEqual(loud['normalization']['requested']['integratedLufs'],-14) @@ -146,6 +147,7 @@ def mix_with(gain): self.assertEqual(loud['voiceover']['file'],'assets/voice/voiceover.wav') self.assertIn('sha256',loud['voiceover']) self.assertTrue(any(line.get('sha256') for line in loud['voiceover']['lines']),'per-line evidence keeps its hash') + self.assertFalse(any(line.get('sha256') for line in quiet['voiceover']['lines']),'missing per-line files remain optional evidence') # 1.0 versus 0.4 is about 8 dB, so the overall gain has to reach the mixed voice track. self.assertLess(mean_volume(quiet_stem),mean_volume(loud_stem)-5)