narrate-pipeline.mjs 12 KB

123456789101112131415161718192021222324252627282930313233343536373839404142434445464748495051525354555657585960616263646566676869707172737475767778798081828384858687888990919293949596979899100101102103104105106107108109110111112113114115116117118119120121122123124125126127128129130131132133134135136137138139140141142143144145146147148149150151152153154155156157158159160161162163164165166167168169170171172173174175176177178179180181182183184185186187188189190191192193194195196197198199200201202203204205206207208209210211212213214215216217218219220221222223224225226227228229230231232233234235236237238239240241242243244245246247248249250251252253254255256257258259260261262263264265266267268269270271272273274275276277278279280281282283284285286287288289290291292293294295296297298299300301302303304305306307308309310311312313314315316317318319320321322323324325326327328329330331332333334335336337338339340341342343344345346347348349350351352353354355356357358359360361362363364365366
  1. #!/usr/bin/env node
  2. /**
  3. * narrate-pipeline.mjs · L2 长解说总指挥
  4. *
  5. * 输入:markdown 解说稿(## scene-id 分段,[[cue:id]] 标关键句)
  6. * 输出:voiceover.mp3(拼接好的整段人声)+ timeline.json(每段 start/end + cues 绝对时间)
  7. *
  8. * 用法:
  9. * node scripts/narrate-pipeline.mjs --script demo.md --out-dir _narration_demo
  10. *
  11. * 解说稿格式:
  12. * ---
  13. * title: 什么是 LLM
  14. * voice: S_JSdgdWk22 # 可选,不填走 .env
  15. * speed: 1.0 # 可选
  16. * gap: 0.3 # 段间静音秒数,默认 0.3
  17. * ---
  18. *
  19. * ## intro
  20. * 大家好,我是花叔。今天我们 5 分钟讲清楚 LLM 是什么。
  21. *
  22. * ## what-is
  23. * LLM 全称 Large Language Model,[[cue:bigmodel]]它是一个有几千亿参数的神经网络。
  24. * 本质是一个文字接龙的预测器。
  25. *
  26. * 输出文件结构(out-dir 下):
  27. * audio/
  28. * intro.mp3
  29. * what-is.mp3
  30. * voiceover.mp3 拼接全部 scene 的整段人声
  31. * timeline.json schema 见 references/voiceover-pipeline.md
  32. *
  33. * 依赖:tts-doubao.mjs、ffmpeg、ffprobe
  34. */
  35. import fs from 'node:fs';
  36. import path from 'node:path';
  37. import { execFileSync } from 'node:child_process';
  38. import { fileURLToPath } from 'node:url';
  39. const __dirname = path.dirname(fileURLToPath(import.meta.url));
  40. const SKILL_ROOT = path.resolve(__dirname, '..');
  41. const TTS_SCRIPT = path.join(__dirname, 'cloud', 'tts-doubao.mjs');
  42. function parseArgs(argv) {
  43. const args = {};
  44. for (let i = 2; i < argv.length; i++) {
  45. const a = argv[i];
  46. if (a === '--script') args.script = argv[++i];
  47. else if (a === '--out-dir') args.outDir = argv[++i];
  48. else if (a === '--no-timestamps') args.noTimestamps = true;
  49. else if (a === '--yes') args.yes = true;
  50. else if (a === '--help' || a === '-h') args.help = true;
  51. }
  52. return args;
  53. }
  54. function usage() {
  55. console.error(`
  56. narrate-pipeline.mjs · L2 长解说总指挥
  57. --script <path> 解说稿 .md 文件(必填)
  58. --out-dir <path> 输出目录(必填)
  59. --no-timestamps 不请求字级时间戳(默认请求,chunks 里带 words 供卡拉OK字幕)
  60. --yes 确认将解说稿文本发送到豆包 TTS 官方接口(或设 HUASHU_CLOUD_OK=1)
  61. 输出:<out-dir>/voiceover.mp3 + <out-dir>/timeline.json
  62. `.trim());
  63. process.exit(1);
  64. }
  65. /**
  66. * Parse frontmatter + scene blocks from markdown
  67. * Returns { meta, scenes: [{ id, raw }] }
  68. */
  69. function parseScript(md) {
  70. const meta = {};
  71. let body = md;
  72. const fmMatch = md.match(/^---\n([\s\S]*?)\n---\n/);
  73. if (fmMatch) {
  74. for (const line of fmMatch[1].split('\n')) {
  75. const idx = line.indexOf(':');
  76. if (idx < 0) continue;
  77. const key = line.slice(0, idx).trim();
  78. const val = line.slice(idx + 1).trim();
  79. meta[key] = val;
  80. }
  81. body = md.slice(fmMatch[0].length);
  82. }
  83. const scenes = [];
  84. const re = /^##\s+([\w-]+)\s*\n([\s\S]*?)(?=^##\s+[\w-]+\s*\n|$(?![\r\n]))/gm;
  85. let m;
  86. while ((m = re.exec(body)) !== null) {
  87. scenes.push({ id: m[1], raw: m[2].trim() });
  88. }
  89. return { meta, scenes };
  90. }
  91. /**
  92. * Split a scene's text by [[cue:id]] markers into chunks.
  93. * Returns: { chunks: [{ text, cueAfter? }] }
  94. * cueAfter is the cue id that follows this chunk (chunk's end = cue position)
  95. *
  96. * Example: "A[[cue:x]]B[[cue:y]]C" =>
  97. * chunks: [
  98. * { text: "A", cueAfter: "x" },
  99. * { text: "B", cueAfter: "y" },
  100. * { text: "C" }
  101. * ]
  102. */
  103. function splitByCues(text) {
  104. const chunks = [];
  105. const re = /\[\[cue:([\w-]+)\]\]/g;
  106. let lastIdx = 0;
  107. let m;
  108. while ((m = re.exec(text)) !== null) {
  109. const before = text.slice(lastIdx, m.index).trim();
  110. chunks.push({ text: before, cueAfter: m[1] });
  111. lastIdx = m.index + m[0].length;
  112. }
  113. const tail = text.slice(lastIdx).trim();
  114. chunks.push({ text: tail });
  115. // 过滤空文本块(cue 紧贴段首/段尾时)
  116. return chunks.filter((c) => c.text.length > 0 || c.cueAfter);
  117. }
  118. function getDuration(filePath) {
  119. const out = execFileSync('ffprobe', [
  120. '-v', 'error',
  121. '-show_entries', 'format=duration',
  122. '-of', 'default=noprint_wrappers=1:nokey=1',
  123. filePath,
  124. ], { encoding: 'utf8' });
  125. return parseFloat(out.trim());
  126. }
  127. let timestampsBroken = false; // 时间戳请求失败一次后,后续 chunk 全部降级,避免反复重试
  128. function callTTS(text, outPath, opts) {
  129. // 同意门已在本管线入口过(见 main),子进程直接带 --yes
  130. const args = ['--text', text, '--out', outPath, '--yes'];
  131. if (opts.voice) args.push('--voice', opts.voice);
  132. if (opts.speed) args.push('--speed', String(opts.speed));
  133. const wantTimestamps = opts.timestamps && !timestampsBroken;
  134. if (wantTimestamps) args.push('--timestamps');
  135. try {
  136. const out = execFileSync('node', [TTS_SCRIPT, ...args], {
  137. encoding: 'utf8',
  138. stdio: ['ignore', 'pipe', 'inherit'],
  139. });
  140. return JSON.parse(out.trim());
  141. } catch (e) {
  142. if (!wantTimestamps) throw e;
  143. // 字级时间戳可能不被当前音色/资源支持(仅 2.0 资源+中英文)——降级重试,不带时间戳
  144. timestampsBroken = true;
  145. console.error('[narrate] ⚠ 带 --timestamps 的 TTS 失败,降级为无时间戳模式(timeline 不含 words,卡拉OK字幕不可用)');
  146. const out = execFileSync('node', [TTS_SCRIPT, ...args.filter((a) => a !== '--timestamps')], {
  147. encoding: 'utf8',
  148. stdio: ['ignore', 'pipe', 'inherit'],
  149. });
  150. return JSON.parse(out.trim());
  151. }
  152. }
  153. function ffmpegConcat(inputs, output) {
  154. // 用 concat demuxer 合并相同编码的 mp3
  155. const listFile = output + '.list';
  156. fs.writeFileSync(
  157. listFile,
  158. inputs.map((p) => `file '${p.replace(/'/g, "'\\''")}'`).join('\n'),
  159. );
  160. execFileSync(
  161. 'ffmpeg',
  162. ['-y', '-f', 'concat', '-safe', '0', '-i', listFile, '-c', 'copy', output],
  163. { stdio: ['ignore', 'pipe', 'pipe'] },
  164. );
  165. fs.unlinkSync(listFile);
  166. }
  167. function makeSilence(duration, outPath) {
  168. execFileSync(
  169. 'ffmpeg',
  170. ['-y', '-f', 'lavfi', '-i', 'anullsrc=r=24000:cl=mono', '-t', String(duration),
  171. '-q:a', '9', '-acodec', 'libmp3lame', outPath],
  172. { stdio: ['ignore', 'pipe', 'pipe'] },
  173. );
  174. }
  175. async function main() {
  176. const args = parseArgs(process.argv);
  177. if (args.help || !args.script || !args.outDir) usage();
  178. if (!args.yes && process.env.HUASHU_CLOUD_OK !== '1') {
  179. console.error(
  180. '[云能力确认] 本管线会把解说稿文本分段发送到豆包TTS官方接口(openspeech.bytedance.com,' +
  181. '使用你自己的key合成配音)。\n确认无误请重跑并加 --yes,或设置环境变量 HUASHU_CLOUD_OK=1。' +
  182. '数据流向声明见 SECURITY.md。',
  183. );
  184. process.exit(2);
  185. }
  186. const scriptPath = path.resolve(args.script);
  187. const outDir = path.resolve(args.outDir);
  188. const audioDir = path.join(outDir, 'audio');
  189. const tmpDir = path.join(outDir, '.tmp');
  190. fs.mkdirSync(audioDir, { recursive: true });
  191. fs.mkdirSync(tmpDir, { recursive: true });
  192. const md = fs.readFileSync(scriptPath, 'utf8');
  193. const { meta, scenes } = parseScript(md);
  194. if (scenes.length === 0) {
  195. console.error('错:解说稿没有 ## scene 段,至少一段。');
  196. process.exit(1);
  197. }
  198. const voice = meta.voice || undefined;
  199. const speed = meta.speed ? parseFloat(meta.speed) : 1.0;
  200. const gap = meta.gap ? parseFloat(meta.gap) : 0.3;
  201. const timestamps = !args.noTimestamps && meta.timestamps !== 'false';
  202. console.error(`[narrate] script=${path.basename(scriptPath)} scenes=${scenes.length} voice=${voice || '(env)'} speed=${speed} gap=${gap}s`);
  203. // 段间静音文件(共用一个)
  204. const gapFile = path.join(tmpDir, 'gap.mp3');
  205. if (gap > 0) makeSilence(gap, gapFile);
  206. const timeline = {
  207. title: meta.title || path.basename(scriptPath, '.md'),
  208. voice: voice || null,
  209. speed,
  210. gap,
  211. totalDuration: 0,
  212. scenes: [],
  213. };
  214. let cursor = 0;
  215. const sceneAudioFiles = [];
  216. for (let i = 0; i < scenes.length; i++) {
  217. const scene = scenes[i];
  218. console.error(`[narrate] (${i + 1}/${scenes.length}) scene="${scene.id}"`);
  219. const chunks = splitByCues(scene.raw);
  220. const chunkFiles = [];
  221. const cueRecords = [];
  222. const chunkRecords = []; // 每个 chunk 的实测 start/end 段内时间,用于字幕显示
  223. let sceneInternalCursor = 0;
  224. for (let j = 0; j < chunks.length; j++) {
  225. const chunk = chunks[j];
  226. if (!chunk.text) {
  227. // 空文本块(cue 紧贴),跳过 TTS 但仍记录 cue 位置
  228. if (chunk.cueAfter) {
  229. cueRecords.push({
  230. id: chunk.cueAfter,
  231. offset: sceneInternalCursor,
  232. });
  233. }
  234. continue;
  235. }
  236. const chunkPath = path.join(tmpDir, `${scene.id}-${j}.mp3`);
  237. const result = callTTS(chunk.text, chunkPath, { voice, speed, timestamps });
  238. const chunkStart = sceneInternalCursor;
  239. chunkFiles.push(chunkPath);
  240. sceneInternalCursor += result.duration;
  241. chunkRecords.push({
  242. text: chunk.text,
  243. start: chunkStart,
  244. end: sceneInternalCursor,
  245. duration: result.duration,
  246. // 字级时间戳(TTS 实测,TN 后文本):换算成段内相对时间
  247. words: (result.words || []).map((w) => ({
  248. text: w.text,
  249. start: chunkStart + w.start,
  250. end: chunkStart + w.end,
  251. })),
  252. });
  253. console.error(` chunk ${j}: ${result.duration.toFixed(2)}s · ${chunk.text.length} 字 · ${chunk.text.slice(0, 30)}${chunk.text.length > 30 ? '…' : ''}`);
  254. if (chunk.cueAfter) {
  255. cueRecords.push({
  256. id: chunk.cueAfter,
  257. offset: sceneInternalCursor,
  258. });
  259. }
  260. }
  261. // 合并段内子段
  262. const sceneAudio = path.join(audioDir, `${scene.id}.mp3`);
  263. if (chunkFiles.length === 1) {
  264. fs.copyFileSync(chunkFiles[0], sceneAudio);
  265. } else {
  266. ffmpegConcat(chunkFiles, sceneAudio);
  267. }
  268. const sceneDuration = getDuration(sceneAudio);
  269. // 拼接到总轨:先加 gap(除了第一段),再加 scene
  270. if (i > 0 && gap > 0) {
  271. sceneAudioFiles.push(gapFile);
  272. cursor += gap;
  273. }
  274. sceneAudioFiles.push(sceneAudio);
  275. timeline.scenes.push({
  276. id: scene.id,
  277. start: cursor,
  278. end: cursor + sceneDuration,
  279. duration: sceneDuration,
  280. audio: path.relative(outDir, sceneAudio),
  281. text: scene.raw.replace(/\[\[cue:[\w-]+\]\]/g, ''),
  282. // chunks: 用于字幕逐句显示。start/end 是段内相对时间,absoluteStart/absoluteEnd 是整轨绝对时间
  283. // words: 字级时间戳(卡拉OK字幕用;TN 后文本,可能与 chunk.text 不完全一致)。空数组=不可用
  284. chunks: chunkRecords.map((c) => ({
  285. text: c.text,
  286. start: c.start,
  287. end: c.end,
  288. absoluteStart: cursor + c.start,
  289. absoluteEnd: cursor + c.end,
  290. words: (c.words || []).map((w) => ({
  291. text: w.text,
  292. start: w.start,
  293. end: w.end,
  294. absoluteStart: cursor + w.start,
  295. absoluteEnd: cursor + w.end,
  296. })),
  297. })),
  298. cues: cueRecords.map((c) => ({
  299. id: c.id,
  300. offset: c.offset,
  301. absoluteTime: cursor + c.offset,
  302. })),
  303. });
  304. cursor += sceneDuration;
  305. }
  306. // 合并整轨
  307. const voiceoverPath = path.join(outDir, 'voiceover.mp3');
  308. ffmpegConcat(sceneAudioFiles, voiceoverPath);
  309. timeline.totalDuration = getDuration(voiceoverPath);
  310. timeline.voiceover = 'voiceover.mp3';
  311. fs.writeFileSync(
  312. path.join(outDir, 'timeline.json'),
  313. JSON.stringify(timeline, null, 2),
  314. );
  315. // 清理 tmp
  316. fs.rmSync(tmpDir, { recursive: true, force: true });
  317. console.error(`\n[narrate] 完成。`);
  318. console.error(` voiceover: ${voiceoverPath}`);
  319. console.error(` timeline: ${path.join(outDir, 'timeline.json')}`);
  320. console.error(` 总时长: ${timeline.totalDuration.toFixed(2)}s (${(timeline.totalDuration / 60).toFixed(2)} min)`);
  321. console.error(` 段数: ${timeline.scenes.length}`);
  322. const totalCues = timeline.scenes.reduce((sum, s) => sum + s.cues.length, 0);
  323. console.error(` cue 数: ${totalCues}`);
  324. const totalWords = timeline.scenes.reduce(
  325. (sum, s) => sum + s.chunks.reduce((a, c) => a + (c.words ? c.words.length : 0), 0), 0);
  326. console.error(` 字级时间戳: ${totalWords > 0 ? `${totalWords} words(<Subtitles karaoke /> 可用)` : '无'}`);
  327. }
  328. main().catch((err) => {
  329. console.error(`narrate-pipeline 失败:${err.message}`);
  330. console.error(err.stack);
  331. process.exit(1);
  332. });