ソースを参照

feat: 卡拉OK逐字字幕——豆包TTS v3字级时间戳(enable_subtitle)接入narration pipeline

- tts-doubao.mjs --timestamps / narrate-pipeline timeline.json新增words[](纯增量)
- Subtitles组件karaoke模式(默认关,读到即变橙,无transition保seek确定性)
- 顺手修存量bug:.env v1 endpoint致500、S_音色误推icl-1.0致403

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
alchain 2 ヶ月 前
親
コミット
05b9d893f1

+ 54 - 2
assets/narration_stage.jsx

@@ -403,12 +403,64 @@ const NarrationStageLib = (() => {
    *   maxLen    单行最大视觉长度,默认 13
    *
    * 深底场景:把 color 改成 '#fff',haloColor 改成 'rgba(0,0,0,0.85)' 即可。
+   *
+   * 卡拉OK模式(字级高亮,需 timeline chunks 里带 words——narrate-pipeline.mjs 默认输出):
+   *   karaoke       true 开启,默认 false。整行显示,读到哪个字哪个字变色
+   *   karaokeColor  已读字的颜色,默认品牌橙 '#e8590c'
+   *   chunk 没有 words 数据时自动回落到普通 chunk 模式,调用方不用做判断。
+   *   注意:words 是 TN 后文本("2025"→"二零二五"),卡拉OK行直接由 words 拼出,
+   *   保证高亮与发音严格对齐(与 chunk.text 原文可能有差异)。
    */
-  function Subtitles({ bottom = 90, fontSize = 32, color = '#1a1a1a', haloColor = 'rgba(245,241,232,0.9)', maxLen = 13 } = {}) {
+  function splitWordsToLines(words, maxLen = 13) {
+    // 把字级时间戳 token 贪心打包成 ≤maxLen 的行;强标点(。!?)后强制换行,绝不跨句号
+    const lines = [];
+    let cur = [];
+    let curLen = 0;
+    for (const w of words) {
+      const wLen = visualLen(w.text);
+      if (cur.length > 0 && curLen + wLen > maxLen) { lines.push(cur); cur = []; curLen = 0; }
+      cur.push(w);
+      curLen += wLen;
+      if (/[。!?]\s*$/.test(w.text)) { lines.push(cur); cur = []; curLen = 0; }
+    }
+    if (cur.length > 0) lines.push(cur);
+    return lines;
+  }
+
+  function Subtitles({ bottom = 90, fontSize = 32, color = '#1a1a1a', haloColor = 'rgba(245,241,232,0.9)', maxLen = 13, karaoke = false, karaokeColor = '#e8590c' } = {}) {
     const { time, scene } = React.useContext(NarrationContext);
     if (!scene || !scene.chunks) return null;
     const active = scene.chunks.find(c => time >= c.absoluteStart && time < c.absoluteEnd);
     if (!active) return null;
+
+    // —— 卡拉OK模式:整行显示 + 逐字高亮(读到即变色,无 CSS transition,seek 渲染确定性)——
+    if (karaoke && active.words && active.words.length > 0) {
+      const wordLines = splitWordsToLines(active.words, maxLen);
+      let activeWLine = wordLines[0];
+      for (const ln of wordLines) {
+        if (time >= ln[0].absoluteStart) activeWLine = ln;
+        else break;
+      }
+      const lineStart = activeWLine[0].absoluteStart;
+      const lineProg = Math.max(0, Math.min(1, (time - (lineStart - 0.15)) / 0.15)); // 行提前 0.15s 淡入
+      return React.createElement('div', {
+        style: { position: 'absolute', left: 0, right: 0, bottom, display: 'flex', justifyContent: 'center', pointerEvents: 'none', zIndex: 50 },
+      }, React.createElement('div', {
+        key: lineStart,
+        style: {
+          fontFamily: '"PingFang SC", "Noto Sans SC", -apple-system, sans-serif',
+          fontSize, fontWeight: 600,
+          letterSpacing: '0.04em', lineHeight: 1.2, textAlign: 'center',
+          textShadow: `0 0 6px ${haloColor}, 0 0 12px ${haloColor}, 0 1px 2px rgba(255,255,255,0.5)`,
+          opacity: lineProg, transform: `translateY(${(1 - lineProg) * 4}px)`,
+        },
+      }, activeWLine.map((w, i) => React.createElement('span', {
+        key: i,
+        style: { color: time >= w.absoluteStart ? karaokeColor : color },
+      }, w.text))));
+    }
+
+    // —— 普通 chunk 模式(原有行为,不变)——
     const lines = splitChunkToLines(active.text, maxLen);
     if (lines.length === 0) return null;
     const totalLen = lines.reduce((s, l) => s + visualLen(l), 0);
@@ -468,7 +520,7 @@ const NarrationStageLib = (() => {
     return Math.max(0, v);
   }
 
-  return { NarrationStage, Scene, Cue, useNarration, useSceneFade, Subtitles, splitChunkToLines };
+  return { NarrationStage, Scene, Cue, useNarration, useSceneFade, Subtitles, splitChunkToLines, splitWordsToLines };
 })();
 
 if (typeof window !== 'undefined') {

+ 16 - 0
references/voiceover-pipeline.md

@@ -240,6 +240,11 @@ LLM 全称 Large Language Model,[[cue:bigmodel]]它是一个有几千亿参数
           end: number,
           absoluteStart: number,   // 整轨绝对时间(对齐 voiceover.mp3)
           absoluteEnd: number,
+          // words: 字级时间戳(TTS enable_subtitle 实测返回,默认带;--no-timestamps 关闭)
+          // 注意 text 是 TN 后文本("2025"→"二零二五"),标点附在前一个字上
+          words: [
+            { text: string, start: number, end: number, absoluteStart: number, absoluteEnd: number }
+          ],
         }
       ],
       cues: [
@@ -284,6 +289,17 @@ const { NarrationStage, Subtitles } = NarrationStageLib;
 
 `<Subtitles />` 默认按以上规则跑,不需要传 props。深底场景:`<Subtitles color="#fff" haloColor="rgba(0,0,0,0.85)" />`。
 
+### 卡拉OK模式(字级高亮)
+
+```jsx
+<Subtitles karaoke />                          {/* 读到哪个字哪个字变品牌橙 #e8590c */}
+<Subtitles karaoke karaokeColor="#0a84ff" />   {/* 自定义高亮色 */}
+```
+
+- 依赖 timeline chunks 里的 `words` 字级时间戳(narrate-pipeline.mjs 默认输出;豆包 TTS v3 `enable_subtitle`,需 2.0 资源,仅中英文)
+- 整行显示、逐字变色,行切分复用 ≤maxLen + 不跨句号规则(由 words 拼行,与发音严格对齐)
+- chunk 没有 words 时自动回落普通 chunk 模式,调用方无需判断
+
 ### 切句算法(已在 narration_stage.jsx 内置)
 
 ```js

+ 42 - 6
scripts/narrate-pipeline.mjs

@@ -48,6 +48,7 @@ function parseArgs(argv) {
     const a = argv[i];
     if (a === '--script') args.script = argv[++i];
     else if (a === '--out-dir') args.outDir = argv[++i];
+    else if (a === '--no-timestamps') args.noTimestamps = true;
     else if (a === '--help' || a === '-h') args.help = true;
   }
   return args;
@@ -59,6 +60,7 @@ narrate-pipeline.mjs · L2 长解说总指挥
 
   --script <path>     解说稿 .md 文件(必填)
   --out-dir <path>    输出目录(必填)
+  --no-timestamps     不请求字级时间戳(默认请求,chunks 里带 words 供卡拉OK字幕)
 
 输出:<out-dir>/voiceover.mp3 + <out-dir>/timeline.json
 `.trim());
@@ -130,15 +132,31 @@ function getDuration(filePath) {
   return parseFloat(out.trim());
 }
 
+let timestampsBroken = false; // 时间戳请求失败一次后,后续 chunk 全部降级,避免反复重试
+
 function callTTS(text, outPath, opts) {
   const args = ['--text', text, '--out', outPath];
   if (opts.voice) args.push('--voice', opts.voice);
   if (opts.speed) args.push('--speed', String(opts.speed));
-  const out = execFileSync('node', [TTS_SCRIPT, ...args], {
-    encoding: 'utf8',
-    stdio: ['ignore', 'pipe', 'inherit'],
-  });
-  return JSON.parse(out.trim());
+  const wantTimestamps = opts.timestamps && !timestampsBroken;
+  if (wantTimestamps) args.push('--timestamps');
+  try {
+    const out = execFileSync('node', [TTS_SCRIPT, ...args], {
+      encoding: 'utf8',
+      stdio: ['ignore', 'pipe', 'inherit'],
+    });
+    return JSON.parse(out.trim());
+  } catch (e) {
+    if (!wantTimestamps) throw e;
+    // 字级时间戳可能不被当前音色/资源支持(仅 2.0 资源+中英文)——降级重试,不带时间戳
+    timestampsBroken = true;
+    console.error('[narrate] ⚠ 带 --timestamps 的 TTS 失败,降级为无时间戳模式(timeline 不含 words,卡拉OK字幕不可用)');
+    const out = execFileSync('node', [TTS_SCRIPT, ...args.filter((a) => a !== '--timestamps')], {
+      encoding: 'utf8',
+      stdio: ['ignore', 'pipe', 'inherit'],
+    });
+    return JSON.parse(out.trim());
+  }
 }
 
 function ffmpegConcat(inputs, output) {
@@ -183,6 +201,7 @@ async function main() {
   const voice = meta.voice || undefined;
   const speed = meta.speed ? parseFloat(meta.speed) : 1.0;
   const gap = meta.gap ? parseFloat(meta.gap) : 0.3;
+  const timestamps = !args.noTimestamps && meta.timestamps !== 'false';
 
   console.error(`[narrate] script=${path.basename(scriptPath)} scenes=${scenes.length} voice=${voice || '(env)'} speed=${speed} gap=${gap}s`);
 
@@ -225,7 +244,7 @@ async function main() {
         continue;
       }
       const chunkPath = path.join(tmpDir, `${scene.id}-${j}.mp3`);
-      const result = callTTS(chunk.text, chunkPath, { voice, speed });
+      const result = callTTS(chunk.text, chunkPath, { voice, speed, timestamps });
       const chunkStart = sceneInternalCursor;
       chunkFiles.push(chunkPath);
       sceneInternalCursor += result.duration;
@@ -234,6 +253,12 @@ async function main() {
         start: chunkStart,
         end: sceneInternalCursor,
         duration: result.duration,
+        // 字级时间戳(TTS 实测,TN 后文本):换算成段内相对时间
+        words: (result.words || []).map((w) => ({
+          text: w.text,
+          start: chunkStart + w.start,
+          end: chunkStart + w.end,
+        })),
       });
       console.error(`  chunk ${j}: ${result.duration.toFixed(2)}s · ${chunk.text.length} 字 · ${chunk.text.slice(0, 30)}${chunk.text.length > 30 ? '…' : ''}`);
       if (chunk.cueAfter) {
@@ -268,12 +293,20 @@ async function main() {
       audio: path.relative(outDir, sceneAudio),
       text: scene.raw.replace(/\[\[cue:[\w-]+\]\]/g, ''),
       // chunks: 用于字幕逐句显示。start/end 是段内相对时间,absoluteStart/absoluteEnd 是整轨绝对时间
+      // words: 字级时间戳(卡拉OK字幕用;TN 后文本,可能与 chunk.text 不完全一致)。空数组=不可用
       chunks: chunkRecords.map((c) => ({
         text: c.text,
         start: c.start,
         end: c.end,
         absoluteStart: cursor + c.start,
         absoluteEnd: cursor + c.end,
+        words: (c.words || []).map((w) => ({
+          text: w.text,
+          start: w.start,
+          end: w.end,
+          absoluteStart: cursor + w.start,
+          absoluteEnd: cursor + w.end,
+        })),
       })),
       cues: cueRecords.map((c) => ({
         id: c.id,
@@ -306,6 +339,9 @@ async function main() {
   console.error(`  段数:      ${timeline.scenes.length}`);
   const totalCues = timeline.scenes.reduce((sum, s) => sum + s.cues.length, 0);
   console.error(`  cue 数:    ${totalCues}`);
+  const totalWords = timeline.scenes.reduce(
+    (sum, s) => sum + s.chunks.reduce((a, c) => a + (c.words ? c.words.length : 0), 0), 0);
+  console.error(`  字级时间戳: ${totalWords > 0 ? `${totalWords} words(<Subtitles karaoke /> 可用)` : '无'}`);
 }
 
 main().catch((err) => {

+ 27 - 4
scripts/tts-doubao.mjs

@@ -5,10 +5,14 @@
  * 用法:
  *   node scripts/tts-doubao.mjs --text "你好" --out demo.mp3
  *   node scripts/tts-doubao.mjs --text-file script.txt --out out.mp3 --speed 1.0
+ *   node scripts/tts-doubao.mjs --text "你好" --out demo.mp3 --timestamps   # 附带字级时间戳
  *
  * 输出:
  *   - mp3 文件写到 --out 路径
  *   - stdout 打印一行 JSON: {"path":"...","duration":12.34,"bytes":54321}
+ *   - 带 --timestamps 时额外含 words: [{text,start,end,confidence}](秒,相对本段音频开头)
+ *     注意:时间戳文本是 TN 后文本(如 "2025" 会变成 "二零二五"),标点附在前一个字上;
+ *     需要 2.0 资源(seed-tts-2.0 / seed-icl-2.0),仅中英文。
  *
  * 依赖:Node 18+(自带 fetch/crypto)、ffprobe(测时长,brew install ffmpeg)
  *
@@ -59,6 +63,7 @@ function parseArgs(argv) {
     else if (a === '--speed') args.speed = argv[++i];
     else if (a === '--voice') args.voice = argv[++i];
     else if (a === '--encoding') args.encoding = argv[++i];
+    else if (a === '--timestamps') args.timestamps = true;
     else if (a === '--help' || a === '-h') args.help = true;
   }
   return args;
@@ -74,6 +79,7 @@ tts-doubao.mjs · 豆包语音 TTS
   --speed <float>       语速倍率,默认 1.0(0.5-2.0)
   --voice <voice_id>    覆盖 .env 里的音色 id
   --encoding <ext>      mp3 / wav / pcm,默认 mp3
+  --timestamps          请求字级时间戳(enable_subtitle),结果 JSON 多一个 words 数组
 `.trim());
   process.exit(1);
 }
@@ -93,7 +99,9 @@ function getDuration(filePath) {
 }
 
 function inferResourceId(voiceId) {
-  if (voiceId.startsWith('S_')) return 'seed-icl-1.0';
+  // 复刻音色默认走 2.0:本账号只开通了 seed-icl-2.0(1.0 会 403 resource not granted),
+  // 且字级时间戳(enable_subtitle)只有 2.0 资源支持。
+  if (voiceId.startsWith('S_')) return 'seed-icl-2.0';
   if (voiceId.includes('uranus')) return 'seed-tts-2.0';
   return 'seed-tts-1.0';
 }
@@ -130,6 +138,7 @@ function buildAuthHeaders({ requestId, resourceId }) {
 async function readV3Audio(res) {
   const text = await res.text();
   const chunks = [];
+  const words = []; // 字级时间戳(enable_subtitle 开启时服务端按句返回 sentence.words)
   let finalCode = null;
   let finalMessage = '';
 
@@ -154,16 +163,26 @@ async function readV3Audio(res) {
       throw new Error(`API 返回错误 code=${code} msg=${json.message || JSON.stringify(json)}`);
     }
     if (json.data) chunks.push(Buffer.from(json.data, 'base64'));
+    if (json.sentence && Array.isArray(json.sentence.words)) {
+      for (const w of json.sentence.words) {
+        words.push({
+          text: w.word,
+          start: w.startTime,
+          end: w.endTime,
+          confidence: w.confidence,
+        });
+      }
+    }
   }
 
   if (!chunks.length) {
     const detail = finalCode ? `结束码 ${finalCode} ${finalMessage}` : text.slice(0, 500);
     throw new Error(`API 响应无音频数据:${detail}`);
   }
-  return Buffer.concat(chunks);
+  return { audio: Buffer.concat(chunks), words };
 }
 
-async function tts({ text, voice, speed, encoding }) {
+async function tts({ text, voice, speed, encoding, timestamps }) {
   const endpoint = process.env.DOUBAO_TTS_ENDPOINT || 'https://openspeech.bytedance.com/api/v3/tts/unidirectional';
   const voiceId = voice || process.env.DOUBAO_TTS_VOICE_ID || process.env.DOUBAO_SPEAKER;
   const resourceId = process.env.DOUBAO_TTS_RESOURCE_ID || inferResourceId(voiceId || '');
@@ -180,6 +199,8 @@ async function tts({ text, voice, speed, encoding }) {
         format: encoding,
         sample_rate: 24000,
         speech_rate: speedToSpeechRate(speed),
+        // 字级时间戳:仅 2.0 资源(seed-tts-2.0 / seed-icl-2.0)支持,中英文 only
+        ...(timestamps ? { enable_subtitle: true } : {}),
       },
     },
   };
@@ -218,11 +239,12 @@ async function main() {
   const outPath = path.resolve(args.out);
   fs.mkdirSync(path.dirname(outPath), { recursive: true });
 
-  const audio = await tts({
+  const { audio, words } = await tts({
     text,
     voice: args.voice,
     speed: args.speed,
     encoding: args.encoding,
+    timestamps: args.timestamps,
   });
 
   fs.writeFileSync(outPath, audio);
@@ -233,6 +255,7 @@ async function main() {
     duration,
     text_chars: text.length,
   };
+  if (args.timestamps) result.words = words;
   console.log(JSON.stringify(result));
 }