mirror of
https://github.com/ksyasuda/SubMiner.git
synced 2026-08-26 00:15:27 -07:00
YouTube's sentence-level ASR emits long caption rows with a placeholder d="3000", while the caption actually displays until the next window event. Trusting `d` made long lines vanish mid-speech and leave a blank gap until the next cue. Rolling auto-caption documents (rows with a="1") now end each cue at the next event timestamp instead of t + d, matching YouTube's own display timing. Manual and non-rolling TimedText keep duration-based timing so real silence gaps are preserved.
106 lines
2.9 KiB
TypeScript
106 lines
2.9 KiB
TypeScript
import assert from 'node:assert/strict';
|
|
import test from 'node:test';
|
|
import { convertYoutubeTimedTextToVtt, normalizeYoutubeAutoVtt } from './timedtext';
|
|
|
|
test('convertYoutubeTimedTextToVtt leaves malformed numeric entities literal', () => {
|
|
const result = convertYoutubeTimedTextToVtt(
|
|
'<timedtext><body><p t="0" d="1000">� � A</p></body></timedtext>',
|
|
);
|
|
|
|
assert.equal(
|
|
result,
|
|
['WEBVTT', '', '00:00:00.000 --> 00:00:01.000', '� � A', ''].join('\n'),
|
|
);
|
|
});
|
|
|
|
test('convertYoutubeTimedTextToVtt does not swallow text after zero-length overlap rows', () => {
|
|
const result = convertYoutubeTimedTextToVtt(
|
|
[
|
|
'<timedtext><body>',
|
|
'<p t="0" d="2000">今日は</p>',
|
|
'<p t="1000" d="0">今日はいい天気ですね</p>',
|
|
'<p t="1000" d="2000">今日はいい天気ですね</p>',
|
|
'</body></timedtext>',
|
|
].join(''),
|
|
);
|
|
|
|
assert.equal(
|
|
result,
|
|
[
|
|
'WEBVTT',
|
|
'',
|
|
'00:00:00.000 --> 00:00:00.999',
|
|
'今日は',
|
|
'',
|
|
'00:00:01.000 --> 00:00:03.000',
|
|
'いい天気ですね',
|
|
'',
|
|
].join('\n'),
|
|
);
|
|
});
|
|
|
|
test('convertYoutubeTimedTextToVtt extends rolling captions to the next window event', () => {
|
|
// Real-world shape of YouTube's sentence-level auto captions: window-append
|
|
// filler rows (a="1", sometimes without d) mark the display timeline, while
|
|
// long text rows carry a placeholder d="3000" far shorter than the speech.
|
|
const result = convertYoutubeTimedTextToVtt(
|
|
[
|
|
'<timedtext><body>',
|
|
'<p t="98550" d="3010" w="1" a="1">\n</p>',
|
|
'<p t="98560" d="3000" w="1"><s ac="0">ありがとうって言えないよね。こんなんじゃ。</s></p>',
|
|
'<p t="106950" w="1" a="1">\n</p>',
|
|
'<p t="106960" d="3799" w="1"><s ac="0">私だったら無理だよ。</s></p>',
|
|
'</body></timedtext>',
|
|
].join('\n'),
|
|
);
|
|
|
|
assert.equal(
|
|
result,
|
|
[
|
|
'WEBVTT',
|
|
'',
|
|
'00:01:38.560 --> 00:01:46.950',
|
|
'ありがとうって言えないよね。こんなんじゃ。',
|
|
'',
|
|
'00:01:46.960 --> 00:01:50.759',
|
|
'私だったら無理だよ。',
|
|
'',
|
|
].join('\n'),
|
|
);
|
|
});
|
|
|
|
test('normalizeYoutubeAutoVtt strips cumulative rolling-caption prefixes', () => {
|
|
const result = normalizeYoutubeAutoVtt(
|
|
[
|
|
'WEBVTT',
|
|
'',
|
|
'00:00:01.000 --> 00:00:02.000',
|
|
'今日は',
|
|
'',
|
|
'00:00:02.000 --> 00:00:03.000',
|
|
'今日はいい天気ですね',
|
|
'',
|
|
'00:00:03.000 --> 00:00:04.000',
|
|
'今日はいい天気ですね本当に',
|
|
'',
|
|
].join('\n'),
|
|
);
|
|
|
|
assert.equal(
|
|
result,
|
|
[
|
|
'WEBVTT',
|
|
'',
|
|
'00:00:01.000 --> 00:00:02.000',
|
|
'今日は',
|
|
'',
|
|
'00:00:02.000 --> 00:00:03.000',
|
|
'いい天気ですね',
|
|
'',
|
|
'00:00:03.000 --> 00:00:04.000',
|
|
'本当に',
|
|
'',
|
|
].join('\n'),
|
|
);
|
|
});
|