fix(subtitles): keep overlapping lines that join an already active cue (#221)

This commit is contained in:
2026-08-27 23:13:29 -07:00
committed by GitHub
parent c2c25c0da6
commit 6e945f0872
14 changed files with 430 additions and 132 deletions
+15
View File
@@ -218,3 +218,18 @@ test('removeLiveGlyphFragmentLines leaves ordinary short lines alone', () => {
const text = 'え\nはい。\nそうだな';
assert.equal(removeLiveGlyphFragmentLines(text), text);
});
test('normalizePlainSubtitleText folds cue-boundary blank lines for text consumers', () => {
// The display layer splits on the blank line before normalizing; everyone else --
// tokenizer, cache key, dedup gate, mined sentence -- wants the plain line form.
assert.equal(
normalizePlainSubtitleText('\u4e00\u884c\u76ee\n\n\u4e8c\u884c\u76ee'),
'\u4e00\u884c\u76ee\n\u4e8c\u884c\u76ee',
);
assert.equal(
normalizePlainSubtitleText('\u4e00\u884c\u76ee\n\n\u4e8c\u884c\u76ee', {
collapseLineBreaks: true,
}),
'\u4e00\u884c\u76ee \u4e8c\u884c\u76ee',
);
});
+4
View File
@@ -153,6 +153,10 @@ export function normalizePlainSubtitleText(
);
if (collapseLineBreaks) {
normalized = normalized.replace(/\n/g, ' ').replace(/\s+/g, ' ');
} else {
// Simultaneous cues reach the display layer separated by a blank line; every other
// consumer wants the plain one-break-per-line form.
normalized = normalized.replace(/\n{2,}/g, '\n');
}
return trim ? normalized.trim() : normalized;
+24 -4
View File
@@ -1927,9 +1927,7 @@ test('parseSubtitleCues does not double a line rendered whole beside its glyph s
['わ', 1022],
['ね', 1064],
] as const;
const wholeLine = glyphs
.map(([glyph]) => `{\\an5\\fad(300,500)\\pos(960,50)}${glyph}`)
.join('');
const wholeLine = glyphs.map(([glyph]) => `{\\an5\\fad(300,500)\\pos(960,50)}${glyph}`).join('');
const content = [
...eventsHeader,
`Dialogue: 1,0:00:17.29,0:00:18.99,OP - JP,,0,0,0,,${wholeLine}`,
@@ -1952,7 +1950,7 @@ test('parseSubtitleCues drops a wall of near-invisible positioned texture string
// faint translation is one or two events and stays published.
const content = [
...eventsHeader,
'Dialogue: 90,0:00:12.66,0:00:14.91,Default,,0,0,0,,We\'ll play as a band, and then...',
"Dialogue: 90,0:00:12.66,0:00:14.91,Default,,0,0,0,,We'll play as a band, and then...",
...Array.from(
{ length: 12 },
(_, index) =>
@@ -1997,3 +1995,25 @@ test('parseSubtitleCues keeps hidden events hidden when a transform animates an
['grows into view', 'wipes into view'],
);
});
test('parseAssCues records the vertical band from style alignment, overrides, and \\pos', () => {
const ass = [
'[Script Info]',
'PlayResY: 720',
'',
'[V4+ Styles]',
'Format: Name, Fontname, Fontsize, PrimaryColour, Bold, Alignment, MarginV, Encoding',
'Style: Bottom,Arial,54,&H00FFFFFF,0,2,30,1',
'Style: TopSong,Arial,54,&H00FFFFFF,0,9,12,1',
'',
'[Events]',
'Format: Layer, Start, End, Style, Name, MarginL, MarginR, MarginV, Effect, Text',
'Dialogue: 0,0:00:01.00,0:00:03.00,Bottom,,0,0,0,,\u4e0b\u306e\u30bb\u30ea\u30d5',
'Dialogue: 0,0:00:01.00,0:00:03.00,TopSong,,0,0,0,,\u6b4c\u8a5e\u306e\u884c',
'Dialogue: 0,0:00:01.00,0:00:03.00,Bottom,,0,0,0,,{\\an8}\u4e0a\u66f8\u304d\u306e\u884c',
'Dialogue: 0,0:00:01.00,0:00:03.00,Bottom,,0,0,0,,{\\pos(640,20)}\u770b\u677f\u306e\u884c',
].join('\n');
const bands = parseAssCues(ass).map((cue) => cue.assLayout?.verticalBand);
assert.deepEqual(bands, ['bottom', 'top', 'top', 'top']);
});
+126 -16
View File
@@ -10,10 +10,13 @@ import {
} from './ass-text';
import { hasAssAnimationEvidence, mergeDuplicateCues } from './subtitle-cue-dedup';
/** Vertical third of the screen a cue is authored to occupy. */
export type AssVerticalBand = 'top' | 'middle' | 'bottom';
export type AssCueLayout =
| { kind: 'positioned'; sourceOrder: number; y: number }
| { kind: 'fragment-grid'; sourceOrder: number }
| { kind: 'source-order'; sourceOrder: number };
| { kind: 'positioned'; sourceOrder: number; y: number; verticalBand?: AssVerticalBand }
| { kind: 'fragment-grid'; sourceOrder: number; verticalBand?: AssVerticalBand }
| { kind: 'source-order'; sourceOrder: number; verticalBand?: AssVerticalBand };
export interface SubtitleCue {
startTime: number;
@@ -481,9 +484,7 @@ function structuralOverrideSignature(cue: AnnotatedSubtitleCue): string {
let signature = structuralSignatureCache.get(cue);
if (signature === undefined) {
const names = new Set(
cue.overrides.map(
(command) => `${command.animated ? '~' : ''}${command.name.toLowerCase()}`,
),
cue.overrides.map((command) => `${command.animated ? '~' : ''}${command.name.toLowerCase()}`),
);
signature = [...names].sort().join(',');
structuralSignatureCache.set(cue, signature);
@@ -544,9 +545,7 @@ function buildCoalescedCopy(members: readonly AnnotatedSubtitleCue[]): Annotated
* while the anchor says one glyph. Merging each stack into a single presence spanning
* the union window lets timing clusters see the authored line instead of its phases.
*/
function coalesceAssAnchorCopies(
events: readonly AnnotatedSubtitleCue[],
): AnnotatedSubtitleCue[] {
function coalesceAssAnchorCopies(events: readonly AnnotatedSubtitleCue[]): AnnotatedSubtitleCue[] {
const buckets = new Map<string, number[]>();
const anchorPoints: (AssFragmentPosition[] | null)[] = events.map(() => null);
events.forEach((event, index) => {
@@ -1362,8 +1361,7 @@ function isRepeatedGlyphText(cue: AnnotatedSubtitleCue): boolean {
function isClippedRepeatedGlyphFragment(cue: AnnotatedSubtitleCue): boolean {
return (
isRepeatedGlyphText(cue) &&
(hasStaticOverride(cue, 'clip') || hasStaticOverride(cue, 'iclip'))
isRepeatedGlyphText(cue) && (hasStaticOverride(cue, 'clip') || hasStaticOverride(cue, 'iclip'))
);
}
@@ -2055,6 +2053,111 @@ function recoverCanonicalAssEvents({
);
}
function bandFromNumpadAlignment(alignment: number): AssVerticalBand | null {
if (alignment >= 7 && alignment <= 9) return 'top';
if (alignment >= 4 && alignment <= 6) return 'middle';
if (alignment >= 1 && alignment <= 3) return 'bottom';
return null;
}
// SSA v4 alignment reuses the legacy `\a` codes: 1-3 bottom, +4 top, +8 middle.
function bandFromLegacyAlignment(alignment: number): AssVerticalBand | null {
if (alignment >= 9 && alignment <= 11) return 'middle';
if (alignment >= 5 && alignment <= 7) return 'top';
if (alignment >= 1 && alignment <= 3) return 'bottom';
return null;
}
interface AssPlacementContext {
playResY: number | null;
/** Lowercased style name -> vertical band from the style's Alignment column. */
styleBands: Map<string, AssVerticalBand>;
}
const EMPTY_PLACEMENT_CONTEXT: AssPlacementContext = { playResY: null, styleBands: new Map() };
function parseAssPlacementContext(content: string): AssPlacementContext {
const styleBands = new Map<string, AssVerticalBand>();
let playResY: number | null = null;
let section: 'info' | 'v4plus' | 'v4' | null = null;
let alignmentIndex = -1;
let nameIndex = -1;
for (const line of content.split(/\r?\n/)) {
const trimmed = line.trim();
if (trimmed.startsWith('[') && trimmed.endsWith(']')) {
const sectionName = trimmed.toLowerCase();
section =
sectionName === '[script info]'
? 'info'
: sectionName === '[v4+ styles]'
? 'v4plus'
: sectionName === '[v4 styles]'
? 'v4'
: null;
alignmentIndex = -1;
nameIndex = -1;
continue;
}
if (section === 'info') {
const resMatch = trimmed.match(/^playresy\s*:\s*(\d+(?:\.\d+)?)\s*$/i);
if (resMatch) playResY = Number(resMatch[1]);
continue;
}
if (section !== 'v4plus' && section !== 'v4') continue;
const separator = trimmed.indexOf(':');
if (separator < 0) continue;
const key = trimmed.slice(0, separator).trim().toLowerCase();
const fields = trimmed.slice(separator + 1).split(',');
if (key === 'format') {
const names = fields.map((field) => field.trim().toLowerCase());
alignmentIndex = names.indexOf('alignment');
nameIndex = names.indexOf('name');
continue;
}
if (key !== 'style' || alignmentIndex < 0 || nameIndex < 0) continue;
const styleName = fields[nameIndex]?.trim().toLowerCase();
const alignment = Number(fields[alignmentIndex]?.trim());
if (!styleName || !Number.isFinite(alignment)) continue;
const band =
section === 'v4plus'
? bandFromNumpadAlignment(alignment)
: bandFromLegacyAlignment(alignment);
if (band) styleBands.set(styleName, band);
}
return { playResY, styleBands };
}
/**
* Where on screen mpv will draw this event: an explicit `\pos`/`\move` coordinate when
* the script declares its coordinate space, else an `\an`/`\a` override, else the
* style's Alignment. Constant for the life of the event, which is what lets simultaneous
* lines keep a stable stacking order in the overlay.
*/
function resolveVerticalBand(
overrides: readonly AssOverrideCommand[],
y: number | null,
style: string,
context: AssPlacementContext,
): AssVerticalBand | undefined {
if (y !== null && context.playResY && context.playResY > 0) {
const ratio = y / context.playResY;
return ratio < 1 / 3 ? 'top' : ratio < 2 / 3 ? 'middle' : 'bottom';
}
for (const command of overrides) {
if (command.animated) continue;
const name = command.name.toLowerCase();
if (name !== 'an' && name !== 'a') continue;
const band =
name === 'an'
? bandFromNumpadAlignment(Number(command.args))
: bandFromLegacyAlignment(Number(command.args));
if (band) return band;
}
return context.styleBands.get(style.trim().toLowerCase());
}
function parseAssCoordinate(value: string | undefined): number | null {
if (!value?.trim()) return null;
const coordinate = Number(value.trim());
@@ -2064,6 +2167,8 @@ function parseAssCoordinate(value: string | undefined): number | null {
function buildAssCueLayout(
overrides: readonly AssOverrideCommand[],
sourceOrder: number,
style: string,
placement: AssPlacementContext,
): AssCueLayout {
let y: number | null = null;
for (const command of overrides) {
@@ -2081,14 +2186,18 @@ function buildAssCueLayout(
y = (startY + endY) / 2;
}
}
return y === null
? { kind: 'source-order', sourceOrder }
: { kind: 'positioned', sourceOrder, y };
const verticalBand = resolveVerticalBand(overrides, y, style, placement);
const base: AssCueLayout =
y === null ? { kind: 'source-order', sourceOrder } : { kind: 'positioned', sourceOrder, y };
return verticalBand ? { ...base, verticalBand } : base;
}
function parseAnnotatedAssEvents(content: string): ParsedAssEvents {
const cues: AnnotatedSubtitleCue[] = [];
const comments: AnnotatedSubtitleCue[] = [];
const placement = content.includes('[')
? parseAssPlacementContext(content)
: EMPTY_PLACEMENT_CONTEXT;
const lines = content.split(/\r?\n/);
let inEventsSection = false;
let eventOrder = 0;
@@ -2185,12 +2294,13 @@ function parseAnnotatedAssEvents(content: string): ParsedAssEvents {
const effect = readField(fields, fieldIndex.effect);
const layer = Number(readField(fields, fieldIndex.layer));
const overrides = collectAssOverrideCommands(rawText);
const style = readField(fields, fieldIndex.style);
const cue: AnnotatedSubtitleCue = {
startTime,
endTime,
text,
rawText,
style: readField(fields, fieldIndex.style),
style,
layer: Number.isFinite(layer) ? layer : 0,
name: readField(fields, fieldIndex.name),
effect,
@@ -2198,7 +2308,7 @@ function parseAnnotatedAssEvents(content: string): ParsedAssEvents {
overrides,
overrideSignature: assOverrideSignature(overrides),
order: eventOrder,
assLayout: buildAssCueLayout(overrides, eventOrder),
assLayout: buildAssCueLayout(overrides, eventOrder, style, placement),
};
eventOrder += 1;
if (eventPrefix === ASS_COMMENT_PREFIX) {
@@ -134,7 +134,7 @@ export function createSubtitleProcessingController(
try {
const cachedTokenized = getCachedTokenization(text);
if (cachedTokenized) {
output = cachedTokenized;
output = { ...cachedTokenized, text };
} else {
// Cache miss: show the plain line on time; the tokenized payload
// upgrades it once ready. Skipped on refreshes of an already
@@ -266,7 +266,7 @@ export function createSubtitleProcessingController(
lastEmittedText = text;
lastEmittedGeneration = cacheGeneration;
lastPlainEmittedText = null;
return cached;
return { ...cached, text };
},
hasCachedSubtitle: (text: string) => {
const cacheKey = normalizeSubtitleCacheKey(text);
+17
View File
@@ -84,6 +84,17 @@ function createDeferred<T>() {
};
}
test('tokenizeSubtitle keeps the blank line separating simultaneous cues', async () => {
// The tokenized payload's text drives display; folding the cue boundary would merge
// two speakers back onto one line the moment tokenization upgrades the plain emit.
const result = await tokenizeSubtitle(
'\u4e00\u884c\u76ee\n\n\u4e8c\u884c\u76ee',
makeDeps({ getYomitanExt: () => null }),
);
assert.equal(result.text, '\u4e00\u884c\u76ee\n\n\u4e8c\u884c\u76ee');
});
test('tokenizeSubtitle splits same-line grammar endings before applying annotations', async () => {
const result = await tokenizeSubtitle(
'猫です',
@@ -1682,6 +1693,12 @@ test('tokenizeSubtitle normalizes newlines before Yomitan parse request', async
assert.equal(result.tokens, null);
});
test('tokenizeSubtitle preserves CRLF boundaries between simultaneous cues', async () => {
const result = await tokenizeSubtitle('a\r\n\r\nb', makeDeps());
assert.deepEqual(result, { text: 'a\n\nb', tokens: null });
});
test('tokenizeSubtitle collapses zero-width separators before Yomitan parse request', async () => {
let parseInput = '';
const result = await tokenizeSubtitle(
+9 -1
View File
@@ -887,7 +887,15 @@ export async function tokenizeSubtitle(
text: string,
deps: TokenizerServiceDeps,
): Promise<SubtitleData> {
const displayText = normalizePlainSubtitleText(text);
// Normalize per cue group: the blank line separating simultaneous cues is display
// structure the payload text must keep, or the tokenized upgrade re-merges lines the
// provisional plain emit already showed apart.
const displayText = text
.replace(/\r\n/g, '\n')
.split(/\n{2,}/)
.map((part) => normalizePlainSubtitleText(part))
.filter(Boolean)
.join('\n\n');
// ASS decoding already happened upstream (cue parser for files, mpv for live text), so
// all this drops is whitespace -- but a whitespace-only line still normalizes to empty.