mirror of
https://github.com/ksyasuda/SubMiner.git
synced 2026-08-25 12:15:26 -07:00
2269 lines
81 KiB
TypeScript
2269 lines
81 KiB
TypeScript
import {
|
||
assOverrideSignature,
|
||
assToPlainText,
|
||
collectAssOverrideCommands,
|
||
hasAssTemporalOverride,
|
||
parseAssEffectField,
|
||
removeAssControlDebrisLines,
|
||
type AssEffectKind,
|
||
type AssOverrideCommand,
|
||
} from './ass-text';
|
||
import { hasAssAnimationEvidence, mergeDuplicateCues } from './subtitle-cue-dedup';
|
||
|
||
export type AssCueLayout =
|
||
| { kind: 'positioned'; sourceOrder: number; y: number }
|
||
| { kind: 'fragment-grid'; sourceOrder: number }
|
||
| { kind: 'source-order'; sourceOrder: number };
|
||
|
||
export interface SubtitleCue {
|
||
startTime: number;
|
||
endTime: number;
|
||
text: string;
|
||
/** How a complete line was recovered from generated ASS animation events. */
|
||
source?: 'canonical-ass' | 'reconstructed-ass';
|
||
/**
|
||
* Full span of the generated animation events a recovered cue replaced. Entrance and
|
||
* exit frames can run past canonical authored timing.
|
||
*/
|
||
animationStartTime?: number;
|
||
animationEndTime?: number;
|
||
/** ASS style retained only for fragment-reconstructed lines. */
|
||
assStyle?: string;
|
||
/** Authored ASS ordering metadata used when flattening simultaneous positioned cues. */
|
||
assLayout?: AssCueLayout;
|
||
}
|
||
|
||
/**
|
||
* Everything the parser knows about a source event, shared only with the dedup engine.
|
||
* Deduplication needs the authoring context -- which style the line belongs to, which
|
||
* override commands it carries, whether the `Effect` column was set -- to tell a karaoke
|
||
* burst apart from two characters saying the same word in turn. None of it is meaningful
|
||
* outside the parser, so the public API exposes only timing, text, and the optional
|
||
* recovery marker used by live subtitle consumers.
|
||
*/
|
||
export interface AnnotatedSubtitleCue extends SubtitleCue {
|
||
/** Text exactly as authored, override blocks and all. */
|
||
rawText: string;
|
||
style: string;
|
||
layer: number;
|
||
/** ASS `Name`/`Actor` column. */
|
||
name: string;
|
||
/** ASS `Effect` column, verbatim. */
|
||
effect: string;
|
||
effectKind: AssEffectKind;
|
||
/** Override commands found in `{...}` blocks, with their arguments. */
|
||
overrides: readonly AssOverrideCommand[];
|
||
/** Canonical form of `overrides`, for spotting values that change across a run. */
|
||
overrideSignature: string;
|
||
/** Position in the source file, so sorting by time stays deterministic across layers. */
|
||
order: number;
|
||
}
|
||
|
||
export type SubtitleSourceFormat = 'ass' | 'srt';
|
||
|
||
const HTML_SUBTITLE_TAG_PATTERN = /<\/?[A-Za-z][^>\n]*>/g;
|
||
|
||
const SRT_TIMING_PATTERN =
|
||
/^\s*(?:(\d{1,2}):)?(\d{2}):(\d{2})[,.](\d{1,3})\s*-->\s*(?:(\d{1,2}):)?(\d{2}):(\d{2})[,.](\d{1,3})/;
|
||
|
||
function parseTimestamp(
|
||
hours: string | undefined,
|
||
minutes: string,
|
||
seconds: string,
|
||
millis: string,
|
||
): number {
|
||
return (
|
||
Number(hours || 0) * 3600 +
|
||
Number(minutes) * 60 +
|
||
Number(seconds) +
|
||
Number(millis.padEnd(3, '0')) / 1000
|
||
);
|
||
}
|
||
|
||
/**
|
||
* The single ASS decode for the file path: cues leave the parser as plain text with real
|
||
* line breaks, matching what mpv hands over for the same line played live. No layer
|
||
* downstream decodes ASS again.
|
||
*/
|
||
function decodeSubtitleCueText(text: string): string {
|
||
return assToPlainText(text, '\n').replace(HTML_SUBTITLE_TAG_PATTERN, '');
|
||
}
|
||
|
||
function sanitizeSubtitleCueText(text: string): string {
|
||
return decodeSubtitleCueText(text).trim();
|
||
}
|
||
|
||
function sanitizeAssCueText(text: string): string {
|
||
return removeAssControlDebrisLines(decodeSubtitleCueText(text)).trim();
|
||
}
|
||
|
||
function attachAssLayout<T extends SubtitleCue>(cue: T, assLayout: AssCueLayout | undefined): T {
|
||
if (assLayout) {
|
||
Object.defineProperty(cue, 'assLayout', { value: assLayout, enumerable: false });
|
||
}
|
||
return cue;
|
||
}
|
||
|
||
function toPublicCues(cues: AnnotatedSubtitleCue[]): SubtitleCue[] {
|
||
return cues.map(
|
||
({
|
||
startTime,
|
||
endTime,
|
||
text,
|
||
source,
|
||
animationStartTime,
|
||
animationEndTime,
|
||
style,
|
||
assLayout,
|
||
}) => {
|
||
const common = {
|
||
startTime,
|
||
endTime,
|
||
text,
|
||
};
|
||
if (source === 'reconstructed-ass') {
|
||
return attachAssLayout(
|
||
{
|
||
...common,
|
||
source,
|
||
animationStartTime,
|
||
animationEndTime,
|
||
assStyle: style,
|
||
},
|
||
assLayout,
|
||
);
|
||
}
|
||
return attachAssLayout(
|
||
source ? { ...common, source, animationStartTime, animationEndTime } : common,
|
||
assLayout,
|
||
);
|
||
},
|
||
);
|
||
}
|
||
|
||
function parseAnnotatedSrtCues(content: string): AnnotatedSubtitleCue[] {
|
||
const cues: AnnotatedSubtitleCue[] = [];
|
||
const lines = content.split(/\r?\n/);
|
||
let i = 0;
|
||
|
||
while (i < lines.length) {
|
||
const line = lines[i]!;
|
||
const timingMatch = SRT_TIMING_PATTERN.exec(line);
|
||
if (!timingMatch) {
|
||
i += 1;
|
||
continue;
|
||
}
|
||
|
||
const startTime = parseTimestamp(
|
||
timingMatch[1],
|
||
timingMatch[2]!,
|
||
timingMatch[3]!,
|
||
timingMatch[4]!,
|
||
);
|
||
const endTime = parseTimestamp(
|
||
timingMatch[5],
|
||
timingMatch[6]!,
|
||
timingMatch[7]!,
|
||
timingMatch[8]!,
|
||
);
|
||
|
||
i += 1;
|
||
const textLines: string[] = [];
|
||
while (i < lines.length && lines[i]!.trim() !== '') {
|
||
textLines.push(lines[i]!);
|
||
i += 1;
|
||
}
|
||
|
||
const rawText = textLines.join('\n');
|
||
const text = sanitizeSubtitleCueText(rawText);
|
||
if (text) {
|
||
cues.push({
|
||
startTime,
|
||
endTime,
|
||
text,
|
||
rawText,
|
||
style: '',
|
||
layer: 0,
|
||
name: '',
|
||
effect: '',
|
||
effectKind: 'none',
|
||
// SRT and VTT carry no authoring metadata, and the dedup engine never reads
|
||
// overrides for those formats -- collecting them would be parsing for nobody.
|
||
overrides: [],
|
||
overrideSignature: '',
|
||
order: cues.length,
|
||
});
|
||
}
|
||
}
|
||
|
||
return cues;
|
||
}
|
||
|
||
export function parseSrtCues(content: string): SubtitleCue[] {
|
||
return toPublicCues(parseAnnotatedSrtCues(content));
|
||
}
|
||
|
||
const ASS_TIMING_PATTERN = /^(\d+):(\d{2}):(\d{2})\.(\d{1,2})$/;
|
||
const ASS_FORMAT_PREFIX = 'Format:';
|
||
const ASS_DIALOGUE_PREFIX = 'Dialogue:';
|
||
const ASS_COMMENT_PREFIX = 'Comment:';
|
||
const ASS_NAME_FIELD_ALIASES = ['name', 'actor'];
|
||
const CANONICAL_MATCH_MARGIN_SECONDS = 1;
|
||
const MIN_CANONICAL_ANIMATION_EVENTS = 3;
|
||
// A tiny animated fragment can itself be composed from still smaller glyph events. It is
|
||
// not enough evidence that the fragment represents an authored line boundary.
|
||
const MIN_CANONICAL_DIALOGUE_TEXT_LENGTH = 4;
|
||
const MIN_FRAGMENT_LINE_EVENTS = 8;
|
||
const MIN_FRAGMENT_LINE_PARTS = 4;
|
||
const MAX_FRAGMENT_MEDIAN_LENGTH = 4;
|
||
const MAX_FRAGMENT_LINE_TIMING_VARIANCE_SECONDS = 2;
|
||
const MAX_FRAGMENT_LINE_VERTICAL_SPAN = 48;
|
||
|
||
function parseAssTimestamp(raw: string): number | null {
|
||
const match = ASS_TIMING_PATTERN.exec(raw.trim());
|
||
if (!match) {
|
||
return null;
|
||
}
|
||
const hours = Number(match[1]);
|
||
const minutes = Number(match[2]);
|
||
const seconds = Number(match[3]);
|
||
const centiseconds = Number(match[4]!.padEnd(2, '0'));
|
||
return hours * 3600 + minutes * 60 + seconds + centiseconds / 100;
|
||
}
|
||
|
||
function readField(fields: string[], index: number): string {
|
||
return index >= 0 && index < fields.length ? fields[index]!.trim() : '';
|
||
}
|
||
|
||
function findFieldIndex(formatFields: string[], aliases: string[]): number {
|
||
for (const alias of aliases) {
|
||
const index = formatFields.indexOf(alias);
|
||
if (index >= 0) {
|
||
return index;
|
||
}
|
||
}
|
||
return -1;
|
||
}
|
||
|
||
interface ParsedAssEvents {
|
||
dialogue: AnnotatedSubtitleCue[];
|
||
comments: AnnotatedSubtitleCue[];
|
||
}
|
||
|
||
// Every candidate line re-reads the compacted text of each event in its window, so on
|
||
// fragment-heavy scripts the same event compacts thousands of times without this cache.
|
||
const compactMatchTextCache = new WeakMap<AnnotatedSubtitleCue, string>();
|
||
|
||
function compactAssMatchText(text: string): string {
|
||
return text.replace(/\s+/gu, '');
|
||
}
|
||
|
||
function compactCueMatchText(cue: AnnotatedSubtitleCue): string {
|
||
let compact = compactMatchTextCache.get(cue);
|
||
if (compact === undefined) {
|
||
compact = compactAssMatchText(cue.text);
|
||
compactMatchTextCache.set(cue, compact);
|
||
}
|
||
return compact;
|
||
}
|
||
|
||
function assEventGroupKey(cue: AnnotatedSubtitleCue): string {
|
||
return `${cue.style}\0${cue.name}`;
|
||
}
|
||
|
||
/**
|
||
* Windowed lookup over one style/name group. Every candidate line queries its time
|
||
* neighborhood, and fragment-heavy scripts put thousands of candidates in one group, so
|
||
* a linear rescan per candidate is quadratic in practice. Events are sorted by start
|
||
* once; `prefixMaxEnd` lets the backward walk stop as soon as no earlier event can still
|
||
* reach the window.
|
||
*/
|
||
interface AssEventGroupIndex {
|
||
byStart: AnnotatedSubtitleCue[];
|
||
prefixMaxEnd: number[];
|
||
}
|
||
|
||
function buildAssEventGroupIndex(events: readonly AnnotatedSubtitleCue[]): AssEventGroupIndex {
|
||
const byStart = [...events].sort((a, b) => a.startTime - b.startTime || a.order - b.order);
|
||
const prefixMaxEnd: number[] = [];
|
||
let maxEnd = -Infinity;
|
||
for (const event of byStart) {
|
||
maxEnd = Math.max(maxEnd, event.endTime);
|
||
prefixMaxEnd.push(maxEnd);
|
||
}
|
||
return { byStart, prefixMaxEnd };
|
||
}
|
||
|
||
/** Group events overlapping `[startTime, endTime]`, returned in source order. */
|
||
function eventsOverlappingWindow(
|
||
index: AssEventGroupIndex,
|
||
startTime: number,
|
||
endTime: number,
|
||
): AnnotatedSubtitleCue[] {
|
||
const { byStart, prefixMaxEnd } = index;
|
||
let low = 0;
|
||
let high = byStart.length;
|
||
while (low < high) {
|
||
const mid = (low + high) >>> 1;
|
||
if (byStart[mid]!.startTime <= endTime) {
|
||
low = mid + 1;
|
||
} else {
|
||
high = mid;
|
||
}
|
||
}
|
||
const matches: AnnotatedSubtitleCue[] = [];
|
||
for (let i = low - 1; i >= 0 && prefixMaxEnd[i]! >= startTime; i -= 1) {
|
||
if (byStart[i]!.endTime >= startTime) {
|
||
matches.push(byStart[i]!);
|
||
}
|
||
}
|
||
return matches.sort((a, b) => a.order - b.order);
|
||
}
|
||
|
||
interface FragmentGroup {
|
||
text: string;
|
||
events: AnnotatedSubtitleCue[];
|
||
}
|
||
|
||
// Anchor extraction walks every override command, and the canonical/fragment passes
|
||
// consult the same events once per candidate neighborhood, so heavy KFX tags make the
|
||
// uncached form quadratic in practice.
|
||
const placementAnchorCache = new WeakMap<AnnotatedSubtitleCue, Set<string>>();
|
||
|
||
function fragmentPlacementAnchors(event: AnnotatedSubtitleCue): Set<string> {
|
||
let anchors = placementAnchorCache.get(event);
|
||
if (anchors) {
|
||
return anchors;
|
||
}
|
||
anchors = new Set<string>();
|
||
for (const command of event.overrides) {
|
||
const name = command.name.toLowerCase();
|
||
const args = command.args.split(',').map((value) => value.trim());
|
||
if (name === 'pos' && args.length >= 2) {
|
||
anchors.add(`pos:${args[0]},${args[1]}`);
|
||
} else if (name === 'move' && args.length >= 4) {
|
||
anchors.add(`move:${args[0]},${args[1]}`);
|
||
anchors.add(`move:${args[2]},${args[3]}`);
|
||
}
|
||
}
|
||
placementAnchorCache.set(event, anchors);
|
||
return anchors;
|
||
}
|
||
|
||
// Drop-shadow layer copies sit a few pixels off their base glyph, while even tightly
|
||
// kerned repeated glyphs in one line ("ii") measure 10px apart or more.
|
||
const LAYER_COPY_OFFSET_TOLERANCE_PX = 6;
|
||
|
||
// Copies chain only through genuine time overlap. Consecutive re-runs of one visual
|
||
// (chant bursts, countdown frames, jitter animation frames) abut or micro-overlap at
|
||
// frame seams, so the different-structure threshold sits above a frame seam while the
|
||
// phases of one effect (a highlight and the exit ghost it launches) overlap far longer.
|
||
const MIN_PHASE_OVERLAP_SECONDS = 0.04;
|
||
|
||
// Decoration is timed to the line it accompanies, but a lead-in echo can end exactly
|
||
// where the recovered line's first sung copy begins; a small tolerance keeps such
|
||
// flush decoration attached to its line.
|
||
const DECORATION_SPAN_TOLERANCE_SECONDS = 0.1;
|
||
|
||
// Guards for pathological event volumes. Real copy stacks and repaint chains stay in
|
||
// the hundreds; a same-text bucket or candidate sweep group in the thousands is a
|
||
// particle field, and the quadratic passes over it would stall the main process.
|
||
const MAX_COALESCE_BUCKET_EVENTS = 1500;
|
||
const MAX_SWEEP_GROUP_EVENTS = 4000;
|
||
// Fragment-layer collapsing repeatedly rescans the remaining parts after each match.
|
||
// Genuine authored lines stay far below this limit; larger groups are particle fields.
|
||
const MAX_FRAGMENT_COLLAPSE_PARTS = 1500;
|
||
|
||
/**
|
||
* Sources behind a coalesced copy chain. Synthetic cues stand in for their sources
|
||
* during fragment recovery, but suppression and copy-count evidence must reach the
|
||
* original events, which are what the published dialogue list still holds.
|
||
*/
|
||
const coalescedSourceEvents = new WeakMap<AnnotatedSubtitleCue, AnnotatedSubtitleCue[]>();
|
||
|
||
function sourceEventsOf(cue: AnnotatedSubtitleCue): readonly AnnotatedSubtitleCue[] {
|
||
return coalescedSourceEvents.get(cue) ?? [cue];
|
||
}
|
||
|
||
function sourceEventCount(events: readonly AnnotatedSubtitleCue[]): number {
|
||
return events.reduce((count, event) => count + sourceEventsOf(event).length, 0);
|
||
}
|
||
|
||
// One representative point per placement command: the `\pos` point or the `\move`
|
||
// midpoint. Comparing raw `\move` endpoints cross-wise misreads a travel distance that
|
||
// matches the glyph advance as a layer copy of a neighboring same-letter glyph.
|
||
const anchorPointCache = new WeakMap<AnnotatedSubtitleCue, AssFragmentPosition[]>();
|
||
|
||
function fragmentAnchorPoints(event: AnnotatedSubtitleCue): AssFragmentPosition[] {
|
||
let points = anchorPointCache.get(event);
|
||
if (points) {
|
||
return points;
|
||
}
|
||
points = [];
|
||
for (const command of event.overrides) {
|
||
const name = command.name.toLowerCase();
|
||
const args = command.args.split(',').map((value) => Number(value.trim()));
|
||
if (name === 'pos' && args.length >= 2 && args.slice(0, 2).every(Number.isFinite)) {
|
||
points.push({ x: args[0]!, y: args[1]! });
|
||
} else if (name === 'move' && args.length >= 4 && args.slice(0, 4).every(Number.isFinite)) {
|
||
points.push({ x: (args[0]! + args[2]!) / 2, y: (args[1]! + args[3]!) / 2 });
|
||
}
|
||
}
|
||
anchorPointCache.set(event, points);
|
||
return points;
|
||
}
|
||
|
||
function isRepeatedFragmentCopy(
|
||
previous: AnnotatedSubtitleCue,
|
||
current: AnnotatedSubtitleCue,
|
||
): boolean {
|
||
const previousAnchors = fragmentPlacementAnchors(previous);
|
||
if ([...fragmentPlacementAnchors(current)].some((anchor) => previousAnchors.has(anchor))) {
|
||
return true;
|
||
}
|
||
const previousPoints = fragmentAnchorPoints(previous);
|
||
const nearbyAnchor = fragmentAnchorPoints(current).some((point) =>
|
||
previousPoints.some(
|
||
(previousPoint) =>
|
||
Math.abs(point.x - previousPoint.x) <= LAYER_COPY_OFFSET_TOLERANCE_PX &&
|
||
Math.abs(point.y - previousPoint.y) <= LAYER_COPY_OFFSET_TOLERANCE_PX,
|
||
),
|
||
);
|
||
if (nearbyAnchor) {
|
||
return true;
|
||
}
|
||
return (
|
||
previous.startTime === current.startTime &&
|
||
previous.endTime === current.endTime &&
|
||
previous.overrideSignature === current.overrideSignature
|
||
);
|
||
}
|
||
|
||
// Coalescing keys on where a copy is anchored: the `\pos` point and both `\move`
|
||
// endpoints, since exit ghosts launch from the glyph anchor and entrance copies
|
||
// converge onto it.
|
||
function coalesceAnchorPoints(event: AnnotatedSubtitleCue): AssFragmentPosition[] {
|
||
const points: AssFragmentPosition[] = [];
|
||
for (const command of event.overrides) {
|
||
const name = command.name.toLowerCase();
|
||
const args = command.args.split(',').map((value) => Number(value.trim()));
|
||
if (name === 'pos' && args.length >= 2 && args.slice(0, 2).every(Number.isFinite)) {
|
||
points.push({ x: args[0]!, y: args[1]! });
|
||
} else if (name === 'move' && args.length >= 4 && args.slice(0, 4).every(Number.isFinite)) {
|
||
points.push({ x: args[0]!, y: args[1]! });
|
||
points.push({ x: args[2]!, y: args[3]! });
|
||
}
|
||
}
|
||
return points;
|
||
}
|
||
|
||
function shareCoalesceAnchor(
|
||
left: readonly AssFragmentPosition[],
|
||
right: readonly AssFragmentPosition[],
|
||
): boolean {
|
||
return right.some((point) =>
|
||
left.some(
|
||
(other) =>
|
||
Math.abs(point.x - other.x) <= LAYER_COPY_OFFSET_TOLERANCE_PX &&
|
||
Math.abs(point.y - other.y) <= LAYER_COPY_OFFSET_TOLERANCE_PX,
|
||
),
|
||
);
|
||
}
|
||
|
||
// The set of distinct command names, ignoring arguments and repetition. Two phases of
|
||
// one effect (pre-echo, highlight, hold) carry different command vocabularies; a re-run
|
||
// of the same visual (chant burst, countdown frame) repeats the same vocabulary with new
|
||
// argument values -- including a different number of animation keyframes, which is why
|
||
// repetition must not count.
|
||
const structuralSignatureCache = new WeakMap<AnnotatedSubtitleCue, string>();
|
||
|
||
function structuralOverrideSignature(cue: AnnotatedSubtitleCue): string {
|
||
let signature = structuralSignatureCache.get(cue);
|
||
if (signature === undefined) {
|
||
const names = new Set(
|
||
cue.overrides.map(
|
||
(command) => `${command.animated ? '~' : ''}${command.name.toLowerCase()}`,
|
||
),
|
||
);
|
||
signature = [...names].sort().join(',');
|
||
structuralSignatureCache.set(cue, signature);
|
||
}
|
||
return signature;
|
||
}
|
||
|
||
// A transparent lead-in ends exactly where its glyph's first sung copy begins, so the
|
||
// echo-to-visible handoff must chain across a small seam.
|
||
const COALESCE_SEAM_TOLERANCE_SECONDS = 0.05;
|
||
|
||
/**
|
||
* Two same-text copies chain when their windows genuinely overlap. Structurally
|
||
* identical events are layers of one visual exactly when their windows substantially
|
||
* coincide; a frame seam or few-millisecond overlap between identical structures is a
|
||
* re-run (the next chant burst, the next countdown frame). Structurally different
|
||
* events are phases of one effect and chain across any above-seam overlap. A bare seam
|
||
* only joins a transparent echo to its visible phase: the lead-in before a highlight
|
||
* belongs to its glyph, while two visible events that merely abut (the next burst of a
|
||
* chant, an exit flash after a hold) are separate showings.
|
||
*/
|
||
function areTimeConnectedCopies(a: AnnotatedSubtitleCue, b: AnnotatedSubtitleCue): boolean {
|
||
const overlap = Math.min(a.endTime, b.endTime) - Math.max(a.startTime, b.startTime);
|
||
if (structuralOverrideSignature(a) === structuralOverrideSignature(b)) {
|
||
const shorterDuration = Math.min(a.endTime - a.startTime, b.endTime - b.startTime);
|
||
return overlap >= Math.max(MIN_PHASE_OVERLAP_SECONDS, shorterDuration / 2);
|
||
}
|
||
if (overlap >= MIN_PHASE_OVERLAP_SECONDS) {
|
||
return true;
|
||
}
|
||
return (
|
||
overlap >= -COALESCE_SEAM_TOLERANCE_SECONDS &&
|
||
isTransparentFillEcho(a) !== isTransparentFillEcho(b)
|
||
);
|
||
}
|
||
|
||
function buildCoalescedCopy(members: readonly AnnotatedSubtitleCue[]): AnnotatedSubtitleCue {
|
||
const ordered = [...members].sort((left, right) => left.order - right.order);
|
||
const representative =
|
||
ordered.find((member) =>
|
||
member.overrides.some((command) => !command.animated && command.name.toLowerCase() === 'pos'),
|
||
) ?? ordered[0]!;
|
||
const synthetic: AnnotatedSubtitleCue = {
|
||
...representative,
|
||
startTime: earliestStartTime(ordered),
|
||
endTime: latestEndTime(ordered),
|
||
order: ordered[0]!.order,
|
||
};
|
||
coalescedSourceEvents.set(synthetic, ordered);
|
||
return synthetic;
|
||
}
|
||
|
||
/**
|
||
* Generated glyph effects render one authored glyph as a stack of copies anchored at the
|
||
* same point: a transparent pre-echo until the syllable is sung, a short highlight, an
|
||
* exit ghost launched from the anchor, and a steady hold to the end of the line. Their
|
||
* windows abut rather than coincide, so per-event timing says four unrelated fragments
|
||
* while the anchor says one glyph. Merging each stack into a single presence spanning
|
||
* the union window lets timing clusters see the authored line instead of its phases.
|
||
*/
|
||
function coalesceAssAnchorCopies(
|
||
events: readonly AnnotatedSubtitleCue[],
|
||
): AnnotatedSubtitleCue[] {
|
||
const buckets = new Map<string, number[]>();
|
||
const anchorPoints: (AssFragmentPosition[] | null)[] = events.map(() => null);
|
||
events.forEach((event, index) => {
|
||
const text = compactCueMatchText(event);
|
||
if (!text) {
|
||
return;
|
||
}
|
||
const points = coalesceAnchorPoints(event);
|
||
if (points.length === 0) {
|
||
return;
|
||
}
|
||
anchorPoints[index] = points;
|
||
const bucket = buckets.get(text);
|
||
if (bucket) {
|
||
bucket.push(index);
|
||
} else {
|
||
buckets.set(text, [index]);
|
||
}
|
||
});
|
||
for (const [text, bucket] of buckets) {
|
||
if (bucket.length > MAX_COALESCE_BUCKET_EVENTS) {
|
||
// Pathological same-text volume (a whole-episode particle field). Pairing would
|
||
// stall the main process; uncoalesced events fall back to the burst/grid paths.
|
||
buckets.delete(text);
|
||
} else {
|
||
bucket.sort((left, right) => events[left]!.startTime - events[right]!.startTime);
|
||
}
|
||
}
|
||
|
||
const parent = events.map((_, index) => index);
|
||
const findRoot = (index: number): number => {
|
||
let root = index;
|
||
while (parent[root] !== root) {
|
||
root = parent[root]!;
|
||
}
|
||
while (parent[index] !== root) {
|
||
const next = parent[index]!;
|
||
parent[index] = root;
|
||
index = next;
|
||
}
|
||
return root;
|
||
};
|
||
const union = (left: number, right: number): void => {
|
||
parent[findRoot(left)] = findRoot(right);
|
||
};
|
||
|
||
for (const bucket of buckets.values()) {
|
||
// Buckets are start-sorted; copies can only connect through time proximity, so the
|
||
// backward scan stops once no earlier copy's window can still reach this one.
|
||
const prefixMaxEnd: number[] = [];
|
||
let maxEnd = -Infinity;
|
||
for (const index of bucket) {
|
||
maxEnd = Math.max(maxEnd, events[index]!.endTime);
|
||
prefixMaxEnd.push(maxEnd);
|
||
}
|
||
for (let i = 1; i < bucket.length; i += 1) {
|
||
const right = bucket[i]!;
|
||
const reachableStart = events[right]!.startTime - COALESCE_SEAM_TOLERANCE_SECONDS;
|
||
for (let j = i - 1; j >= 0 && prefixMaxEnd[j]! >= reachableStart; j -= 1) {
|
||
const left = bucket[j]!;
|
||
if (
|
||
shareCoalesceAnchor(anchorPoints[left]!, anchorPoints[right]!) &&
|
||
areTimeConnectedCopies(events[left]!, events[right]!)
|
||
) {
|
||
union(left, right);
|
||
}
|
||
}
|
||
}
|
||
}
|
||
|
||
const componentsByRoot = new Map<number, AnnotatedSubtitleCue[]>();
|
||
events.forEach((event, index) => {
|
||
const root = findRoot(index);
|
||
const members = componentsByRoot.get(root);
|
||
if (members) {
|
||
members.push(event);
|
||
} else {
|
||
componentsByRoot.set(root, [event]);
|
||
}
|
||
});
|
||
if (componentsByRoot.size === events.length) {
|
||
return [...events];
|
||
}
|
||
return [...componentsByRoot.values()]
|
||
.map((members) => (members.length === 1 ? members[0]! : buildCoalescedCopy(members)))
|
||
.sort((left, right) => left.order - right.order);
|
||
}
|
||
|
||
function hasRelaxedAssFragmentEvidence(events: readonly AnnotatedSubtitleCue[]): boolean {
|
||
if (
|
||
sourceEventCount(events) < 2 ||
|
||
!events.every((event) => fragmentPlacementAnchors(event).size > 0)
|
||
) {
|
||
return false;
|
||
}
|
||
|
||
const latestStart = events.reduce(
|
||
(latest, event) => Math.max(latest, event.startTime),
|
||
-Infinity,
|
||
);
|
||
const earliestEnd = events.reduce(
|
||
(earliest, event) => Math.min(earliest, event.endTime),
|
||
Infinity,
|
||
);
|
||
if (latestStart >= earliestEnd) {
|
||
return false;
|
||
}
|
||
|
||
const first = events[0]!;
|
||
const hasChangingOverrides = events.some(
|
||
(event) => event.overrideSignature !== first.overrideSignature,
|
||
);
|
||
const hasPositionedLayerCopy =
|
||
events.some((event) => sourceEventsOf(event).length > 1) ||
|
||
events.some((event, index) =>
|
||
events
|
||
.slice(0, index)
|
||
.some(
|
||
(previous) =>
|
||
compactCueMatchText(previous) === compactCueMatchText(event) &&
|
||
isRepeatedFragmentCopy(previous, event),
|
||
),
|
||
);
|
||
return hasChangingOverrides || hasPositionedLayerCopy;
|
||
}
|
||
|
||
interface AssFragmentPart {
|
||
cue: AnnotatedSubtitleCue;
|
||
text: string;
|
||
}
|
||
|
||
interface AssFragmentPosition {
|
||
x: number;
|
||
y: number;
|
||
}
|
||
|
||
const MIN_LATIN_POSITION_GAP_SAMPLES = 4;
|
||
const LATIN_FRAGMENT_WORD_GAP_RATIO = 1.16;
|
||
const LATIN_GLYPH_WORD_GAP_RATIO = 1.4;
|
||
// Word-space advance beyond the width-predicted glyph advance, as a fraction of the
|
||
// line's common unit. Measured corpus extremes: widest within-word excess 0.32 (`pp`
|
||
// with tracking), narrowest word gap 0.40 (`s w` across a wide glyph). That margin only
|
||
// holds when the common unit is estimated from enough glyph pairs; a short single-word
|
||
// line (`Swelling`) skews the unit low and its ordinary advances read as word gaps.
|
||
const LATIN_GLYPH_WORD_EXCESS_RATIO = 0.36;
|
||
const MIN_LATIN_GLYPH_EXCESS_GAP_SAMPLES = 10;
|
||
// Multi-character syllable chunks average out proportional-font variation, so their
|
||
// advances track the width model far more closely than single glyphs do. Measured on a
|
||
// chunked lyric line, within-word excess stayed under 0.07 of the common unit while every
|
||
// word gap cleared 0.31, so a tighter margin separates them without splitting words.
|
||
const LATIN_CHUNK_WORD_EXCESS_RATIO = 0.2;
|
||
const MIN_LATIN_CHUNK_EXCESS_GAP_SAMPLES = 6;
|
||
const LATIN_TWO_GLYPH_WORD_NEXT_GAP_RATIO = 1.2;
|
||
|
||
function fragmentPosition(cue: AnnotatedSubtitleCue): AssFragmentPosition | null {
|
||
for (const command of cue.overrides) {
|
||
if (command.animated) continue;
|
||
const name = command.name.toLowerCase();
|
||
const args = command.args.split(',').map((value) => Number(value.trim()));
|
||
if (
|
||
name === 'pos' &&
|
||
args.length >= 2 &&
|
||
Number.isFinite(args[0]) &&
|
||
Number.isFinite(args[1])
|
||
) {
|
||
return { x: args[0]!, y: args[1]! };
|
||
}
|
||
if (
|
||
name === 'move' &&
|
||
args.length >= 4 &&
|
||
args.slice(0, 4).every((value) => Number.isFinite(value))
|
||
) {
|
||
return { x: (args[0]! + args[2]!) / 2, y: (args[1]! + args[3]!) / 2 };
|
||
}
|
||
}
|
||
return null;
|
||
}
|
||
|
||
function latinGlyphWidthWeight(glyph: string): number {
|
||
if (/[ilIj]/u.test(glyph)) return 0.6;
|
||
if (/[tfr]/u.test(glyph)) return 0.8;
|
||
if (/[mwMW]/u.test(glyph)) return 1.4;
|
||
if (/[A-Z]/u.test(glyph)) return 1.1;
|
||
return 1;
|
||
}
|
||
|
||
function latinFragmentWidthWeight(text: string): number | null {
|
||
if (!/^[A-Za-z0-9'’.,!?;:-]+$/u.test(text)) return null;
|
||
const punctuationWeight = /^[A-Za-z0-9]['’.,!?;:-]$/u.test(text) ? 0.5 : 0.25;
|
||
return [...text].reduce(
|
||
(width, glyph) =>
|
||
width + (/['’.,!?;:-]/u.test(glyph) ? punctuationWeight : latinGlyphWidthWeight(glyph)),
|
||
0,
|
||
);
|
||
}
|
||
|
||
function isSingleLatinGlyphFragment(text: string): boolean {
|
||
return [...text].filter((glyph) => /[A-Za-z0-9]/u.test(glyph)).length <= 1;
|
||
}
|
||
|
||
interface LatinFragmentGapMeasure {
|
||
distance: number;
|
||
meanWeight: number;
|
||
}
|
||
|
||
function latinFragmentGapMeasure(
|
||
previous: AssFragmentPart,
|
||
current: AssFragmentPart,
|
||
): LatinFragmentGapMeasure | null {
|
||
const previousWeight = latinFragmentWidthWeight(previous.text);
|
||
const currentWeight = latinFragmentWidthWeight(current.text);
|
||
const previousPosition = fragmentPosition(previous.cue);
|
||
const currentPosition = fragmentPosition(current.cue);
|
||
if (previousWeight === null || currentWeight === null || !previousPosition || !currentPosition) {
|
||
return null;
|
||
}
|
||
const xDistance = currentPosition.x - previousPosition.x;
|
||
const yDistance = Math.abs(currentPosition.y - previousPosition.y);
|
||
if (yDistance <= 2 && xDistance <= 0) return null;
|
||
|
||
// A wrapped authored line can return to the left on its next visual row. Preserve
|
||
// that measured row transition as a separator without treating backwards movement
|
||
// on the same row as a word gap.
|
||
const distance = yDistance <= 2 ? xDistance : Math.abs(xDistance) + yDistance;
|
||
return { distance, meanWeight: (previousWeight + currentWeight) / 2 };
|
||
}
|
||
|
||
function normalizedLatinFragmentGap(
|
||
previous: AssFragmentPart,
|
||
current: AssFragmentPart,
|
||
): number | null {
|
||
const measure = latinFragmentGapMeasure(previous, current);
|
||
return measure === null ? null : measure.distance / measure.meanWeight;
|
||
}
|
||
|
||
function startsNewPositionedFragmentSequence(
|
||
previous: AssFragmentPart,
|
||
current: AssFragmentPart,
|
||
): boolean {
|
||
const previousPosition = fragmentPosition(previous.cue);
|
||
const currentPosition = fragmentPosition(current.cue);
|
||
return Boolean(
|
||
previousPosition &&
|
||
currentPosition &&
|
||
Math.abs(currentPosition.y - previousPosition.y) <= 2 &&
|
||
currentPosition.x <= previousPosition.x &&
|
||
current.cue.startTime > previous.cue.startTime,
|
||
);
|
||
}
|
||
|
||
function commonLatinFragmentGap(values: readonly number[]): number {
|
||
const sorted = [...values].sort((left, right) => left - right);
|
||
// Romaji lines contain many short particles, so real word gaps can outnumber
|
||
// within-word transitions. A lower quantile still represents ordinary glyph advance
|
||
// while ignoring the narrowest character pair as an outlier.
|
||
return sorted[Math.floor((sorted.length - 1) * 0.35)]!;
|
||
}
|
||
|
||
function isLikelyTwoGlyphCapitalizedWord(options: {
|
||
parts: readonly AssFragmentPart[];
|
||
index: number;
|
||
gap: number;
|
||
wordGapThreshold: number;
|
||
}): boolean {
|
||
const first = options.parts[options.index - 1]!;
|
||
const second = options.parts[options.index]!;
|
||
if (!/^[A-Z]$/u.test(first.text) || !/^[a-z]$/u.test(second.text)) {
|
||
return false;
|
||
}
|
||
|
||
const precedingGap =
|
||
options.index > 1 ? normalizedLatinFragmentGap(options.parts[options.index - 2]!, first) : null;
|
||
const following = options.parts[options.index + 1];
|
||
const followingGap = following ? normalizedLatinFragmentGap(second, following) : null;
|
||
const startsAtWordBoundary =
|
||
options.index === 1 || (precedingGap !== null && precedingGap > options.wordGapThreshold);
|
||
|
||
return (
|
||
startsAtWordBoundary &&
|
||
followingGap !== null &&
|
||
followingGap > options.wordGapThreshold &&
|
||
followingGap > options.gap * LATIN_TWO_GLYPH_WORD_NEXT_GAP_RATIO
|
||
);
|
||
}
|
||
|
||
/**
|
||
* Character-by-character typesetting often omits literal spaces because the authored
|
||
* word gap exists only in each glyph's `\pos`. Estimate the normal adjacent-glyph
|
||
* advance within that one line, then preserve only materially larger horizontal gaps.
|
||
* Normalizing each gap by the neighboring fragment widths supports both single glyphs
|
||
* and multi-character karaoke syllables without guessing from the text itself. Per-glyph
|
||
* runs use a wider safety margin because proportional fonts vary more than syllable chunks.
|
||
*
|
||
* The ratio test alone under-detects a word gap next to a wide fragment (`waves within`
|
||
* measured across `s`/`w`, or `Choices presumably` across two three-letter chunks, both
|
||
* normalize to nearly a common advance), so a gap also counts as a word boundary when its
|
||
* advance exceeds the width-predicted advance by a material fraction of the line's common
|
||
* unit -- a word space adds a roughly constant extra distance no matter how wide its
|
||
* neighbors are. Chunk runs use a tighter margin than per-glyph runs because their
|
||
* advances deviate less from the width model.
|
||
*
|
||
* A line may mix both conventions: one fragment carrying a literal space while its
|
||
* neighbors rely on position alone. Whitespace-bearing fragments have no width weight, so
|
||
* they drop out of the estimate and their own boundary comes from the authored space,
|
||
* leaving the surrounding positional gaps to be recovered normally.
|
||
*/
|
||
function joinAssFragmentParts(parts: readonly AssFragmentPart[]): string {
|
||
const normalizedGaps: number[] = [];
|
||
for (let index = 1; index < parts.length; index += 1) {
|
||
const gap = normalizedLatinFragmentGap(parts[index - 1]!, parts[index]!);
|
||
if (gap !== null) normalizedGaps.push(gap);
|
||
}
|
||
const isGlyphRun = parts.every((part) => isSingleLatinGlyphFragment(part.text));
|
||
const commonGap =
|
||
normalizedGaps.length >= MIN_LATIN_POSITION_GAP_SAMPLES
|
||
? commonLatinFragmentGap(normalizedGaps)
|
||
: null;
|
||
const wordGapThreshold =
|
||
commonGap === null
|
||
? Infinity
|
||
: commonGap * (isGlyphRun ? LATIN_GLYPH_WORD_GAP_RATIO : LATIN_FRAGMENT_WORD_GAP_RATIO);
|
||
|
||
let text = parts[0]?.text ?? '';
|
||
for (let index = 1; index < parts.length; index += 1) {
|
||
const previous = parts[index - 1]!;
|
||
const current = parts[index]!;
|
||
const hasAuthoredSpace = /\s$/u.test(previous.text) || /^\s/u.test(current.text);
|
||
const measure = latinFragmentGapMeasure(previous, current);
|
||
const normalizedGap = measure === null ? null : measure.distance / measure.meanWeight;
|
||
// A capital into lowercase is almost always a capitalized word's own first letters
|
||
// (`S|miles`), and capitals overrun the width table too easily, so the excess rule
|
||
// never fires there. A lone capital word like `I` is narrow enough for the ratio
|
||
// test to catch its word gap on its own.
|
||
const excessRatio = isGlyphRun ? LATIN_GLYPH_WORD_EXCESS_RATIO : LATIN_CHUNK_WORD_EXCESS_RATIO;
|
||
const minimumExcessSamples = isGlyphRun
|
||
? MIN_LATIN_GLYPH_EXCESS_GAP_SAMPLES
|
||
: MIN_LATIN_CHUNK_EXCESS_GAP_SAMPLES;
|
||
const hasAdvanceExcess =
|
||
commonGap !== null &&
|
||
normalizedGaps.length >= minimumExcessSamples &&
|
||
measure !== null &&
|
||
!(/^[A-Z]$/u.test(previous.text) && /^[a-z]$/u.test(current.text)) &&
|
||
measure.distance - measure.meanWeight * commonGap > excessRatio * commonGap;
|
||
const hasPositionedWordGap =
|
||
startsNewPositionedFragmentSequence(previous, current) ||
|
||
(normalizedGap !== null &&
|
||
(normalizedGap > wordGapThreshold || hasAdvanceExcess) &&
|
||
!isLikelyTwoGlyphCapitalizedWord({
|
||
parts,
|
||
index,
|
||
gap: normalizedGap,
|
||
wordGapThreshold,
|
||
}));
|
||
if (!hasAuthoredSpace && hasPositionedWordGap) {
|
||
text += ' ';
|
||
}
|
||
text += current.text;
|
||
}
|
||
return text.trim();
|
||
}
|
||
|
||
// A tall multi-part layout is only a visual grid when its parts read like tiling
|
||
// rather than prose: a couple of texts repeated across many fragments (sign walls),
|
||
// the same text re-shown at the same spot over time (countdown/animation frames),
|
||
// nothing but scattered single glyphs, or cells aligned into table columns. Wrapped
|
||
// lyric rows with repeated karaoke syllables and CC-style dialogue blocks (speaker
|
||
// labels plus a sentence) share the same tall geometry but stay publishable.
|
||
function looksLikeFragmentGridParts(parts: readonly AssFragmentPart[]): boolean {
|
||
const positioned = parts
|
||
.map((part) => ({
|
||
text: part.text.trim(),
|
||
layout: part.cue.assLayout,
|
||
position: fragmentPosition(part.cue),
|
||
startTime: part.cue.startTime,
|
||
}))
|
||
.filter((part) => part.text && part.layout?.kind === 'positioned');
|
||
if (positioned.length === 0) return true;
|
||
|
||
const uniqueTexts = new Set(positioned.map((part) => part.text));
|
||
if (uniqueTexts.size * 3 <= positioned.length) return true;
|
||
|
||
if (positioned.every((part) => [...part.text].length <= 1)) return true;
|
||
|
||
const seenPlacements = new Map<string, number>();
|
||
for (const part of positioned) {
|
||
if (part.layout?.kind !== 'positioned' || !part.position) continue;
|
||
const placement = `${part.text}@${Math.round(part.position.x)},${Math.round(part.position.y)}`;
|
||
const earlierStart = seenPlacements.get(placement);
|
||
if (earlierStart !== undefined && Math.abs(part.startTime - earlierStart) > 0.01) {
|
||
return true;
|
||
}
|
||
seenPlacements.set(placement, part.startTime);
|
||
}
|
||
|
||
// Table cells align into columns: several x values each reused on multiple rows.
|
||
// Requiring two such columns holding at least half the parts keeps a wrapped lyric
|
||
// whose rows accidentally share one x coordinate out of the grid bucket.
|
||
const columnRows = new Map<number, Set<number>>();
|
||
for (const part of positioned) {
|
||
if (!part.position) continue;
|
||
const x = Math.round(part.position.x);
|
||
const rows = columnRows.get(x) ?? new Set<number>();
|
||
rows.add(Math.round(part.position.y));
|
||
columnRows.set(x, rows);
|
||
}
|
||
let alignedColumns = 0;
|
||
let alignedParts = 0;
|
||
for (const part of positioned) {
|
||
if (!part.position) continue;
|
||
if ((columnRows.get(Math.round(part.position.x))?.size ?? 0) >= 2) alignedParts += 1;
|
||
}
|
||
for (const rows of columnRows.values()) {
|
||
if (rows.size >= 2) alignedColumns += 1;
|
||
}
|
||
return alignedColumns >= 2 && alignedParts * 2 >= positioned.length;
|
||
}
|
||
|
||
function reconstructedAssFragmentLayout(
|
||
parts: readonly AssFragmentPart[],
|
||
owner: AnnotatedSubtitleCue,
|
||
): AssCueLayout | undefined {
|
||
let positionedPartCount = 0;
|
||
let minimumY = Infinity;
|
||
let maximumY = -Infinity;
|
||
for (const part of parts) {
|
||
const layout = part.cue.assLayout;
|
||
if (layout?.kind !== 'positioned') continue;
|
||
positionedPartCount += 1;
|
||
minimumY = Math.min(minimumY, layout.y);
|
||
maximumY = Math.max(maximumY, layout.y);
|
||
}
|
||
|
||
if (
|
||
positionedPartCount >= MIN_FRAGMENT_LINE_PARTS &&
|
||
maximumY - minimumY > MAX_FRAGMENT_LINE_VERTICAL_SPAN &&
|
||
looksLikeFragmentGridParts(parts)
|
||
) {
|
||
return { kind: 'fragment-grid', sourceOrder: owner.order };
|
||
}
|
||
return owner.assLayout;
|
||
}
|
||
|
||
interface AssFragmentTimingCluster {
|
||
events: AnnotatedSubtitleCue[];
|
||
minStartTime: number;
|
||
maxStartTime: number;
|
||
minEndTime: number;
|
||
maxEndTime: number;
|
||
}
|
||
|
||
function addToFragmentTimingCluster(
|
||
cluster: AssFragmentTimingCluster,
|
||
cue: AnnotatedSubtitleCue,
|
||
): void {
|
||
cluster.events.push(cue);
|
||
cluster.minStartTime = Math.min(cluster.minStartTime, cue.startTime);
|
||
cluster.maxStartTime = Math.max(cluster.maxStartTime, cue.startTime);
|
||
cluster.minEndTime = Math.min(cluster.minEndTime, cue.endTime);
|
||
cluster.maxEndTime = Math.max(cluster.maxEndTime, cue.endTime);
|
||
}
|
||
|
||
function fragmentTimingDistance(
|
||
cluster: AssFragmentTimingCluster,
|
||
cue: AnnotatedSubtitleCue,
|
||
): number {
|
||
const nextMinStart = Math.min(cluster.minStartTime, cue.startTime);
|
||
const nextMaxStart = Math.max(cluster.maxStartTime, cue.startTime);
|
||
const nextMinEnd = Math.min(cluster.minEndTime, cue.endTime);
|
||
const nextMaxEnd = Math.max(cluster.maxEndTime, cue.endTime);
|
||
if (
|
||
nextMaxStart - nextMinStart > MAX_FRAGMENT_LINE_TIMING_VARIANCE_SECONDS ||
|
||
nextMaxEnd - nextMinEnd > MAX_FRAGMENT_LINE_TIMING_VARIANCE_SECONDS
|
||
) {
|
||
return Infinity;
|
||
}
|
||
return (
|
||
Math.abs(cue.startTime - (cluster.minStartTime + cluster.maxStartTime) / 2) +
|
||
Math.abs(cue.endTime - (cluster.minEndTime + cluster.maxEndTime) / 2)
|
||
);
|
||
}
|
||
|
||
function clusterAssFragmentEvents(
|
||
events: readonly AnnotatedSubtitleCue[],
|
||
): AssFragmentTimingCluster[] {
|
||
const clusters: AssFragmentTimingCluster[] = [];
|
||
for (const cue of events) {
|
||
let nearest: AssFragmentTimingCluster | null = null;
|
||
let nearestDistance = Infinity;
|
||
for (const cluster of clusters) {
|
||
const distance = fragmentTimingDistance(cluster, cue);
|
||
if (distance < nearestDistance) {
|
||
nearest = cluster;
|
||
nearestDistance = distance;
|
||
}
|
||
}
|
||
if (nearest) {
|
||
addToFragmentTimingCluster(nearest, cue);
|
||
} else {
|
||
clusters.push({
|
||
events: [cue],
|
||
minStartTime: cue.startTime,
|
||
maxStartTime: cue.startTime,
|
||
minEndTime: cue.endTime,
|
||
maxEndTime: cue.endTime,
|
||
});
|
||
}
|
||
}
|
||
return clusters;
|
||
}
|
||
|
||
interface FragmentInterval {
|
||
startTime: number;
|
||
endTime: number;
|
||
}
|
||
|
||
/** Event time ranges with repeated same-text, same-time layer copies collapsed. */
|
||
function distinctFragmentIntervals(events: readonly AnnotatedSubtitleCue[]): FragmentInterval[] {
|
||
const intervals: FragmentInterval[] = [];
|
||
const previousEvents: AnnotatedSubtitleCue[] = [];
|
||
for (const event of events) {
|
||
const compactText = compactCueMatchText(event);
|
||
const isLayerCopy = previousEvents.some(
|
||
(previous) =>
|
||
compactCueMatchText(previous) === compactText &&
|
||
previous.startTime === event.startTime &&
|
||
previous.endTime === event.endTime &&
|
||
isRepeatedFragmentCopy(previous, event),
|
||
);
|
||
previousEvents.push(event);
|
||
if (isLayerCopy) continue;
|
||
intervals.push({ startTime: event.startTime, endTime: event.endTime });
|
||
}
|
||
return intervals.sort((a, b) => a.startTime - b.startTime || a.endTime - b.endTime);
|
||
}
|
||
|
||
function intervalsNeverCoexist(intervals: readonly FragmentInterval[]): boolean {
|
||
let latestEnd = -Infinity;
|
||
for (const interval of intervals) {
|
||
if (interval.startTime < latestEnd - 0.001) {
|
||
return false;
|
||
}
|
||
latestEnd = Math.max(latestEnd, interval.endTime);
|
||
}
|
||
return true;
|
||
}
|
||
|
||
// A sweep progresses through the syllables of a lyric, so its events carry different
|
||
// texts. Sequential same-text repaints are one shaking line being redrawn, and its text
|
||
// must survive to the raw/burst path rather than be suppressed as decoration.
|
||
function hasMultipleFragmentTexts(events: readonly AnnotatedSubtitleCue[]): boolean {
|
||
const first = events[0] ? compactCueMatchText(events[0]) : '';
|
||
return events.some((event) => compactCueMatchText(event) !== first);
|
||
}
|
||
|
||
/**
|
||
* A karaoke highlight sweep repaints one syllable at a time over an already-visible
|
||
* lyric line: each event ends as the next begins, so the cluster's concatenated text is
|
||
* never on screen as a whole. Publishing it would emit rolling partial copies of the
|
||
* lyric ("to sou omo" beside "akenakute ii to sou omotteta"). Layer copies share one
|
||
* placement and timing, so the test is whether any two distinct placements coexist.
|
||
*/
|
||
function isProgressiveHighlightSweep(events: readonly AnnotatedSubtitleCue[]): boolean {
|
||
if (!hasMultipleFragmentTexts(events)) {
|
||
return false;
|
||
}
|
||
const intervals = distinctFragmentIntervals(events);
|
||
return intervals.length >= 2 && intervalsNeverCoexist(intervals);
|
||
}
|
||
|
||
/**
|
||
* Timing clusters split a long sweep unevenly, leaving stragglers the per-cluster check
|
||
* cannot judge: a two-event tail reconstructs on relaxed evidence, and a lone held
|
||
* syllable stays raw and publishes as its own flickering cue. When an entire style group
|
||
* reads as one chained repaint -- many short positioned animated fragments, no two ever
|
||
* on screen together, transitions mostly back-to-back -- the whole group is highlight
|
||
* decoration and none of it is publishable text. Independent one-off signs sharing a
|
||
* style stay published: they are few, longer, or separated by real gaps.
|
||
*/
|
||
function isProgressiveHighlightSweepGroup(events: readonly AnnotatedSubtitleCue[]): boolean {
|
||
if (
|
||
events.length > MAX_SWEEP_GROUP_EVENTS ||
|
||
sourceEventCount(events) < MIN_FRAGMENT_LINE_EVENTS ||
|
||
!hasMultipleFragmentTexts(events) ||
|
||
!events.every((event) => fragmentPlacementAnchors(event).size > 0) ||
|
||
!hasAssAnimationEvidence(events)
|
||
) {
|
||
return false;
|
||
}
|
||
const lengths = events
|
||
.map((event) => compactCueMatchText(event).length)
|
||
.sort((left, right) => left - right);
|
||
if ((lengths[Math.floor(lengths.length / 2)] ?? Infinity) > MAX_FRAGMENT_MEDIAN_LENGTH) {
|
||
return false;
|
||
}
|
||
const intervals = distinctFragmentIntervals(events);
|
||
if (intervals.length < 2 || !intervalsNeverCoexist(intervals)) {
|
||
return false;
|
||
}
|
||
let abutting = 0;
|
||
for (let index = 1; index < intervals.length; index += 1) {
|
||
if (Math.abs(intervals[index]!.startTime - intervals[index - 1]!.endTime) <= 0.1) {
|
||
abutting += 1;
|
||
}
|
||
}
|
||
return abutting * 2 >= intervals.length - 1;
|
||
}
|
||
|
||
function decodeSingleAssFragment(cue: AnnotatedSubtitleCue): string | null {
|
||
const visibleLines = decodeSubtitleCueText(cue.rawText)
|
||
.split('\n')
|
||
.filter((line) => line.trim().length > 0);
|
||
return visibleLines.length === 1 ? visibleLines[0]! : null;
|
||
}
|
||
|
||
/**
|
||
* One cluster can hold the same authored text at two granularities: a whole-line event
|
||
* and the per-glyph events that spell it (an assembly effect renders the line while its
|
||
* glyph particles converge). Joining both doubles the line. A consecutive run of two or
|
||
* more parts that concatenates to exactly another part's text is that part's fragment
|
||
* layer; the whole part keeps the authored spacing, so the run is dropped. Single equal
|
||
* parts are never dropped -- a repeated word in a lyric is real text, not a layer.
|
||
*
|
||
* Spelling alone is not proof: repeated digits after a thousands group also concatenate
|
||
* to the earlier fragment's text while being real continuation. A duplicate layer sits
|
||
* on top of its fragments, so the whole part's anchor must fall inside the run's
|
||
* positional span; text that merely continues the line sits beyond it.
|
||
*/
|
||
function isWholePartOverItsRun(whole: AssFragmentPart, run: readonly AssFragmentPart[]): boolean {
|
||
const wholePosition = fragmentPosition(whole.cue);
|
||
if (!wholePosition) {
|
||
return true;
|
||
}
|
||
const xs = run
|
||
.map((part) => fragmentPosition(part.cue)?.x)
|
||
.filter((x): x is number => Number.isFinite(x));
|
||
if (xs.length === 0) {
|
||
return true;
|
||
}
|
||
return wholePosition.x >= Math.min(...xs) && wholePosition.x <= Math.max(...xs);
|
||
}
|
||
|
||
function dropFragmentRunsCoveredByWholeParts(parts: AssFragmentPart[]): AssFragmentPart[] {
|
||
if (parts.length > MAX_FRAGMENT_COLLAPSE_PARTS) {
|
||
return parts;
|
||
}
|
||
const kept = [...parts];
|
||
let changed = true;
|
||
while (changed) {
|
||
changed = false;
|
||
const wholes = [...kept].sort(
|
||
(left, right) =>
|
||
compactAssMatchText(right.text).length - compactAssMatchText(left.text).length,
|
||
);
|
||
for (const whole of wholes) {
|
||
const wholeText = compactAssMatchText(whole.text);
|
||
if ([...wholeText].length < 2) {
|
||
break;
|
||
}
|
||
for (let start = 0; start < kept.length && !changed; start += 1) {
|
||
if (kept[start] === whole) {
|
||
continue;
|
||
}
|
||
let combined = '';
|
||
for (let end = start; end < kept.length; end += 1) {
|
||
if (kept[end] === whole) {
|
||
break;
|
||
}
|
||
combined += compactAssMatchText(kept[end]!.text);
|
||
if (!wholeText.startsWith(combined)) {
|
||
break;
|
||
}
|
||
if (combined === wholeText) {
|
||
const run = kept.slice(start, end + 1);
|
||
if (end > start && isWholePartOverItsRun(whole, run)) {
|
||
kept.splice(start, end - start + 1);
|
||
changed = true;
|
||
}
|
||
break;
|
||
}
|
||
}
|
||
}
|
||
if (changed) {
|
||
break;
|
||
}
|
||
}
|
||
}
|
||
return kept;
|
||
}
|
||
|
||
function reconstructAssFragmentLine(
|
||
events: readonly AnnotatedSubtitleCue[],
|
||
): AnnotatedSubtitleCue | null {
|
||
const hasRelaxedEvidence = hasRelaxedAssFragmentEvidence(events);
|
||
const minimumEvents = hasRelaxedEvidence ? 2 : MIN_FRAGMENT_LINE_EVENTS;
|
||
// Coalesced copies stand in for their source events, so animation-volume thresholds
|
||
// count sources: a merged four-phase glyph is still four generated events of evidence.
|
||
if (sourceEventCount(events) < minimumEvents || !hasAssAnimationEvidence(events)) {
|
||
return null;
|
||
}
|
||
|
||
const collected: AssFragmentPart[] = [];
|
||
for (const cue of events) {
|
||
const text = decodeSingleAssFragment(cue);
|
||
if (text === null) {
|
||
return null;
|
||
}
|
||
const compactText = compactAssMatchText(text);
|
||
const isLayerCopy = collected.some(
|
||
(part) =>
|
||
compactAssMatchText(part.text) === compactText && isRepeatedFragmentCopy(part.cue, cue),
|
||
);
|
||
if (!isLayerCopy) {
|
||
collected.push({ cue, text });
|
||
}
|
||
}
|
||
const minimumParts = hasRelaxedEvidence ? 1 : MIN_FRAGMENT_LINE_PARTS;
|
||
if (
|
||
collected.length < minimumParts ||
|
||
(!hasRelaxedEvidence && collected.length === sourceEventCount(events))
|
||
) {
|
||
return null;
|
||
}
|
||
// A cluster whose every part is one identical glyph is a particle field, not a line.
|
||
// Reconstructing it would also break the surviving particles' same-text chain that
|
||
// burst deduplication collapses downstream.
|
||
if (collected.length >= 2) {
|
||
const firstText = compactAssMatchText(collected[0]!.text);
|
||
if (
|
||
[...firstText].length === 1 &&
|
||
collected.every((part) => compactAssMatchText(part.text) === firstText)
|
||
) {
|
||
return null;
|
||
}
|
||
}
|
||
// The granularity gate judges the cluster as authored, so it runs before redundant
|
||
// whole-vs-fragments layers collapse: a line plus its glyph swarm is fragment-sized
|
||
// work even though only the whole-line part survives into the join.
|
||
const lengths = collected
|
||
.map((part) => compactAssMatchText(part.text).length)
|
||
.sort((left, right) => left - right);
|
||
if ((lengths[Math.floor(lengths.length / 2)] ?? Infinity) > MAX_FRAGMENT_MEDIAN_LENGTH) {
|
||
return null;
|
||
}
|
||
const parts = dropFragmentRunsCoveredByWholeParts(collected);
|
||
|
||
const text = joinAssFragmentParts(parts);
|
||
if (!text) {
|
||
return null;
|
||
}
|
||
const owner = parts[0]!.cue;
|
||
const animationStartTime = earliestStartTime(events);
|
||
const animationEndTime = latestEndTime(events);
|
||
// Publish the window where the line reads as sung text. Transparent pre-echoes render
|
||
// the upcoming line before its first syllable, and exit ghosts fade past the hold, so
|
||
// the raw span makes consecutive lyrics overlap on screen. The line starts with its
|
||
// first opaque copy and holds until the last statically anchored one ends; a line
|
||
// placed entirely by `\move` has no hold phase and keeps its full visible span.
|
||
const sources = events.flatMap((event) => [...sourceEventsOf(event)]);
|
||
const visibleSources = sources.filter((source) => !isTransparentFillEcho(source));
|
||
const displaySources = visibleSources.length > 0 ? visibleSources : sources;
|
||
const heldSources = displaySources.filter((source) =>
|
||
source.overrides.some((command) => !command.animated && command.name.toLowerCase() === 'pos'),
|
||
);
|
||
return {
|
||
...owner,
|
||
startTime: earliestStartTime(displaySources),
|
||
endTime: latestEndTime(heldSources.length > 0 ? heldSources : displaySources),
|
||
text,
|
||
rawText: text,
|
||
source: 'reconstructed-ass',
|
||
animationStartTime,
|
||
animationEndTime,
|
||
assLayout: reconstructedAssFragmentLayout(parts, owner),
|
||
overrides: [],
|
||
overrideSignature: '',
|
||
};
|
||
}
|
||
|
||
// `\fnSplit splat splodge` tokenizes as name `fnSplit` + args `splat splodge`, while
|
||
// `\fnArial` is all name and `\fn04b` is all args, so the font is both pieces rejoined.
|
||
function staticFontOverride(cue: AnnotatedSubtitleCue): string | null {
|
||
let font: string | null = null;
|
||
for (const command of cue.overrides) {
|
||
if (command.animated || !command.name.toLowerCase().startsWith('fn')) continue;
|
||
font = [command.name.slice(2), command.args].filter(Boolean).join(' ').trim().toLowerCase();
|
||
}
|
||
return font;
|
||
}
|
||
|
||
const MIN_TEXTURE_GLYPH_RUN = 8;
|
||
const MIN_TEXTURE_ALPHA_OVERRIDES = 6;
|
||
// Texture payloads switch secondary alpha at nearly every glyph. Authored text that a
|
||
// typesetter styles in syllable or word chunks measures two or more glyphs per
|
||
// override, so the seed test demands per-glyph density.
|
||
const MAX_TEXTURE_GLYPHS_PER_ALPHA_OVERRIDE = 1.5;
|
||
const MIN_TEXTURE_LAYER_ALPHA = 0xe0;
|
||
const MAX_TEXTURE_PAYLOAD_FONT_SIZE = 12;
|
||
const MIN_TEXTURE_PAYLOAD_LINES = 3;
|
||
const ASS_ALPHA_VALUE_PATTERN = /^&?H([0-9a-f]{1,2})&?$/iu;
|
||
const ASS_FONT_WEIGHT_SUFFIX_PATTERN =
|
||
/\s+(?:black|bold|heavy|light|medium|regular|semibold|thin)$/u;
|
||
|
||
function hasStaticOverride(cue: AnnotatedSubtitleCue, expectedName: string): boolean {
|
||
return cue.overrides.some(
|
||
(command) => !command.animated && command.name.toLowerCase() === expectedName,
|
||
);
|
||
}
|
||
|
||
function isRepeatedGlyphText(cue: AnnotatedSubtitleCue): boolean {
|
||
const glyphs = [...compactCueMatchText(cue)];
|
||
return glyphs.length > 0 && glyphs.every((glyph) => glyph === glyphs[0]);
|
||
}
|
||
|
||
function isClippedRepeatedGlyphFragment(cue: AnnotatedSubtitleCue): boolean {
|
||
return (
|
||
isRepeatedGlyphText(cue) &&
|
||
(hasStaticOverride(cue, 'clip') || hasStaticOverride(cue, 'iclip'))
|
||
);
|
||
}
|
||
|
||
/**
|
||
* Some ASS signs build image textures from clipped placeholder glyphs, optionally through
|
||
* a texture font. A long clipped single-glyph run or frequent changing secondary alpha
|
||
* tags identifies the effect without guessing from its visible text or font name.
|
||
*/
|
||
function isAssTextureSeed(cue: AnnotatedSubtitleCue): boolean {
|
||
if (fragmentPosition(cue) === null) {
|
||
return false;
|
||
}
|
||
|
||
// A truncated capture can leave tag debris after the placeholder run, so the seed
|
||
// test looks for a long same-glyph run inside the clipped text rather than requiring
|
||
// the whole event to be uniform.
|
||
const glyphs = [...compactCueMatchText(cue)];
|
||
if (hasStaticOverride(cue, 'clip') || hasStaticOverride(cue, 'iclip')) {
|
||
let longestRun = 0;
|
||
let run = 0;
|
||
let previous = '';
|
||
for (const glyph of glyphs) {
|
||
run = glyph === previous ? run + 1 : 1;
|
||
previous = glyph;
|
||
longestRun = Math.max(longestRun, run);
|
||
}
|
||
if (longestRun >= MIN_TEXTURE_GLYPH_RUN) {
|
||
return true;
|
||
}
|
||
}
|
||
|
||
if (staticFontOverride(cue) === null) {
|
||
return false;
|
||
}
|
||
|
||
const secondaryAlpha = cue.overrides.filter(
|
||
(command) => !command.animated && command.name.toLowerCase() === '2a',
|
||
);
|
||
if (secondaryAlpha.length < MIN_TEXTURE_ALPHA_OVERRIDES) {
|
||
return false;
|
||
}
|
||
// Real signs can alternate secondary alpha between words. Texture payloads switch it
|
||
// per glyph, so sparse word-level styling must not seed a texture-font group.
|
||
if (glyphs.length > secondaryAlpha.length * MAX_TEXTURE_GLYPHS_PER_ALPHA_OVERRIDE) {
|
||
return false;
|
||
}
|
||
const alphaValues = secondaryAlpha.map((command) => command.args.toLowerCase());
|
||
return new Set(alphaValues).size >= 2;
|
||
}
|
||
|
||
function staticGlobalAlpha(cue: AnnotatedSubtitleCue): number | null {
|
||
return staticAlphaOverride(cue, 'alpha');
|
||
}
|
||
|
||
function staticAlphaOverride(cue: AnnotatedSubtitleCue, expectedName: string): number | null {
|
||
let alpha: number | null = null;
|
||
for (const command of cue.overrides) {
|
||
if (command.animated || command.name.toLowerCase() !== expectedName) continue;
|
||
const match = ASS_ALPHA_VALUE_PATTERN.exec(command.args.trim());
|
||
const alphaValue = match?.[1];
|
||
if (alphaValue !== undefined) {
|
||
alpha = Number.parseInt(alphaValue, 16);
|
||
}
|
||
}
|
||
return alpha;
|
||
}
|
||
|
||
function hasAnimatedAlphaOverride(cue: AnnotatedSubtitleCue): boolean {
|
||
return cue.overrides.some((command) => {
|
||
if (!command.animated) return false;
|
||
const name = command.name.toLowerCase();
|
||
return name === '1a' || name === 'alpha';
|
||
});
|
||
}
|
||
|
||
// `\alpha&HFF&` blanks all four layers, but a later component override can turn one
|
||
// back on: chant overlays render entirely through `\4a&H00&` shadows. Any re-enabled
|
||
// layer means the event draws real text.
|
||
const ASS_COMPONENT_ALPHA_NAMES = ['1a', '2a', '3a', '4a'] as const;
|
||
|
||
function hasVisibleComponentAlpha(cue: AnnotatedSubtitleCue): boolean {
|
||
return ASS_COMPONENT_ALPHA_NAMES.some((name) => {
|
||
const value = staticAlphaOverride(cue, name);
|
||
return value !== null && value < 0xff;
|
||
});
|
||
}
|
||
|
||
/**
|
||
* A glyph copy whose fill is statically fully transparent and never animated back in is
|
||
* a glow or outline echo of the real glyph, not the text itself. A coalesced copy chain
|
||
* counts only when every phase in it is such an echo; one opaque phase means the chain
|
||
* carries the authored glyph.
|
||
*/
|
||
function isTransparentFillEcho(cue: AnnotatedSubtitleCue): boolean {
|
||
return sourceEventsOf(cue).every(
|
||
(source) =>
|
||
(staticAlphaOverride(source, '1a') === 0xff ||
|
||
staticAlphaOverride(source, 'alpha') === 0xff) &&
|
||
!hasAnimatedAlphaOverride(source) &&
|
||
!hasVisibleComponentAlpha(source),
|
||
);
|
||
}
|
||
|
||
function staticFontSize(cue: AnnotatedSubtitleCue): number | null {
|
||
let fontSize: number | null = null;
|
||
for (const command of cue.overrides) {
|
||
if (command.animated || command.name.toLowerCase() !== 'fs') continue;
|
||
const value = Number(command.args.trim());
|
||
if (Number.isFinite(value) && value > 0) {
|
||
fontSize = value;
|
||
}
|
||
}
|
||
return fontSize;
|
||
}
|
||
|
||
function isNearlyTransparentPositionedText(cue: AnnotatedSubtitleCue): boolean {
|
||
const alpha = staticGlobalAlpha(cue);
|
||
return (
|
||
alpha !== null &&
|
||
alpha >= MIN_TEXTURE_LAYER_ALPHA &&
|
||
staticFontOverride(cue) !== null &&
|
||
fragmentPosition(cue) !== null
|
||
);
|
||
}
|
||
|
||
function hasAssTextureCandidateEvidence(cue: AnnotatedSubtitleCue): boolean {
|
||
return (
|
||
isClippedRepeatedGlyphFragment(cue) ||
|
||
isNearlyTransparentPositionedText(cue) ||
|
||
hasStaticOverride(cue, '2a')
|
||
);
|
||
}
|
||
|
||
function textureFontFamilyKey(font: string): string {
|
||
return font.replace(ASS_FONT_WEIGHT_SUFFIX_PATTERN, '');
|
||
}
|
||
|
||
function isTextureFontPayload(
|
||
cue: AnnotatedSubtitleCue,
|
||
textureFontFamilies: ReadonlySet<string>,
|
||
): boolean {
|
||
const font = staticFontOverride(cue);
|
||
const fontSize = staticFontSize(cue);
|
||
const visibleLines = cue.text.split('\n').filter((line) => line.trim().length > 0);
|
||
return (
|
||
font !== null &&
|
||
textureFontFamilies.has(textureFontFamilyKey(font)) &&
|
||
fontSize !== null &&
|
||
fontSize <= MAX_TEXTURE_PAYLOAD_FONT_SIZE &&
|
||
staticGlobalAlpha(cue) !== null &&
|
||
fragmentPosition(cue) !== null &&
|
||
visibleLines.length >= MIN_TEXTURE_PAYLOAD_LINES
|
||
);
|
||
}
|
||
|
||
// A sign translation can legitimately render faint text through one or two positioned
|
||
// events. Dozens of them sharing one window is a texture: near-invisible glyph strings
|
||
// laid out as pixels of an image, with no visible-text sibling to anchor them.
|
||
const MIN_TEXTURE_WALL_EVENTS = 6;
|
||
|
||
function textureWallGroupKey(cue: AnnotatedSubtitleCue): string {
|
||
return `${cue.style}\0${cue.startTime}\0${cue.endTime}`;
|
||
}
|
||
|
||
/** Static zero scale or a degenerate static clip renders nothing, unless animation can
|
||
* still bring the event into view (an entrance growing from `\fscx0`, a clip wipe).
|
||
* Only an animated scale or clip reveals; `\t(...)` wrapping some other property leaves
|
||
* the event invisible. Nested `\t(...)` needs no special case because the tags it
|
||
* animates are recorded as animated in their own right. */
|
||
function isInvisiblyRenderedEvent(cue: AnnotatedSubtitleCue): boolean {
|
||
let staticZeroScale = false;
|
||
let staticZeroClip = false;
|
||
let animatedReveal = false;
|
||
for (const command of cue.overrides) {
|
||
const name = command.name.toLowerCase();
|
||
if (command.animated) {
|
||
if (name === 'fscx' || name === 'fscy' || name === 'clip') {
|
||
animatedReveal = true;
|
||
}
|
||
continue;
|
||
}
|
||
if (name === 'fscx' || name === 'fscy') {
|
||
if (Number(command.args.trim()) === 0) {
|
||
staticZeroScale = true;
|
||
}
|
||
} else if (name === 'clip') {
|
||
const args = command.args.split(',').map((value) => Number(value.trim()));
|
||
if (
|
||
args.length >= 4 &&
|
||
args.slice(0, 4).every(Number.isFinite) &&
|
||
(args[0]! >= args[2]! || args[1]! >= args[3]!)
|
||
) {
|
||
staticZeroClip = true;
|
||
}
|
||
}
|
||
}
|
||
return (staticZeroScale || staticZeroClip) && !animatedReveal;
|
||
}
|
||
|
||
function assFontTextureGroupKey(cue: AnnotatedSubtitleCue): string | null {
|
||
const font = staticFontOverride(cue);
|
||
return font === null ? null : `${cue.style}\0${cue.startTime}\0${cue.endTime}\0${font}`;
|
||
}
|
||
|
||
function assTextureTimingGroupKey(cue: AnnotatedSubtitleCue): string {
|
||
return `${cue.style}\0${cue.startTime}\0${cue.endTime}`;
|
||
}
|
||
|
||
function removeAssFontTextureEvents(events: ParsedAssEvents): ParsedAssEvents {
|
||
const seeds = events.dialogue.filter(isAssTextureSeed);
|
||
const seedSet = new Set(seeds);
|
||
// Short pieces can share the seeded font effect under another actor. Matching the
|
||
// seed's style, timing, and font only narrows the candidates; each piece must still
|
||
// carry structural texture evidence.
|
||
const textureGroups = new Set(
|
||
seeds.map(assFontTextureGroupKey).filter((key): key is string => key !== null),
|
||
);
|
||
const noFontTextureTimings = new Set(
|
||
seeds
|
||
.filter((seed) => staticFontOverride(seed) === null)
|
||
.map((seed) => assTextureTimingGroupKey(seed)),
|
||
);
|
||
// Some signs switch actor and font between the texture mask and its payload. A nearly
|
||
// transparent text event that overlaps a proven seed in the same style is another input
|
||
// to that visual effect. Opaque authored text in the same sign remains publishable.
|
||
const seedsByStyle = new Map<string, AnnotatedSubtitleCue[]>();
|
||
for (const seed of seeds) {
|
||
const styleSeeds = seedsByStyle.get(seed.style);
|
||
if (styleSeeds) {
|
||
styleSeeds.push(seed);
|
||
} else {
|
||
seedsByStyle.set(seed.style, [seed]);
|
||
}
|
||
}
|
||
const seedIndexesByStyle = new Map(
|
||
[...seedsByStyle].map(([style, styleSeeds]) => [style, buildAssEventGroupIndex(styleSeeds)]),
|
||
);
|
||
const isAssociatedWithTextureSeed = (cue: AnnotatedSubtitleCue): boolean => {
|
||
const styleSeedIndex = seedIndexesByStyle.get(cue.style);
|
||
return (
|
||
styleSeedIndex !== undefined &&
|
||
eventsOverlappingWindow(styleSeedIndex, cue.startTime, cue.endTime).length > 0
|
||
);
|
||
};
|
||
const textureFontFamilies = new Set(
|
||
[
|
||
...seeds,
|
||
...events.dialogue.filter(
|
||
(cue) => isNearlyTransparentPositionedText(cue) && isAssociatedWithTextureSeed(cue),
|
||
),
|
||
...events.comments.filter(
|
||
(cue) => isNearlyTransparentPositionedText(cue) && isAssociatedWithTextureSeed(cue),
|
||
),
|
||
]
|
||
.map(staticFontOverride)
|
||
.filter((font): font is string => font !== null)
|
||
.map(textureFontFamilyKey),
|
||
);
|
||
|
||
// Wall membership additionally requires that no component alpha turns a layer back
|
||
// on: `\alpha&HFF&` plus a visible `\4a` renders real text through its shadow, and a
|
||
// sign typeset entirely from such layers must not read as a texture.
|
||
const isTextureWallCandidate = (cue: AnnotatedSubtitleCue): boolean =>
|
||
isNearlyTransparentPositionedText(cue) && !hasVisibleComponentAlpha(cue);
|
||
const transparentWallCounts = new Map<string, number>();
|
||
for (const cue of events.dialogue) {
|
||
if (isTextureWallCandidate(cue)) {
|
||
const key = textureWallGroupKey(cue);
|
||
transparentWallCounts.set(key, (transparentWallCounts.get(key) ?? 0) + 1);
|
||
}
|
||
}
|
||
|
||
return {
|
||
dialogue: events.dialogue.filter((cue) => {
|
||
if (isInvisiblyRenderedEvent(cue)) {
|
||
return false;
|
||
}
|
||
if (seedSet.has(cue)) {
|
||
return false;
|
||
}
|
||
if (
|
||
isTextureWallCandidate(cue) &&
|
||
(transparentWallCounts.get(textureWallGroupKey(cue)) ?? 0) >= MIN_TEXTURE_WALL_EVENTS
|
||
) {
|
||
return false;
|
||
}
|
||
if (
|
||
staticFontOverride(cue) === null &&
|
||
noFontTextureTimings.has(assTextureTimingGroupKey(cue)) &&
|
||
isClippedRepeatedGlyphFragment(cue)
|
||
) {
|
||
return false;
|
||
}
|
||
const key = assFontTextureGroupKey(cue);
|
||
if (key !== null && textureGroups.has(key) && hasAssTextureCandidateEvidence(cue)) {
|
||
return false;
|
||
}
|
||
if (isTextureFontPayload(cue, textureFontFamilies)) {
|
||
return false;
|
||
}
|
||
if (!isNearlyTransparentPositionedText(cue)) {
|
||
return true;
|
||
}
|
||
return !isAssociatedWithTextureSeed(cue);
|
||
}),
|
||
comments: events.comments,
|
||
};
|
||
}
|
||
|
||
/**
|
||
* Generated lyric effects often layer decoration over the real syllables: single letters
|
||
* positioned above each glyph, animated in, and rendered through a `\fn` override to a
|
||
* symbol font where `a` draws as a sparkle rather than a letter. Reading them as text
|
||
* corrupts the reconstructed line (`sotto mimi ni ateru to` gains a trailing `a z x`).
|
||
* Within one style/name group, a font used only for scattered animated single glyphs --
|
||
* while the group's actual text renders in another font -- marks those events as
|
||
* decoration rather than dialogue.
|
||
*/
|
||
function decorativeGlyphEvents(events: readonly AnnotatedSubtitleCue[]): Set<AnnotatedSubtitleCue> {
|
||
const byFont = new Map<string, AnnotatedSubtitleCue[]>();
|
||
for (const cue of events) {
|
||
const font = staticFontOverride(cue);
|
||
if (font === null) continue;
|
||
const group = byFont.get(font);
|
||
if (group) {
|
||
group.push(cue);
|
||
} else {
|
||
byFont.set(font, [cue]);
|
||
}
|
||
}
|
||
|
||
const decorative = new Set<AnnotatedSubtitleCue>();
|
||
for (const fontEvents of byFont.values()) {
|
||
if (fontEvents.length * 2 >= events.length) continue;
|
||
const allScatteredGlyphs = fontEvents.every(
|
||
(cue) =>
|
||
[...compactCueMatchText(cue)].length === 1 &&
|
||
fragmentPosition(cue) !== null &&
|
||
hasAssTemporalOverride(cue.overrides),
|
||
);
|
||
if (allScatteredGlyphs) {
|
||
fontEvents.forEach((cue) => decorative.add(cue));
|
||
}
|
||
}
|
||
return decorative;
|
||
}
|
||
|
||
function recoverFragmentOnlyAssLines(dialogue: AnnotatedSubtitleCue[]): AnnotatedSubtitleCue[] {
|
||
const groups = new Map<string, AnnotatedSubtitleCue[]>();
|
||
for (const cue of dialogue) {
|
||
if (cue.source !== undefined) {
|
||
continue;
|
||
}
|
||
const key = assEventGroupKey(cue);
|
||
const group = groups.get(key);
|
||
if (group) {
|
||
group.push(cue);
|
||
} else {
|
||
groups.set(key, [cue]);
|
||
}
|
||
}
|
||
|
||
const recovered: AnnotatedSubtitleCue[] = [];
|
||
const suppressed = new Set<AnnotatedSubtitleCue>();
|
||
// The published dialogue list holds source events, so a suppressed coalesced copy
|
||
// chain must suppress every event behind it.
|
||
const suppress = (event: AnnotatedSubtitleCue): void => {
|
||
for (const source of sourceEventsOf(event)) {
|
||
suppressed.add(source);
|
||
}
|
||
};
|
||
for (const events of groups.values()) {
|
||
const units = coalesceAssAnchorCopies(events);
|
||
const decorative = decorativeGlyphEvents(units);
|
||
for (const unit of units) {
|
||
if (
|
||
!decorative.has(unit) &&
|
||
fragmentPlacementAnchors(unit).size > 0 &&
|
||
isTransparentFillEcho(unit)
|
||
) {
|
||
decorative.add(unit);
|
||
}
|
||
}
|
||
const lineEvents = decorative.size ? units.filter((event) => !decorative.has(event)) : units;
|
||
if (isProgressiveHighlightSweepGroup(lineEvents)) {
|
||
lineEvents.forEach(suppress);
|
||
const spanStart = Math.min(...lineEvents.map((event) => event.startTime));
|
||
const spanEnd = Math.max(...lineEvents.map((event) => event.endTime));
|
||
for (const overlay of decorative) {
|
||
if (
|
||
overlay.startTime < spanEnd + DECORATION_SPAN_TOLERANCE_SECONDS &&
|
||
overlay.endTime > spanStart - DECORATION_SPAN_TOLERANCE_SECONDS
|
||
) {
|
||
suppress(overlay);
|
||
}
|
||
}
|
||
continue;
|
||
}
|
||
for (const cluster of clusterAssFragmentEvents(lineEvents)) {
|
||
const line = reconstructAssFragmentLine(cluster.events);
|
||
if (!line) {
|
||
continue;
|
||
}
|
||
// A sweep only re-highlights the lyric it decorates: hide its events without
|
||
// publishing the reconstruction.
|
||
if (isProgressiveHighlightSweep(cluster.events)) {
|
||
cluster.events.forEach(suppress);
|
||
continue;
|
||
}
|
||
recovered.push(line);
|
||
cluster.events.forEach(suppress);
|
||
// Decoration is timed to the line it overlays, so it disappears with the line's
|
||
// full animation span. Decoration outside any recovered span stays published.
|
||
const spanStart = line.animationStartTime ?? line.startTime;
|
||
const spanEnd = line.animationEndTime ?? line.endTime;
|
||
for (const overlay of decorative) {
|
||
if (
|
||
overlay.startTime < spanEnd + DECORATION_SPAN_TOLERANCE_SECONDS &&
|
||
overlay.endTime > spanStart - DECORATION_SPAN_TOLERANCE_SECONDS
|
||
) {
|
||
suppress(overlay);
|
||
}
|
||
}
|
||
}
|
||
}
|
||
if (recovered.length === 0 && suppressed.size === 0) {
|
||
return dialogue;
|
||
}
|
||
return [
|
||
...dialogue.filter((cue) => !suppressed.has(cue)),
|
||
...mergeAbuttingRecoveredLines(recovered),
|
||
].sort(
|
||
(left, right) =>
|
||
left.startTime - right.startTime || left.endTime - right.endTime || left.order - right.order,
|
||
);
|
||
}
|
||
|
||
// One authored line can reconstruct twice from consecutive effect stages -- its steady
|
||
// glyphs, then an exit animation replaying the same text. Publishing both would show the
|
||
// line restarting, so identical recoveries that touch in time collapse into one span.
|
||
const RECOVERED_LINE_MERGE_GAP_SECONDS = 0.1;
|
||
|
||
function mergeAbuttingRecoveredLines(recovered: AnnotatedSubtitleCue[]): AnnotatedSubtitleCue[] {
|
||
const byText = new Map<string, AnnotatedSubtitleCue[]>();
|
||
for (const line of recovered) {
|
||
const key = `${assEventGroupKey(line)}\0${compactAssMatchText(line.text)}`;
|
||
const bucket = byText.get(key);
|
||
if (bucket) {
|
||
bucket.push(line);
|
||
} else {
|
||
byText.set(key, [line]);
|
||
}
|
||
}
|
||
|
||
const merged: AnnotatedSubtitleCue[] = [];
|
||
for (const bucket of byText.values()) {
|
||
bucket.sort((left, right) => left.startTime - right.startTime || left.order - right.order);
|
||
let current = bucket[0]!;
|
||
for (let index = 1; index < bucket.length; index += 1) {
|
||
const next = bucket[index]!;
|
||
if (next.startTime <= current.endTime + RECOVERED_LINE_MERGE_GAP_SECONDS) {
|
||
const endTime = Math.max(current.endTime, next.endTime);
|
||
current = {
|
||
...current,
|
||
endTime,
|
||
animationEndTime: Math.max(
|
||
current.animationEndTime ?? current.endTime,
|
||
next.animationEndTime ?? next.endTime,
|
||
),
|
||
};
|
||
} else {
|
||
merged.push(current);
|
||
current = next;
|
||
}
|
||
}
|
||
merged.push(current);
|
||
}
|
||
return merged;
|
||
}
|
||
|
||
function groupConsecutiveAssFragments(events: readonly AnnotatedSubtitleCue[]): FragmentGroup[] {
|
||
const groups: FragmentGroup[] = [];
|
||
for (const event of events) {
|
||
const text = compactCueMatchText(event);
|
||
if (!text) {
|
||
continue;
|
||
}
|
||
const previous = groups.at(-1);
|
||
if (
|
||
previous?.text === text &&
|
||
previous.events.some((previousEvent) => isRepeatedFragmentCopy(previousEvent, event))
|
||
) {
|
||
previous.events.push(event);
|
||
} else {
|
||
groups.push({ text, events: [event] });
|
||
}
|
||
}
|
||
return groups;
|
||
}
|
||
|
||
function findCanonicalFragmentEvents(
|
||
events: readonly AnnotatedSubtitleCue[],
|
||
canonicalText: string,
|
||
): AnnotatedSubtitleCue[] {
|
||
const groups = groupConsecutiveAssFragments(events);
|
||
const matches = new Set<AnnotatedSubtitleCue>();
|
||
|
||
for (let start = 0; start < groups.length; start += 1) {
|
||
let combined = '';
|
||
for (let end = start; end < groups.length; end += 1) {
|
||
const group = groups[end]!;
|
||
// A complete rendered copy cannot prove that the neighboring events are its
|
||
// fragments. Exact full-line animation is handled separately for comments.
|
||
if (group.text.length >= canonicalText.length) {
|
||
break;
|
||
}
|
||
const next = combined + group.text;
|
||
if (!canonicalText.startsWith(next)) {
|
||
break;
|
||
}
|
||
combined = next;
|
||
if (combined !== canonicalText) {
|
||
continue;
|
||
}
|
||
for (let index = start; index <= end; index += 1) {
|
||
for (const event of groups[index]!.events) {
|
||
matches.add(event);
|
||
}
|
||
}
|
||
start = end;
|
||
break;
|
||
}
|
||
}
|
||
|
||
return [...matches];
|
||
}
|
||
|
||
function matchingAssAnimationEvents(options: {
|
||
candidate: AnnotatedSubtitleCue;
|
||
group: AssEventGroupIndex;
|
||
allowFullLineFrames: boolean;
|
||
}): AnnotatedSubtitleCue[] {
|
||
const canonicalText = compactCueMatchText(options.candidate);
|
||
// The group index already restricts to the candidate's style and name.
|
||
const nearby = eventsOverlappingWindow(
|
||
options.group,
|
||
options.candidate.startTime - CANONICAL_MATCH_MARGIN_SECONDS,
|
||
options.candidate.endTime + CANONICAL_MATCH_MARGIN_SECONDS,
|
||
);
|
||
const fragments = findCanonicalFragmentEvents(nearby, canonicalText);
|
||
if (fragments.length >= MIN_CANONICAL_ANIMATION_EVENTS && hasAssAnimationEvidence(fragments)) {
|
||
return fragments;
|
||
}
|
||
|
||
if (!options.allowFullLineFrames) {
|
||
return [];
|
||
}
|
||
const fullLineFrames = nearby.filter((cue) => compactCueMatchText(cue) === canonicalText);
|
||
return fullLineFrames.length >= MIN_CANONICAL_ANIMATION_EVENTS &&
|
||
hasAssAnimationEvidence(fullLineFrames)
|
||
? fullLineFrames
|
||
: [];
|
||
}
|
||
|
||
// Reductions rather than `Math.min(...events)`: one generated line can carry an
|
||
// unbounded number of events, and spreading them all as arguments risks the engine's
|
||
// argument-count limit.
|
||
function earliestStartTime(events: readonly AnnotatedSubtitleCue[], seed = Infinity): number {
|
||
return events.reduce((earliest, event) => Math.min(earliest, event.startTime), seed);
|
||
}
|
||
|
||
function latestEndTime(events: readonly AnnotatedSubtitleCue[], seed = -Infinity): number {
|
||
return events.reduce((latest, event) => Math.max(latest, event.endTime), seed);
|
||
}
|
||
|
||
function includeCanonicalBoundaryEvents(options: {
|
||
candidate: AnnotatedSubtitleCue;
|
||
group: AssEventGroupIndex;
|
||
animationEvents: readonly AnnotatedSubtitleCue[];
|
||
}): AnnotatedSubtitleCue[] {
|
||
const canonicalText = compactCueMatchText(options.candidate);
|
||
const startTime = earliestStartTime(options.animationEvents);
|
||
const endTime = latestEndTime(options.animationEvents);
|
||
return eventsOverlappingWindow(
|
||
options.group,
|
||
startTime - CANONICAL_MATCH_MARGIN_SECONDS,
|
||
endTime + CANONICAL_MATCH_MARGIN_SECONDS,
|
||
).filter((cue) => compactCueMatchText(cue) === canonicalText);
|
||
}
|
||
|
||
function recoverCanonicalAssEvents({
|
||
dialogue,
|
||
comments,
|
||
}: ParsedAssEvents): AnnotatedSubtitleCue[] {
|
||
const recovered: AnnotatedSubtitleCue[] = [];
|
||
const suppressed = new Set<AnnotatedSubtitleCue>();
|
||
// A recovery is only as good as its owning event. When a later candidate proves that
|
||
// an earlier candidate was itself a generated frame of its animation, the earlier
|
||
// recovery is a duplicate of the same authored line and must be withdrawn.
|
||
const recoveredByOwner = new Map<AnnotatedSubtitleCue, AnnotatedSubtitleCue>();
|
||
const withdrawn = new Set<AnnotatedSubtitleCue>();
|
||
const eventsByGroup = new Map<string, AnnotatedSubtitleCue[]>();
|
||
for (const cue of dialogue) {
|
||
const key = assEventGroupKey(cue);
|
||
const group = eventsByGroup.get(key);
|
||
if (group) {
|
||
group.push(cue);
|
||
} else {
|
||
eventsByGroup.set(key, [cue]);
|
||
}
|
||
}
|
||
const indexByGroup = new Map<string, AssEventGroupIndex>();
|
||
for (const [key, events] of eventsByGroup) {
|
||
indexByGroup.set(key, buildAssEventGroupIndex(events));
|
||
}
|
||
const emptyGroupIndex: AssEventGroupIndex = { byStart: [], prefixMaxEnd: [] };
|
||
const candidates = [
|
||
...comments.map((cue) => ({ cue, kind: 'comment' as const })),
|
||
...dialogue
|
||
.filter(
|
||
(cue) =>
|
||
compactCueMatchText(cue).length >= MIN_CANONICAL_DIALOGUE_TEXT_LENGTH &&
|
||
(cue.assLayout?.kind === 'source-order' || hasAssAnimationEvidence([cue])),
|
||
)
|
||
.sort((left, right) => right.text.length - left.text.length || left.order - right.order)
|
||
.map((cue) => ({ cue, kind: 'dialogue' as const })),
|
||
];
|
||
|
||
for (const { cue: candidate, kind } of candidates) {
|
||
if (candidate.endTime <= candidate.startTime || suppressed.has(candidate)) {
|
||
continue;
|
||
}
|
||
const canonicalText = compactCueMatchText(candidate);
|
||
if (!canonicalText) {
|
||
continue;
|
||
}
|
||
|
||
const group = indexByGroup.get(assEventGroupKey(candidate)) ?? emptyGroupIndex;
|
||
const animationEvents = matchingAssAnimationEvents({
|
||
candidate,
|
||
group,
|
||
allowFullLineFrames: kind === 'comment',
|
||
});
|
||
if (animationEvents.length === 0) {
|
||
continue;
|
||
}
|
||
|
||
const boundaryEvents = includeCanonicalBoundaryEvents({
|
||
candidate,
|
||
group,
|
||
animationEvents,
|
||
});
|
||
const generatedEvents = [...new Set([...animationEvents, ...boundaryEvents])];
|
||
const animationStartTime = earliestStartTime(generatedEvents, candidate.startTime);
|
||
const animationEndTime = latestEndTime(generatedEvents, candidate.endTime);
|
||
const startTime = kind === 'comment' ? candidate.startTime : animationStartTime;
|
||
const endTime = kind === 'comment' ? candidate.endTime : animationEndTime;
|
||
const recoveredCue: AnnotatedSubtitleCue = {
|
||
...candidate,
|
||
startTime,
|
||
endTime,
|
||
animationStartTime,
|
||
animationEndTime,
|
||
source: 'canonical-ass',
|
||
};
|
||
recovered.push(recoveredCue);
|
||
recoveredByOwner.set(candidate, recoveredCue);
|
||
for (const event of generatedEvents) {
|
||
suppressed.add(event);
|
||
if (event === candidate) {
|
||
continue;
|
||
}
|
||
const priorRecovery = recoveredByOwner.get(event);
|
||
if (priorRecovery) {
|
||
// No text is lost by withdrawing: a fragment claim means the withdrawn line is
|
||
// a contiguous piece of this candidate's text, and a boundary claim means the
|
||
// texts are equal, so the surviving canonical cue always contains it.
|
||
withdrawn.add(priorRecovery);
|
||
}
|
||
}
|
||
}
|
||
|
||
const survivingRecovered = recovered.filter((cue) => !withdrawn.has(cue));
|
||
if (survivingRecovered.length === 0) {
|
||
return dialogue;
|
||
}
|
||
return [...dialogue.filter((cue) => !suppressed.has(cue)), ...survivingRecovered].sort(
|
||
(a, b) => a.startTime - b.startTime || a.endTime - b.endTime || a.order - b.order,
|
||
);
|
||
}
|
||
|
||
function parseAssCoordinate(value: string | undefined): number | null {
|
||
if (!value?.trim()) return null;
|
||
const coordinate = Number(value.trim());
|
||
return Number.isFinite(coordinate) ? coordinate : null;
|
||
}
|
||
|
||
function buildAssCueLayout(
|
||
overrides: readonly AssOverrideCommand[],
|
||
sourceOrder: number,
|
||
): AssCueLayout {
|
||
let y: number | null = null;
|
||
for (const command of overrides) {
|
||
if (command.animated) continue;
|
||
const name = command.name.toLowerCase();
|
||
const args = command.args.split(',');
|
||
if (name === 'pos') {
|
||
y = parseAssCoordinate(args[1]) ?? y;
|
||
continue;
|
||
}
|
||
if (name !== 'move') continue;
|
||
const startY = parseAssCoordinate(args[1]);
|
||
const endY = parseAssCoordinate(args[3]);
|
||
if (startY !== null && endY !== null) {
|
||
y = (startY + endY) / 2;
|
||
}
|
||
}
|
||
return y === null
|
||
? { kind: 'source-order', sourceOrder }
|
||
: { kind: 'positioned', sourceOrder, y };
|
||
}
|
||
|
||
function parseAnnotatedAssEvents(content: string): ParsedAssEvents {
|
||
const cues: AnnotatedSubtitleCue[] = [];
|
||
const comments: AnnotatedSubtitleCue[] = [];
|
||
const lines = content.split(/\r?\n/);
|
||
let inEventsSection = false;
|
||
let eventOrder = 0;
|
||
const fieldIndex = {
|
||
start: -1,
|
||
end: -1,
|
||
text: -1,
|
||
style: -1,
|
||
layer: -1,
|
||
name: -1,
|
||
effect: -1,
|
||
};
|
||
|
||
const resetFieldIndex = () => {
|
||
fieldIndex.start = -1;
|
||
fieldIndex.end = -1;
|
||
fieldIndex.text = -1;
|
||
fieldIndex.style = -1;
|
||
fieldIndex.layer = -1;
|
||
fieldIndex.name = -1;
|
||
fieldIndex.effect = -1;
|
||
};
|
||
|
||
for (const line of lines) {
|
||
const trimmed = line.trim();
|
||
// Event text can end in an authored space. Fragmented karaoke commonly uses that
|
||
// space to retain word boundaries when its separately positioned events are joined
|
||
// back into a line, so only remove indentation before slicing the event fields.
|
||
const eventLine = line.trimStart();
|
||
|
||
if (trimmed.startsWith('[') && trimmed.endsWith(']')) {
|
||
inEventsSection = trimmed.toLowerCase() === '[events]';
|
||
if (!inEventsSection) {
|
||
resetFieldIndex();
|
||
}
|
||
continue;
|
||
}
|
||
|
||
if (!inEventsSection) {
|
||
continue;
|
||
}
|
||
|
||
if (trimmed.startsWith(ASS_FORMAT_PREFIX)) {
|
||
const formatFields = trimmed
|
||
.slice(ASS_FORMAT_PREFIX.length)
|
||
.split(',')
|
||
.map((field) => field.trim().toLowerCase());
|
||
fieldIndex.start = formatFields.indexOf('start');
|
||
fieldIndex.end = formatFields.indexOf('end');
|
||
fieldIndex.text = formatFields.indexOf('text');
|
||
fieldIndex.style = formatFields.indexOf('style');
|
||
fieldIndex.layer = formatFields.indexOf('layer');
|
||
// Aegisub writes the speaker column as `Actor`; the v4+ spec calls it `Name`.
|
||
// Missing it costs the burst check its speaker guard, so both spellings count.
|
||
fieldIndex.name = findFieldIndex(formatFields, ASS_NAME_FIELD_ALIASES);
|
||
fieldIndex.effect = formatFields.indexOf('effect');
|
||
continue;
|
||
}
|
||
|
||
const eventPrefix = eventLine.startsWith(ASS_DIALOGUE_PREFIX)
|
||
? ASS_DIALOGUE_PREFIX
|
||
: eventLine.startsWith(ASS_COMMENT_PREFIX)
|
||
? ASS_COMMENT_PREFIX
|
||
: null;
|
||
if (!eventPrefix) {
|
||
continue;
|
||
}
|
||
|
||
if (fieldIndex.start < 0 || fieldIndex.end < 0 || fieldIndex.text < 0) {
|
||
continue;
|
||
}
|
||
|
||
const fields = eventLine.slice(eventPrefix.length).split(',');
|
||
if (
|
||
fieldIndex.start >= fields.length ||
|
||
fieldIndex.end >= fields.length ||
|
||
fieldIndex.text >= fields.length
|
||
) {
|
||
continue;
|
||
}
|
||
|
||
const startTime = parseAssTimestamp(fields[fieldIndex.start]!);
|
||
const endTime = parseAssTimestamp(fields[fieldIndex.end]!);
|
||
if (startTime === null || endTime === null || endTime <= startTime) {
|
||
continue;
|
||
}
|
||
|
||
const rawText = fields.slice(fieldIndex.text).join(',');
|
||
const text = sanitizeAssCueText(rawText);
|
||
if (!text) {
|
||
continue;
|
||
}
|
||
|
||
const effect = readField(fields, fieldIndex.effect);
|
||
const layer = Number(readField(fields, fieldIndex.layer));
|
||
const overrides = collectAssOverrideCommands(rawText);
|
||
const cue: AnnotatedSubtitleCue = {
|
||
startTime,
|
||
endTime,
|
||
text,
|
||
rawText,
|
||
style: readField(fields, fieldIndex.style),
|
||
layer: Number.isFinite(layer) ? layer : 0,
|
||
name: readField(fields, fieldIndex.name),
|
||
effect,
|
||
effectKind: parseAssEffectField(effect),
|
||
overrides,
|
||
overrideSignature: assOverrideSignature(overrides),
|
||
order: eventOrder,
|
||
assLayout: buildAssCueLayout(overrides, eventOrder),
|
||
};
|
||
eventOrder += 1;
|
||
if (eventPrefix === ASS_COMMENT_PREFIX) {
|
||
comments.push(cue);
|
||
} else {
|
||
cues.push(cue);
|
||
}
|
||
}
|
||
|
||
return { dialogue: cues, comments };
|
||
}
|
||
|
||
function parseAnnotatedAssCues(content: string): AnnotatedSubtitleCue[] {
|
||
const events = removeAssFontTextureEvents(parseAnnotatedAssEvents(content));
|
||
return recoverFragmentOnlyAssLines(recoverCanonicalAssEvents(events));
|
||
}
|
||
|
||
export function parseAssCues(content: string): SubtitleCue[] {
|
||
return toPublicCues(parseAnnotatedAssCues(content));
|
||
}
|
||
|
||
function detectSubtitleFormat(source: string): 'srt' | 'vtt' | 'ass' | 'ssa' | null {
|
||
const [normalizedSource = source] =
|
||
(() => {
|
||
try {
|
||
return /^[a-z]+:\/\//i.test(source) ? new URL(source).pathname : source;
|
||
} catch {
|
||
return source;
|
||
}
|
||
})().split(/[?#]/, 1)[0] ?? '';
|
||
const ext = normalizedSource.split('.').pop()?.toLowerCase() ?? '';
|
||
if (ext === 'srt') return 'srt';
|
||
if (ext === 'vtt') return 'vtt';
|
||
if (ext === 'ass' || ext === 'ssa') return 'ass';
|
||
return null;
|
||
}
|
||
|
||
export function parseSubtitleCues(content: string, filename: string): SubtitleCue[] {
|
||
const format = detectSubtitleFormat(filename);
|
||
let cues: AnnotatedSubtitleCue[];
|
||
let sourceFormat: SubtitleSourceFormat = 'srt';
|
||
|
||
switch (format) {
|
||
case 'srt':
|
||
case 'vtt':
|
||
cues = parseAnnotatedSrtCues(content);
|
||
break;
|
||
case 'ass':
|
||
case 'ssa':
|
||
cues = parseAnnotatedAssCues(content);
|
||
sourceFormat = 'ass';
|
||
break;
|
||
default:
|
||
cues = [];
|
||
}
|
||
|
||
if (cues.length === 0) {
|
||
const assCues = parseAnnotatedAssCues(content);
|
||
const srtCues = parseAnnotatedSrtCues(content);
|
||
const preferAss = assCues.length >= srtCues.length;
|
||
cues = preferAss ? assCues : srtCues;
|
||
sourceFormat = preferAss && assCues.length > 0 ? 'ass' : 'srt';
|
||
}
|
||
|
||
cues.sort((a, b) => a.startTime - b.startTime || a.endTime - b.endTime || a.order - b.order);
|
||
return toPublicCues(mergeDuplicateCues(cues, sourceFormat));
|
||
}
|