fix(subtitle-prefetch): reuse startup runs and prefetch overlap lines

- Warm combined text for overlapping subtitle cues.
- Preserve valid in-flight tokenization across startup restarts while rejecting invalidated cache results.
This commit is contained in:
2026-09-28 01:17:59 -07:00
parent 4648c5e5a1
commit a4325740b7
9 changed files with 274 additions and 6 deletions
+5
View File
@@ -0,0 +1,5 @@
type: fixed
area: annotations
- Subtitle lines shown while two cues overlap (common in Jimaku and Jellyfin SRTs, and usually the long lines) are now prefetched, so their annotations appear immediately instead of after a delay.
- Subtitle prefetching no longer restarts each time the same subtitle file is re-selected during startup, and a line tokenized before a restart or seek is kept instead of discarded.
@@ -300,3 +300,65 @@ test('prefetch service deduplicates repeated cue text within a run', async () =>
);
assert.ok(tokenizedTexts.includes('other'));
});
test('prefetch service keeps an in-flight result from a superseded run', async () => {
const cues = makeCues(5);
const cachedTexts: string[] = [];
let releaseFirst!: () => void;
const firstGate = new Promise<void>((resolve) => {
releaseFirst = resolve;
});
const service = createSubtitlePrefetchService({
cues,
tokenizeSubtitle: async (text) => {
if (text === 'line-0') await firstGate;
return { text, tokens: [] };
},
preCacheTokenization: (text) => {
cachedTexts.push(text);
},
getCacheGeneration: () => 0,
priorityWindowSize: 3,
});
service.start(0);
await flushMicrotasks();
service.stop();
releaseFirst();
await flushMicrotasks();
assert.deepEqual(cachedTexts, ['line-0']);
});
test('prefetch service drops an in-flight result after cache invalidation', async () => {
const cues = makeCues(5);
const cachedTexts: string[] = [];
let generation = 0;
let releaseFirst!: () => void;
const firstGate = new Promise<void>((resolve) => {
releaseFirst = resolve;
});
const service = createSubtitlePrefetchService({
cues,
tokenizeSubtitle: async (text) => {
if (text === 'line-0') await firstGate;
return { text, tokens: [] };
},
preCacheTokenization: (text) => {
cachedTexts.push(text);
},
getCacheGeneration: () => generation,
priorityWindowSize: 3,
});
service.start(0);
await flushMicrotasks();
service.stop();
generation += 1;
releaseFirst();
await flushMicrotasks();
assert.deepEqual(cachedTexts, []);
});
+10 -1
View File
@@ -7,6 +7,12 @@ export interface SubtitlePrefetchServiceDeps {
tokenizeSubtitle: (text: string) => Promise<SubtitleData | null>;
preCacheTokenization: (text: string, data: SubtitleData) => void;
hasCachedTokenization?: (text: string) => boolean;
/**
* Bumped whenever the tokenization cache is invalidated. A result is cached only if
* the generation is unchanged since its tokenization began; without this dep every
* result counts as fresh.
*/
getCacheGeneration?: () => number;
priorityWindowSize?: number;
}
@@ -88,9 +94,12 @@ export function createSubtitlePrefetchService(
}
warmedKeys.add(cacheKey);
// Freshness follows the cache generation, not the run: a seek or restart that
// supersedes this run leaves the result valid, and the parser time is already spent.
const generation = deps.getCacheGeneration?.();
try {
const result = await deps.tokenizeSubtitle(cue.text);
if (result && !stopped && runId === currentRunId) {
if (result && generation === deps.getCacheGeneration?.()) {
deps.preCacheTokenization(cue.text, result);
}
} catch {
@@ -43,6 +43,8 @@ export interface SubtitleProcessingController {
*/
notePlainSubtitleEmitted: (text: string) => void;
invalidateTokenizationCache: () => void;
/** Incremented by invalidateTokenizationCache; lets async writers detect stale results. */
getCacheGeneration: () => number;
preCacheTokenization: (text: string, data: SubtitleData) => void;
consumeCachedSubtitle: (text: string) => SubtitleData | null;
hasCachedSubtitle: (text: string) => boolean;
@@ -253,6 +255,7 @@ export function createSubtitleProcessingController(
tokenizationCache.clear();
cacheGeneration += 1;
},
getCacheGeneration: () => cacheGeneration,
preCacheTokenization: (text: string, data: SubtitleData) => {
setCachedTokenization(text, data);
},
+1
View File
@@ -2152,6 +2152,7 @@ const subtitlePrefetchInitController = createSubtitlePrefetchInitController({
subtitleProcessingController.preCacheTokenization(text, data);
},
hasCachedTokenization: (text) => subtitleProcessingController.hasCachedSubtitle(text),
getCacheGeneration: () => subtitleProcessingController.getCacheGeneration(),
logInfo: (message) => logger.info(message),
logWarn: (message) => logger.warn(message),
onParsedSubtitleCuesChanged: (cues, sourceKey) => {
@@ -251,3 +251,99 @@ test('subtitle prefetch init logs a warning when the source parses to zero cues'
assert.equal(warnings.length, 1);
assert.match(warnings[0]!, /\[subtitle-prefetch\].*0 cues.*\/tmp\/broken\.ass/);
});
function createTrackingController(loadContent: () => string) {
let currentService: SubtitlePrefetchService | null = null;
const events: string[] = [];
const cueUpdates: Array<string | null> = [];
let serviceCount = 0;
const controller = createSubtitlePrefetchInitController({
getCurrentService: () => currentService,
setCurrentService: (service) => {
currentService = service;
},
loadSubtitleSourceText: async () => loadContent(),
parseSubtitleCues: (content): SubtitleCue[] => [{ startTime: 0, endTime: 1, text: content }],
createSubtitlePrefetchService: () => {
const id = ++serviceCount;
return {
start: () => events.push(`start:${id}`),
stop: () => events.push(`stop:${id}`),
onSeek: () => {},
pause: () => {},
resume: () => {},
};
},
tokenizeSubtitle: async () => null,
preCacheTokenization: () => {},
logInfo: () => {},
logWarn: () => {},
onParsedSubtitleCuesChanged: (cues, source) => {
cueUpdates.push(cues ? `${source}:${cues[0]!.text}` : null);
},
});
return { controller, events, cueUpdates };
}
test('re-initializing an unchanged source keeps the running service and republishes cues', async () => {
const { controller, events, cueUpdates } = createTrackingController(() => 'content');
await controller.initSubtitlePrefetch('track-1.srt', 0);
await controller.initSubtitlePrefetch('track-1.srt', 3);
assert.deepEqual(events, ['start:1']);
assert.deepEqual(cueUpdates, ['track-1.srt:content', 'track-1.srt:content']);
});
test('re-initializing a source whose content changed restarts the service', async () => {
let content = 'first';
const { controller, events } = createTrackingController(() => content);
await controller.initSubtitlePrefetch('track-1.srt', 0);
content = 'second';
await controller.initSubtitlePrefetch('track-1.srt', 0);
assert.deepEqual(events, ['start:1', 'stop:1', 'start:2']);
});
test('subtitle prefetch init warms overlap lines without publishing them as cues', async () => {
let prefetchedTexts: string[] = [];
let publishedTexts: string[] = [];
let currentService: SubtitlePrefetchService | null = null;
const controller = createSubtitlePrefetchInitController({
getCurrentService: () => currentService,
setCurrentService: (service) => {
currentService = service;
},
loadSubtitleSourceText: async () => 'content',
parseSubtitleCues: (): SubtitleCue[] => [
{ startTime: 0, endTime: 4, text: 'first' },
{ startTime: 2, endTime: 6, text: 'second' },
],
createSubtitlePrefetchService: ({ cues }) => {
prefetchedTexts = cues.map((cue) => cue.text);
return {
start: () => {},
stop: () => {},
onSeek: () => {},
pause: () => {},
resume: () => {},
};
},
tokenizeSubtitle: async () => null,
preCacheTokenization: () => {},
logInfo: () => {},
logWarn: () => {},
onParsedSubtitleCuesChanged: (cues) => {
publishedTexts = cues?.map((cue) => cue.text) ?? [];
},
});
await controller.initSubtitlePrefetch('track.srt', 0);
assert.deepEqual(prefetchedTexts, ['first', 'second', 'first\n\nsecond']);
assert.deepEqual(publishedTexts, ['first', 'second']);
});
+35 -5
View File
@@ -4,6 +4,7 @@ import type {
} from '../../core/services/subtitle-prefetch';
import type { SubtitleData } from '../../types';
import type { SubtitleCue } from '../../types';
import { buildOverlapPrefetchCues } from './subtitle-prefetch-overlaps';
export interface SubtitlePrefetchInitControllerDeps {
getCurrentService: () => SubtitlePrefetchService | null;
@@ -14,6 +15,7 @@ export interface SubtitlePrefetchInitControllerDeps {
tokenizeSubtitle: (text: string) => Promise<SubtitleData | null>;
preCacheTokenization: (text: string, data: SubtitleData) => void;
hasCachedTokenization?: (text: string) => boolean;
getCacheGeneration?: () => number;
logInfo: (message: string) => void;
logWarn: (message: string) => void;
onParsedSubtitleCuesChanged?: (cues: SubtitleCue[] | null, sourceKey: string | null) => void;
@@ -32,11 +34,20 @@ export function createSubtitlePrefetchInitController(
deps: SubtitlePrefetchInitControllerDeps,
): SubtitlePrefetchInitController {
let initRevision = 0;
// What the running service was built from. Startup fires several inits for one source
// (Jellyfin preload seed, sid change, each track-list update); restarting on each would
// discard the in-flight tokenization and re-queue the priority window behind it.
let activeSource: { key: string; content: string; cues: SubtitleCue[] } | null = null;
const stopCurrentService = (): void => {
deps.getCurrentService()?.stop();
deps.setCurrentService(null);
activeSource = null;
};
const cancelPendingInit = (): void => {
initRevision += 1;
deps.getCurrentService()?.stop();
deps.setCurrentService(null);
stopCurrentService();
deps.onParsedSubtitleCuesChanged?.(null, null);
};
@@ -46,14 +57,26 @@ export function createSubtitlePrefetchInitController(
sourceKey = sourcePath,
): Promise<void> => {
const revision = ++initRevision;
deps.getCurrentService()?.stop();
deps.setCurrentService(null);
// The same source may still be current once loaded; keep its service running until then.
if (activeSource?.key !== sourceKey) {
stopCurrentService();
}
try {
const content = await deps.loadSubtitleSourceText(sourcePath);
if (revision !== initRevision) {
return;
}
if (
activeSource?.key === sourceKey &&
activeSource.content === content &&
deps.getCurrentService()
) {
// A track change clears the published cues synchronously, so republish them.
deps.onParsedSubtitleCuesChanged?.(activeSource.cues, sourceKey);
return;
}
stopCurrentService();
const cues = deps.parseSubtitleCues(content, sourcePath);
if (revision !== initRevision || cues.length === 0) {
@@ -66,11 +89,16 @@ export function createSubtitlePrefetchInitController(
return;
}
// Overlap cues only feed the cache; the published cue list stays as authored.
const prefetchCues = [...cues, ...buildOverlapPrefetchCues(cues)].sort(
(a, b) => a.startTime - b.startTime,
);
const nextService = deps.createSubtitlePrefetchService({
cues,
cues: prefetchCues,
tokenizeSubtitle: (text) => deps.tokenizeSubtitle(text),
preCacheTokenization: (text, data) => deps.preCacheTokenization(text, data),
hasCachedTokenization: (text) => deps.hasCachedTokenization?.(text) ?? false,
getCacheGeneration: deps.getCacheGeneration,
});
if (revision !== initRevision) {
@@ -78,6 +106,7 @@ export function createSubtitlePrefetchInitController(
}
deps.setCurrentService(nextService);
activeSource = { key: sourceKey, content, cues };
deps.onParsedSubtitleCuesChanged?.(cues, sourceKey);
nextService.start(currentTimePos);
deps.logInfo(
@@ -85,6 +114,7 @@ export function createSubtitlePrefetchInitController(
);
} catch (error) {
if (revision === initRevision) {
stopCurrentService();
deps.onParsedSubtitleCuesChanged?.(null, null);
deps.logWarn(`[subtitle-prefetch] failed to initialize: ${(error as Error).message}`);
}
@@ -0,0 +1,24 @@
import assert from 'node:assert/strict';
import test from 'node:test';
import type { SubtitleCue } from '../../types';
import { buildOverlapPrefetchCues } from './subtitle-prefetch-overlaps';
test('overlapping cues yield the combined line the live path resolves', () => {
const cues: SubtitleCue[] = [
{ startTime: 0, endTime: 4, text: 'このまま頑張ったって―' },
{ startTime: 2, endTime: 6, text: 'ちょっとカズマ\n聞こえてんの?' },
];
assert.deepEqual(buildOverlapPrefetchCues(cues), [
{ startTime: 2, endTime: 4, text: 'このまま頑張ったって―\n\nちょっとカズマ\n聞こえてんの?' },
]);
});
test('back-to-back cues yield no overlap lines', () => {
const cues: SubtitleCue[] = [
{ startTime: 0, endTime: 2, text: 'first' },
{ startTime: 2, endTime: 4, text: 'second' },
];
assert.deepEqual(buildOverlapPrefetchCues(cues), []);
});
@@ -0,0 +1,38 @@
import type { SubtitleCue } from '../../types';
import { resolvePrimarySubtitleText } from './primary-subtitle-text';
/**
* While cues overlap, mpv publishes their texts joined as one `sub-text`, and the live
* path tokenizes and caches that combined line as its own entry. Prefetching single cues
* never warms it, so overlapping lines (usually the long ones) always miss the cache.
*
* Returns one synthetic cue per overlap span carrying the text the live path resolves
* for it, so the prefetcher can warm those lines in timeline order with the rest.
*/
export function buildOverlapPrefetchCues(cues: readonly SubtitleCue[]): SubtitleCue[] {
const boundaries = [...new Set(cues.flatMap((cue) => [cue.startTime, cue.endTime]))].sort(
(a, b) => a - b,
);
const singleTexts = new Set(cues.map((cue) => cue.text));
const seen = new Set<string>();
const overlapCues: SubtitleCue[] = [];
for (let i = 0; i + 1 < boundaries.length; i += 1) {
const startTime = boundaries[i]!;
const endTime = boundaries[i + 1]!;
const midpoint = (startTime + endTime) / 2;
const active = cues.filter((cue) => cue.startTime <= midpoint && cue.endTime > midpoint);
if (active.length < 2) continue;
const text = resolvePrimarySubtitleText({
liveText: active.map((cue) => cue.text).join('\n'),
currentTimeSec: midpoint,
cues,
});
if (!text.trim() || singleTexts.has(text) || seen.has(text)) continue;
seen.add(text);
overlapCues.push({ startTime, endTime, text });
}
return overlapCues;
}