mirror of
https://github.com/ksyasuda/SubMiner.git
synced 2026-08-07 19:21:32 -07:00
261 lines
9.3 KiB
TypeScript
261 lines
9.3 KiB
TypeScript
import type { SubtitleData } from '../../types';
|
|
|
|
export interface SubtitleProcessingControllerDeps {
|
|
tokenizeSubtitle: (text: string) => Promise<SubtitleData | null>;
|
|
emitSubtitle: (payload: SubtitleData) => void;
|
|
/**
|
|
* Fires when the controller runs out of work: every scheduled line has been
|
|
* processed, whether it ended in an emit, a suppressed duplicate, or a
|
|
* tokenizer failure. Callers that hold a resource for the duration of
|
|
* processing (prefetch pausing) release it here rather than on an emit,
|
|
* which is not guaranteed to happen.
|
|
*/
|
|
onProcessingSettled?: () => void;
|
|
logDebug?: (message: string) => void;
|
|
now?: () => number;
|
|
cacheLimit?: number;
|
|
}
|
|
|
|
/**
|
|
* Pure memory bound on the LRU, not a coverage limit: prefetching runs to the end of a
|
|
* file regardless of cache pressure. Sized to hold a feature-length title (a 24-minute
|
|
* episode runs 300-400 lines, a 2-hour film ~2000) plus room for lines that repeat across
|
|
* episodes of a series, so openings and endings stay warm between titles.
|
|
*/
|
|
export const DEFAULT_SUBTITLE_TOKENIZATION_CACHE_LIMIT = 2500;
|
|
|
|
export interface SubtitleProcessingController {
|
|
/**
|
|
* Returns whether processing is now scheduled or already in flight for this
|
|
* event. A false return means the controller is idle and will do nothing, so
|
|
* onProcessingSettled will not fire; callers that pause work for the duration
|
|
* of processing (such as subtitle prefetching) must release it themselves.
|
|
*/
|
|
onSubtitleChange: (text: string) => boolean;
|
|
/** Same contract as onSubtitleChange: whether processing is pending. */
|
|
refreshCurrentSubtitle: (textOverride?: string) => boolean;
|
|
/**
|
|
* Records that this exact text has already been shown plain by someone else
|
|
* (autoplay priming paints its first frame before scheduling tokenization),
|
|
* so the controller does not repeat that payload on its way to the tokenized
|
|
* one.
|
|
*/
|
|
notePlainSubtitleEmitted: (text: string) => void;
|
|
invalidateTokenizationCache: () => void;
|
|
preCacheTokenization: (text: string, data: SubtitleData) => void;
|
|
consumeCachedSubtitle: (text: string) => SubtitleData | null;
|
|
hasCachedSubtitle: (text: string) => boolean;
|
|
}
|
|
|
|
export function normalizeSubtitleCacheKey(text: string): string {
|
|
return text.replace(/\r\n/g, '\n').replace(/\\N/g, '\n').replace(/\\n/g, '\n').trim();
|
|
}
|
|
|
|
export function createSubtitleProcessingController(
|
|
deps: SubtitleProcessingControllerDeps,
|
|
): SubtitleProcessingController {
|
|
const SUBTITLE_TOKENIZATION_CACHE_LIMIT =
|
|
deps.cacheLimit && deps.cacheLimit > 0
|
|
? deps.cacheLimit
|
|
: DEFAULT_SUBTITLE_TOKENIZATION_CACHE_LIMIT;
|
|
let latestText = '';
|
|
let lastEmittedText = '';
|
|
// Tracks the latest provisional plain emit across rapid changes and loop retries
|
|
// so the same line is never shown plain twice.
|
|
let lastPlainEmittedText: string | null = null;
|
|
let cacheGeneration = 0;
|
|
let lastEmittedGeneration = 0;
|
|
let processing = false;
|
|
let staleDropCount = 0;
|
|
const tokenizationCache = new Map<string, SubtitleData>();
|
|
const now = deps.now ?? (() => Date.now());
|
|
|
|
const getCachedTokenization = (text: string): SubtitleData | null => {
|
|
const cacheKey = normalizeSubtitleCacheKey(text);
|
|
const cached = tokenizationCache.get(cacheKey);
|
|
if (!cached) {
|
|
return null;
|
|
}
|
|
|
|
tokenizationCache.delete(cacheKey);
|
|
tokenizationCache.set(cacheKey, cached);
|
|
return cached;
|
|
};
|
|
|
|
const setCachedTokenization = (text: string, payload: SubtitleData): void => {
|
|
tokenizationCache.set(normalizeSubtitleCacheKey(text), payload);
|
|
while (tokenizationCache.size > SUBTITLE_TOKENIZATION_CACHE_LIMIT) {
|
|
const firstKey = tokenizationCache.keys().next().value;
|
|
if (firstKey !== undefined) {
|
|
tokenizationCache.delete(firstKey);
|
|
}
|
|
}
|
|
};
|
|
|
|
const processLatest = (): void => {
|
|
if (processing) {
|
|
return;
|
|
}
|
|
|
|
processing = true;
|
|
|
|
void (async () => {
|
|
while (true) {
|
|
const text = latestText;
|
|
const generation = cacheGeneration;
|
|
const startedAtMs = now();
|
|
|
|
if (!text.trim()) {
|
|
if (lastPlainEmittedText !== text) {
|
|
deps.emitSubtitle({ text, tokens: null });
|
|
}
|
|
lastEmittedText = text;
|
|
lastEmittedGeneration = generation;
|
|
lastPlainEmittedText = null;
|
|
break;
|
|
}
|
|
|
|
let output: SubtitleData = { text, tokens: null };
|
|
try {
|
|
const cachedTokenized = getCachedTokenization(text);
|
|
if (cachedTokenized) {
|
|
output = cachedTokenized;
|
|
} else {
|
|
// Cache miss: show the plain line on time; the tokenized payload
|
|
// upgrades it once ready. Skipped on refreshes of an already
|
|
// emitted line so downstream consumers never see a downgrade.
|
|
if (text !== lastEmittedText && text !== lastPlainEmittedText) {
|
|
deps.emitSubtitle({ text, tokens: null });
|
|
lastPlainEmittedText = text;
|
|
}
|
|
const tokenized = await deps.tokenizeSubtitle(text);
|
|
// A null result is a transient tokenizer failure, not a verdict on
|
|
// the line: caching the plain fallback would pin it untokenized for
|
|
// every later occurrence.
|
|
if (tokenized) {
|
|
output = tokenized;
|
|
// A result computed before an invalidation must not repopulate the
|
|
// fresh cache, or the retry below would serve the stale entry.
|
|
if (generation === cacheGeneration) {
|
|
setCachedTokenization(text, tokenized);
|
|
}
|
|
}
|
|
}
|
|
} catch (error) {
|
|
deps.logDebug?.(`Subtitle tokenization failed: ${(error as Error).message}`);
|
|
}
|
|
|
|
if (latestText !== text) {
|
|
staleDropCount += 1;
|
|
deps.logDebug?.(
|
|
`Dropped stale subtitle tokenization result; dropped=${staleDropCount}, elapsed=${now() - startedAtMs}ms`,
|
|
);
|
|
continue;
|
|
}
|
|
|
|
if (generation !== cacheGeneration) {
|
|
deps.logDebug?.(
|
|
`Dropped stale subtitle tokenization result after cache invalidation; elapsed=${now() - startedAtMs}ms`,
|
|
);
|
|
continue;
|
|
}
|
|
|
|
// An untokenized result adds nothing when this line was already shown,
|
|
// either provisionally or as an earlier full emit (failed refresh) —
|
|
// emitting it would duplicate or downgrade what is on screen.
|
|
const plainAlreadyShown = lastPlainEmittedText === text || lastEmittedText === text;
|
|
if (!(output.tokens === null && output.text === text && plainAlreadyShown)) {
|
|
deps.emitSubtitle(output);
|
|
}
|
|
lastEmittedText = text;
|
|
lastEmittedGeneration = generation;
|
|
lastPlainEmittedText = null;
|
|
deps.logDebug?.(
|
|
`Subtitle tokenization delivered; elapsed=${now() - startedAtMs}ms, staleDrops=${staleDropCount}`,
|
|
);
|
|
break;
|
|
}
|
|
})()
|
|
.catch((error) => {
|
|
deps.logDebug?.(`Subtitle processing loop failed: ${(error as Error).message}`);
|
|
})
|
|
.finally(() => {
|
|
processing = false;
|
|
if (
|
|
latestText !== lastEmittedText ||
|
|
(latestText.trim() && cacheGeneration !== lastEmittedGeneration)
|
|
) {
|
|
processLatest();
|
|
return;
|
|
}
|
|
// Nothing left to do: signal completion even when this run emitted
|
|
// nothing (suppressed duplicate, tokenizer failure), or callers waiting
|
|
// on the controller would wait forever.
|
|
deps.onProcessingSettled?.();
|
|
});
|
|
};
|
|
|
|
return {
|
|
onSubtitleChange: (text: string) => {
|
|
if (text === latestText) {
|
|
// A run already in flight for this text will still emit for it.
|
|
return processing;
|
|
}
|
|
latestText = text;
|
|
if (
|
|
processing &&
|
|
text !== lastPlainEmittedText &&
|
|
!tokenizationCache.has(normalizeSubtitleCacheKey(text))
|
|
) {
|
|
deps.emitSubtitle({ text, tokens: null });
|
|
lastPlainEmittedText = text;
|
|
}
|
|
processLatest();
|
|
return true;
|
|
},
|
|
refreshCurrentSubtitle: (textOverride?: string) => {
|
|
if (typeof textOverride === 'string') {
|
|
latestText = textOverride;
|
|
}
|
|
if (!latestText.trim()) {
|
|
// A run in flight will pick this up and emit the empty subtitle, so
|
|
// the caller is still waiting on an emit.
|
|
return processing;
|
|
}
|
|
if (processing) {
|
|
return true;
|
|
}
|
|
if (latestText === lastEmittedText && cacheGeneration === lastEmittedGeneration) {
|
|
return false;
|
|
}
|
|
processLatest();
|
|
return true;
|
|
},
|
|
notePlainSubtitleEmitted: (text: string) => {
|
|
lastPlainEmittedText = text;
|
|
},
|
|
invalidateTokenizationCache: () => {
|
|
tokenizationCache.clear();
|
|
cacheGeneration += 1;
|
|
},
|
|
preCacheTokenization: (text: string, data: SubtitleData) => {
|
|
setCachedTokenization(text, data);
|
|
},
|
|
consumeCachedSubtitle: (text: string) => {
|
|
const cached = getCachedTokenization(text);
|
|
if (!cached) {
|
|
return null;
|
|
}
|
|
|
|
latestText = text;
|
|
lastEmittedText = text;
|
|
lastEmittedGeneration = cacheGeneration;
|
|
lastPlainEmittedText = null;
|
|
return cached;
|
|
},
|
|
hasCachedSubtitle: (text: string) => {
|
|
return tokenizationCache.has(normalizeSubtitleCacheKey(text));
|
|
},
|
|
};
|
|
}
|