fix(tokenizer): bound retry ladder and cache, fix name-match edge cases

- Cap only blind shrinking-window retries (not backend-guided shrinks); a line that exhausts the cap escalates to one parseText fallback instead of dropping to raw text
- Bound the cross-line termsFind cache by retained dictionary-entry weight, not just key count, and re-check it when a lookup resolves
- Keep halfwidth katakana character names in the greedy pre-pass, and let a generic word beat a name it fully contains
- Share one Han code-point table between the character dictionary and the scanner's name pre-pass; narrow the mob-disambiguator filter to the split letters, not every one-character term
- Release the subtitle prefetch pause on a new onProcessingSettled signal instead of the tokenized emit, so duplicate/suppressed/failed lines no longer pause prefetch indefinitely
- Split the Yomitan scan runtime's injected helper script into its own file
This commit is contained in:
2026-08-05 01:22:10 -07:00
parent 3c24597724
commit afa66ee508
19 changed files with 1352 additions and 587 deletions
@@ -215,6 +215,7 @@ function createPrimingRuntimeWithRealController(options: {
text: string;
calls: string[];
onTokenize: () => void;
tokenize?: (text: string) => SubtitleData | null | Promise<SubtitleData | null>;
cacheLimit?: number;
}) {
const { text, calls } = options;
@@ -222,15 +223,42 @@ function createPrimingRuntimeWithRealController(options: {
let currentSubtitleData: SubtitleData | null = null;
const mediaPath = '/media/video.mkv';
const prefetchService = {
pause: () => calls.push('prefetch:pause'),
resume: () => calls.push('prefetch:resume'),
};
// Mirrors main.ts emitSubtitlePayload: an emit resumes prefetching unless it
// is explicitly marked as not the end of the work for the line, and every
// controller emit is so marked.
const emitSubtitlePayload = (
payload: SubtitleData,
emitOptions?: { resumePrefetch?: boolean },
): void => {
currentSubtitleData = payload;
calls.push(
emitOptions?.resumePrefetch === false
? `emit-raw:${payload.text}`
: `emit-direct:${payload.text}`,
);
if (emitOptions?.resumePrefetch !== false) {
prefetchService.resume();
}
};
const subtitleProcessingController = createSubtitleProcessingController({
tokenizeSubtitle: async (subtitleText) => {
options.onTokenize();
return { text: subtitleText, tokens: [] };
return options.tokenize ? options.tokenize(subtitleText) : { text: subtitleText, tokens: [] };
},
// main.ts routes controller emits through emitSubtitlePayload with
// resumePrefetch: false, so they never release the pause on their own.
emitSubtitle: (payload) => {
currentSubtitleData = payload;
calls.push(`emit:${payload.text}:tokens=${payload.tokens === null ? 'none' : 'yes'}`);
},
onProcessingSettled: () => {
prefetchService.resume();
},
...(options.cacheLimit === undefined ? {} : { cacheLimit: options.cacheLimit }),
});
@@ -249,17 +277,8 @@ function createPrimingRuntimeWithRealController(options: {
getActiveParsedSubtitleCues: () => [],
setActiveParsedSubtitleMediaPath: () => {},
subtitleProcessingController,
emitSubtitlePayload: (payload, emitOptions) => {
if (emitOptions?.resumePrefetch === false) {
calls.push(`emit-raw:${payload.text}`);
return;
}
calls.push(`emit-direct:${payload.text}`);
},
getSubtitlePrefetchService: () => ({
pause: () => calls.push('prefetch:pause'),
resume: () => calls.push('prefetch:resume'),
}),
emitSubtitlePayload,
getSubtitlePrefetchService: () => prefetchService,
getLastObservedTimePos: () => 12,
getVisibleOverlayVisible: () => true,
emitSecondarySubtitle: () => {},
@@ -336,3 +355,68 @@ test('primeCurrentSubtitleForAutoplay releases the prefetch pause when nothing i
assert.deepEqual(calls, ['prefetch:pause', `emit-raw:${text}`, 'prefetch:resume']);
});
test('primeCurrentSubtitleForAutoplay releases the prefetch pause when tokenization emits nothing', async () => {
const calls: string[] = [];
const text = '起動字幕';
const { runtime, subtitleProcessingController, mediaPath } =
createPrimingRuntimeWithRealController({
text,
calls,
onTokenize: () => {},
// Transient tokenizer failure: the controller falls back to plain text it
// has already shown, so it suppresses the emit entirely.
tokenize: () => null,
});
subtitleProcessingController.onSubtitleChange(text);
await new Promise((resolve) => setTimeout(resolve, 0));
subtitleProcessingController.invalidateTokenizationCache();
calls.length = 0;
await runtime.primeCurrentSubtitleForAutoplay(mediaPath);
await new Promise((resolve) => setTimeout(resolve, 0));
assert.ok(
!calls.some((call) => call.startsWith('emit:')),
`expected no controller emit, saw ${JSON.stringify(calls)}`,
);
assert.equal(
calls.filter((call) => call === 'prefetch:resume').length,
1,
`expected the prefetch pause to be released, saw ${JSON.stringify(calls)}`,
);
});
test('prefetch stays paused until tokenization of an uncached line completes', async () => {
const calls: string[] = [];
const text = '起動字幕';
let finishTokenization = (): void => {};
const tokenizationGate = new Promise<void>((resolve) => {
finishTokenization = resolve;
});
const { subtitleProcessingController } = createPrimingRuntimeWithRealController({
text,
calls,
onTokenize: () => {},
tokenize: async (subtitleText) => {
await tokenizationGate;
return { text: subtitleText, tokens: [] };
},
});
subtitleProcessingController.onSubtitleChange(text);
await new Promise((resolve) => setTimeout(resolve, 0));
// The provisional plain emit must not release the pause: the expensive scan
// is still ahead of it and would compete with prefetching for the parser.
assert.deepEqual(calls, [`emit:${text}:tokens=none`]);
finishTokenization();
await new Promise((resolve) => setTimeout(resolve, 0));
assert.deepEqual(calls, [
`emit:${text}:tokens=none`,
`emit:${text}:tokens=yes`,
'prefetch:resume',
]);
});