mirror of
https://github.com/ksyasuda/SubMiner.git
synced 2026-08-04 07:21:32 -07:00
refactor(tokenizer): address review feedback on Yomitan scan runtime
- extract the injected scan runtime (helpers, install script, call-script builder) into tokenizer/yomitan-scan-runtime-script.ts; the host module drops from ~2700 to ~1900 lines - append the kana run to the reading as well as the surface when an unparsed run extends the previous token, so the reading keeps covering the surface and the known-word reading fallback stays enabled (bumps scan runtime version) - stop annotateMs before character-image resolution so the stage timing measures the annotation stage only
This commit is contained in:
@@ -8,3 +8,4 @@ area: subtitles
|
|||||||
- Tokenizer runtime dependencies are built once instead of per line, fixing a JLPT lookup cache that never hit (it was keyed on a per-call closure identity and leaked a Map per line) and a `which mecab` availability check that re-ran synchronously on every line when MeCab is absent.
|
- Tokenizer runtime dependencies are built once instead of per line, fixing a JLPT lookup cache that never hit (it was keyed on a per-call closure identity and leaked a Map per line) and a `which mecab` availability check that re-ran synchronously on every line when MeCab is absent.
|
||||||
- Subtitle changes no longer restart the prefetch run per line (which discarded in-flight tokenization work); prefetch now only pauses for the live line and restarts on real seeks, cache invalidation, or option changes. Prefetch also stays paused across a provisional raw-subtitle emit and resumes only after the tokenized payload lands, so it never competes with the on-screen line for the parser window.
|
- Subtitle changes no longer restart the prefetch run per line (which discarded in-flight tokenization work); prefetch now only pauses for the live line and restarts on real seeks, cache invalidation, or option changes. Prefetch also stays paused across a provisional raw-subtitle emit and resumes only after the tokenized payload lands, so it never competes with the on-screen line for the parser window.
|
||||||
- Added per-stage debug timings (`scanMs`, `mecabMs`, `frequencyMs`, `annotateMs`) to the subtitle tokenization pipeline log.
|
- Added per-stage debug timings (`scanMs`, `mecabMs`, `frequencyMs`, `annotateMs`) to the subtitle tokenization pipeline log.
|
||||||
|
- Fixed a reading that stopped covering its surface when an unmatched kana run extended the preceding token (for example a trailing る on 待ち合わせ), which silently disabled the known-word reading fallback for those tokens.
|
||||||
|
|||||||
@@ -920,8 +920,8 @@ export async function tokenizeSubtitle(
|
|||||||
if (yomitanTokens && yomitanTokens.length > 0) {
|
if (yomitanTokens && yomitanTokens.length > 0) {
|
||||||
const annotateStartedAtMs = Date.now();
|
const annotateStartedAtMs = Date.now();
|
||||||
const annotatedTokens = await applyAnnotationStage(yomitanTokens, deps, annotationOptions);
|
const annotatedTokens = await applyAnnotationStage(yomitanTokens, deps, annotationOptions);
|
||||||
const renderedTokens = applyCharacterNameImages(annotatedTokens, deps, annotationOptions);
|
|
||||||
stageTimings.annotateMs = Date.now() - annotateStartedAtMs;
|
stageTimings.annotateMs = Date.now() - annotateStartedAtMs;
|
||||||
|
const renderedTokens = applyCharacterNameImages(annotatedTokens, deps, annotationOptions);
|
||||||
logStageTimings(renderedTokens.length);
|
logStageTimings(renderedTokens.length);
|
||||||
return {
|
return {
|
||||||
text: displayText,
|
text: displayText,
|
||||||
|
|||||||
@@ -787,6 +787,52 @@ test('requestYomitanScanTokens warns when active Yomitan profile has no dictiona
|
|||||||
});
|
});
|
||||||
});
|
});
|
||||||
|
|
||||||
|
test('requestYomitanScanTokens keeps reading aligned when a kana run extends the previous token', async () => {
|
||||||
|
const deps = createScanDeps((action, params) => {
|
||||||
|
if (action === 'optionsGetFull') {
|
||||||
|
return {
|
||||||
|
profileCurrent: 0,
|
||||||
|
profiles: [{ options: { scanning: { length: 40 } } }],
|
||||||
|
};
|
||||||
|
}
|
||||||
|
if (action === 'getDictionaryInfo') {
|
||||||
|
return [];
|
||||||
|
}
|
||||||
|
const text = (params as { text?: string } | undefined)?.text ?? '';
|
||||||
|
// 待ち合わせ matches, the trailing る does not, so the kana run extends the
|
||||||
|
// previous token instead of becoming its own filler token.
|
||||||
|
if (text.startsWith('待ち合わせ')) {
|
||||||
|
return {
|
||||||
|
originalTextLength: 5,
|
||||||
|
dictionaryEntries: [
|
||||||
|
{
|
||||||
|
headwords: [
|
||||||
|
{
|
||||||
|
term: '待ち合わせる',
|
||||||
|
reading: 'まちあわせる',
|
||||||
|
sources: [{ originalText: '待ち合わせ', isPrimary: true, matchType: 'exact' }],
|
||||||
|
},
|
||||||
|
],
|
||||||
|
},
|
||||||
|
],
|
||||||
|
};
|
||||||
|
}
|
||||||
|
return { originalTextLength: 0, dictionaryEntries: [] };
|
||||||
|
});
|
||||||
|
|
||||||
|
const result = await requestYomitanScanTokens('待ち合わせる', deps, {
|
||||||
|
error: () => undefined,
|
||||||
|
});
|
||||||
|
|
||||||
|
assert.equal(result?.length, 1);
|
||||||
|
assert.equal(result?.[0]?.surface, '待ち合わせる');
|
||||||
|
assert.equal(result?.[0]?.endPos, 6);
|
||||||
|
// The reading must grow with the surface: a short reading fails
|
||||||
|
// isCompleteReadingForSurface and silently disables the known-word reading
|
||||||
|
// fallback downstream.
|
||||||
|
assert.equal(result?.[0]?.reading, 'まちあわせる');
|
||||||
|
});
|
||||||
|
|
||||||
test('requestYomitanScanTokens emits unparsed filler runs for text the scanner skips', async () => {
|
test('requestYomitanScanTokens emits unparsed filler runs for text the scanner skips', async () => {
|
||||||
const deps = createScanDeps((action, params) => {
|
const deps = createScanDeps((action, params) => {
|
||||||
if (action === 'optionsGetFull') {
|
if (action === 'optionsGetFull') {
|
||||||
|
|||||||
@@ -3,6 +3,13 @@ import * as fs from 'fs';
|
|||||||
import * as http from 'http';
|
import * as http from 'http';
|
||||||
import * as path from 'path';
|
import * as path from 'path';
|
||||||
import { selectYomitanParseTokens } from './parser-selection-stage';
|
import { selectYomitanParseTokens } from './parser-selection-stage';
|
||||||
|
import {
|
||||||
|
buildYomitanScanCallScript,
|
||||||
|
CHARACTER_DICTIONARY_TITLE_PREFIX,
|
||||||
|
YOMITAN_SCAN_RUNTIME_INSTALL_SCRIPT,
|
||||||
|
YOMITAN_SCAN_RUNTIME_MISSING_SENTINEL,
|
||||||
|
type YomitanFrequencyMode,
|
||||||
|
} from './yomitan-scan-runtime-script';
|
||||||
|
|
||||||
interface LoggerLike {
|
interface LoggerLike {
|
||||||
error: (message: string, ...args: unknown[]) => void;
|
error: (message: string, ...args: unknown[]) => void;
|
||||||
@@ -22,8 +29,6 @@ interface YomitanParserRuntimeDeps {
|
|||||||
createYomitanExtensionWindow?: (pageName: string) => Promise<BrowserWindow | null>;
|
createYomitanExtensionWindow?: (pageName: string) => Promise<BrowserWindow | null>;
|
||||||
}
|
}
|
||||||
|
|
||||||
type YomitanFrequencyMode = 'occurrence-based' | 'rank-based';
|
|
||||||
|
|
||||||
export interface YomitanDictionaryInfo {
|
export interface YomitanDictionaryInfo {
|
||||||
title: string;
|
title: string;
|
||||||
revision?: string | number;
|
revision?: string | number;
|
||||||
@@ -74,7 +79,6 @@ export interface YomitanAddNoteResult {
|
|||||||
}
|
}
|
||||||
|
|
||||||
const DEFAULT_YOMITAN_SCAN_LENGTH = 40;
|
const DEFAULT_YOMITAN_SCAN_LENGTH = 40;
|
||||||
const CHARACTER_DICTIONARY_TITLE_PREFIX = 'SubMiner Character Dictionary';
|
|
||||||
const yomitanProfileMetadataByWindow = new WeakMap<BrowserWindow, YomitanProfileMetadata>();
|
const yomitanProfileMetadataByWindow = new WeakMap<BrowserWindow, YomitanProfileMetadata>();
|
||||||
const yomitanProfileDiagnosticsLoggedByWindow = new WeakSet<BrowserWindow>();
|
const yomitanProfileDiagnosticsLoggedByWindow = new WeakSet<BrowserWindow>();
|
||||||
const yomitanFrequencyCacheByWindow = new WeakMap<
|
const yomitanFrequencyCacheByWindow = new WeakMap<
|
||||||
@@ -826,788 +830,6 @@ async function serveDictionaryZipOnce<T>(
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
const YOMITAN_SCANNING_HELPERS = String.raw`
|
|
||||||
const HIRAGANA_CONVERSION_RANGE = [0x3041, 0x3096];
|
|
||||||
const KATAKANA_CONVERSION_RANGE = [0x30a1, 0x30f6];
|
|
||||||
const KANA_PROLONGED_SOUND_MARK_CODE_POINT = 0x30fc;
|
|
||||||
const KATAKANA_SMALL_KA_CODE_POINT = 0x30f5;
|
|
||||||
const KATAKANA_SMALL_KE_CODE_POINT = 0x30f6;
|
|
||||||
const KANA_RANGES = [[0x3040, 0x309f], [0x30a0, 0x30ff]];
|
|
||||||
const JAPANESE_RANGES = [[0x3040, 0x30ff], [0x3400, 0x9fff]];
|
|
||||||
function isCodePointInRange(codePoint, range) { return codePoint >= range[0] && codePoint <= range[1]; }
|
|
||||||
function isCodePointInRanges(codePoint, ranges) { return ranges.some((range) => isCodePointInRange(codePoint, range)); }
|
|
||||||
function isCodePointKana(codePoint) { return isCodePointInRanges(codePoint, KANA_RANGES); }
|
|
||||||
function isCodePointJapanese(codePoint) { return isCodePointInRanges(codePoint, JAPANESE_RANGES); }
|
|
||||||
function createFuriganaSegment(text, reading) { return {text, reading}; }
|
|
||||||
function getSegmentReadingContribution(segment) {
|
|
||||||
if (typeof segment.reading === "string" && segment.reading.length > 0) { return segment.reading; }
|
|
||||||
const segmentText = typeof segment.text === "string" ? segment.text : "";
|
|
||||||
const isKanaOnly = segmentText.length > 0 && [...segmentText].every((char) => isCodePointKana(char.codePointAt(0)));
|
|
||||||
return isKanaOnly ? segmentText : "";
|
|
||||||
}
|
|
||||||
function getProlongedHiragana(previousCharacter) {
|
|
||||||
switch (previousCharacter) {
|
|
||||||
case "あ": case "か": case "が": case "さ": case "ざ": case "た": case "だ": case "な": case "は": case "ば": case "ぱ": case "ま": case "や": case "ら": case "わ": case "ぁ": case "ゃ": case "ゎ": return "あ";
|
|
||||||
case "い": case "き": case "ぎ": case "し": case "じ": case "ち": case "ぢ": case "に": case "ひ": case "び": case "ぴ": case "み": case "り": case "ぃ": return "い";
|
|
||||||
case "う": case "く": case "ぐ": case "す": case "ず": case "つ": case "づ": case "ぬ": case "ふ": case "ぶ": case "ぷ": case "む": case "ゆ": case "る": case "ぅ": case "ゅ": return "う";
|
|
||||||
case "え": case "け": case "げ": case "せ": case "ぜ": case "て": case "で": case "ね": case "へ": case "べ": case "ぺ": case "め": case "れ": case "ぇ": return "え";
|
|
||||||
case "お": case "こ": case "ご": case "そ": case "ぞ": case "と": case "ど": case "の": case "ほ": case "ぼ": case "ぽ": case "も": case "よ": case "ろ": case "を": case "ぉ": case "ょ": return "う";
|
|
||||||
default: return null;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
function getFuriganaKanaSegments(text, reading) {
|
|
||||||
const newSegments = [];
|
|
||||||
let start = 0;
|
|
||||||
let state = (reading[0] === text[0]);
|
|
||||||
for (let i = 1; i < text.length; ++i) {
|
|
||||||
const newState = (reading[i] === text[i]);
|
|
||||||
if (state === newState) { continue; }
|
|
||||||
newSegments.push(createFuriganaSegment(text.substring(start, i), state ? '' : reading.substring(start, i)));
|
|
||||||
state = newState;
|
|
||||||
start = i;
|
|
||||||
}
|
|
||||||
newSegments.push(createFuriganaSegment(text.substring(start), state ? '' : reading.substring(start)));
|
|
||||||
return newSegments;
|
|
||||||
}
|
|
||||||
function convertKatakanaToHiragana(text, keepProlongedSoundMarks = false) {
|
|
||||||
let result = '';
|
|
||||||
const offset = (HIRAGANA_CONVERSION_RANGE[0] - KATAKANA_CONVERSION_RANGE[0]);
|
|
||||||
for (let char of text) {
|
|
||||||
const codePoint = char.codePointAt(0);
|
|
||||||
switch (codePoint) {
|
|
||||||
case KATAKANA_SMALL_KA_CODE_POINT:
|
|
||||||
case KATAKANA_SMALL_KE_CODE_POINT:
|
|
||||||
break;
|
|
||||||
case KANA_PROLONGED_SOUND_MARK_CODE_POINT:
|
|
||||||
if (!keepProlongedSoundMarks && result.length > 0) {
|
|
||||||
const char2 = getProlongedHiragana(result[result.length - 1]);
|
|
||||||
if (char2 !== null) { char = char2; }
|
|
||||||
}
|
|
||||||
break;
|
|
||||||
default:
|
|
||||||
if (isCodePointInRange(codePoint, KATAKANA_CONVERSION_RANGE)) {
|
|
||||||
char = String.fromCodePoint(codePoint + offset);
|
|
||||||
}
|
|
||||||
break;
|
|
||||||
}
|
|
||||||
result += char;
|
|
||||||
}
|
|
||||||
return result;
|
|
||||||
}
|
|
||||||
function segmentizeFurigana(reading, readingNormalized, groups, groupsStart) {
|
|
||||||
const groupCount = groups.length - groupsStart;
|
|
||||||
if (groupCount <= 0) { return reading.length === 0 ? [] : null; }
|
|
||||||
const group = groups[groupsStart];
|
|
||||||
const {isKana, text} = group;
|
|
||||||
if (isKana) {
|
|
||||||
if (group.textNormalized !== null && readingNormalized.startsWith(group.textNormalized)) {
|
|
||||||
const segments = segmentizeFurigana(reading.substring(text.length), readingNormalized.substring(text.length), groups, groupsStart + 1);
|
|
||||||
if (segments !== null) {
|
|
||||||
if (reading.startsWith(text)) { segments.unshift(createFuriganaSegment(text, '')); }
|
|
||||||
else { segments.unshift(...getFuriganaKanaSegments(text, reading)); }
|
|
||||||
return segments;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
return null;
|
|
||||||
}
|
|
||||||
let result = null;
|
|
||||||
for (let i = reading.length; i >= text.length; --i) {
|
|
||||||
const segments = segmentizeFurigana(reading.substring(i), readingNormalized.substring(i), groups, groupsStart + 1);
|
|
||||||
if (segments !== null) {
|
|
||||||
if (result !== null) { return null; }
|
|
||||||
segments.unshift(createFuriganaSegment(text, reading.substring(0, i)));
|
|
||||||
result = segments;
|
|
||||||
}
|
|
||||||
if (groupCount === 1) { break; }
|
|
||||||
}
|
|
||||||
return result;
|
|
||||||
}
|
|
||||||
function distributeFurigana(term, reading) {
|
|
||||||
if (reading === term) { return [createFuriganaSegment(term, '')]; }
|
|
||||||
const groups = [];
|
|
||||||
let groupPre = null;
|
|
||||||
let isKanaPre = null;
|
|
||||||
for (const c of term) {
|
|
||||||
const isKana = isCodePointKana(c.codePointAt(0));
|
|
||||||
if (isKana === isKanaPre) { groupPre.text += c; }
|
|
||||||
else {
|
|
||||||
groupPre = {isKana, text: c, textNormalized: null};
|
|
||||||
groups.push(groupPre);
|
|
||||||
isKanaPre = isKana;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
for (const group of groups) {
|
|
||||||
if (group.isKana) { group.textNormalized = convertKatakanaToHiragana(group.text); }
|
|
||||||
}
|
|
||||||
const segments = segmentizeFurigana(reading, convertKatakanaToHiragana(reading), groups, 0);
|
|
||||||
return segments !== null ? segments : [createFuriganaSegment(term, reading)];
|
|
||||||
}
|
|
||||||
function getStemLength(text1, text2) {
|
|
||||||
const minLength = Math.min(text1.length, text2.length);
|
|
||||||
if (minLength === 0) { return 0; }
|
|
||||||
let i = 0;
|
|
||||||
while (true) {
|
|
||||||
const char1 = text1.codePointAt(i);
|
|
||||||
const char2 = text2.codePointAt(i);
|
|
||||||
if (char1 !== char2) { break; }
|
|
||||||
const charLength = String.fromCodePoint(char1).length;
|
|
||||||
i += charLength;
|
|
||||||
if (i >= minLength) {
|
|
||||||
if (i > minLength) { i -= charLength; }
|
|
||||||
break;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
return i;
|
|
||||||
}
|
|
||||||
function distributeFuriganaInflected(term, reading, source) {
|
|
||||||
const termNormalized = convertKatakanaToHiragana(term);
|
|
||||||
const readingNormalized = convertKatakanaToHiragana(reading);
|
|
||||||
const sourceNormalized = convertKatakanaToHiragana(source);
|
|
||||||
let mainText = term;
|
|
||||||
let stemLength = getStemLength(termNormalized, sourceNormalized);
|
|
||||||
const readingStemLength = getStemLength(readingNormalized, sourceNormalized);
|
|
||||||
if (readingStemLength > 0 && readingStemLength >= stemLength) {
|
|
||||||
mainText = reading;
|
|
||||||
stemLength = readingStemLength;
|
|
||||||
reading = source.substring(0, stemLength) + reading.substring(stemLength);
|
|
||||||
}
|
|
||||||
const segments = [];
|
|
||||||
if (stemLength > 0) {
|
|
||||||
mainText = source.substring(0, stemLength) + mainText.substring(stemLength);
|
|
||||||
const segments2 = distributeFurigana(mainText, reading);
|
|
||||||
let consumed = 0;
|
|
||||||
for (const segment of segments2) {
|
|
||||||
const start = consumed;
|
|
||||||
consumed += segment.text.length;
|
|
||||||
if (consumed < stemLength) { segments.push(segment); }
|
|
||||||
else if (consumed === stemLength) { segments.push(segment); break; }
|
|
||||||
else {
|
|
||||||
if (start < stemLength) { segments.push(createFuriganaSegment(mainText.substring(start, stemLength), '')); }
|
|
||||||
break;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
if (stemLength < source.length) {
|
|
||||||
const remainder = source.substring(stemLength);
|
|
||||||
const last = segments[segments.length - 1];
|
|
||||||
if (last && last.reading.length === 0) { last.text += remainder; }
|
|
||||||
else { segments.push(createFuriganaSegment(remainder, '')); }
|
|
||||||
}
|
|
||||||
return segments;
|
|
||||||
}
|
|
||||||
function parsePositiveFrequencyNumber(value) {
|
|
||||||
if (typeof value === 'number' && Number.isFinite(value) && value > 0) {
|
|
||||||
return Math.max(1, Math.floor(value));
|
|
||||||
}
|
|
||||||
if (typeof value === 'string') {
|
|
||||||
const numericMatch = value.trim().match(/[+-]?(\d+(\.\d*)?|\.\d+)([eE][+-]?\d+)?/)?.[0];
|
|
||||||
if (!numericMatch) { return null; }
|
|
||||||
const parsed = Number.parseFloat(numericMatch);
|
|
||||||
if (!Number.isFinite(parsed) || parsed <= 0) { return null; }
|
|
||||||
return Math.max(1, Math.floor(parsed));
|
|
||||||
}
|
|
||||||
if (Array.isArray(value)) {
|
|
||||||
for (const item of value) {
|
|
||||||
const parsed = parsePositiveFrequencyNumber(item);
|
|
||||||
if (parsed !== null) { return parsed; }
|
|
||||||
}
|
|
||||||
}
|
|
||||||
return null;
|
|
||||||
}
|
|
||||||
function parseDisplayFrequencyNumber(value) {
|
|
||||||
if (typeof value === 'string') {
|
|
||||||
const leadingDigits = value.trim().match(/^\d+/)?.[0];
|
|
||||||
if (!leadingDigits) { return null; }
|
|
||||||
const parsed = Number.parseInt(leadingDigits, 10);
|
|
||||||
return Number.isFinite(parsed) && parsed > 0 ? parsed : null;
|
|
||||||
}
|
|
||||||
return parsePositiveFrequencyNumber(value);
|
|
||||||
}
|
|
||||||
function getFrequencyDictionaryName(frequency) {
|
|
||||||
const candidates = [
|
|
||||||
frequency?.dictionary,
|
|
||||||
frequency?.dictionaryName,
|
|
||||||
frequency?.name,
|
|
||||||
frequency?.title,
|
|
||||||
frequency?.dictionaryTitle,
|
|
||||||
frequency?.dictionaryAlias
|
|
||||||
];
|
|
||||||
for (const candidate of candidates) {
|
|
||||||
if (typeof candidate === 'string' && candidate.trim().length > 0) {
|
|
||||||
return candidate.trim();
|
|
||||||
}
|
|
||||||
}
|
|
||||||
return null;
|
|
||||||
}
|
|
||||||
function getBestFrequencyRank(dictionaryEntry, headwordIndex, dictionaryPriorityByName, dictionaryFrequencyModeByName) {
|
|
||||||
let best = null;
|
|
||||||
const headwordCount = Array.isArray(dictionaryEntry?.headwords) ? dictionaryEntry.headwords.length : 0;
|
|
||||||
for (const frequency of dictionaryEntry?.frequencies || []) {
|
|
||||||
if (!frequency || typeof frequency !== 'object') { continue; }
|
|
||||||
const frequencyHeadwordIndex = frequency.headwordIndex;
|
|
||||||
if (typeof frequencyHeadwordIndex === 'number') {
|
|
||||||
if (frequencyHeadwordIndex !== headwordIndex) { continue; }
|
|
||||||
} else if (headwordCount > 1) {
|
|
||||||
continue;
|
|
||||||
}
|
|
||||||
const dictionary = getFrequencyDictionaryName(frequency);
|
|
||||||
if (!dictionary) { continue; }
|
|
||||||
if (dictionaryFrequencyModeByName[dictionary] === 'occurrence-based') { continue; }
|
|
||||||
const rank =
|
|
||||||
parseDisplayFrequencyNumber(frequency.displayValue) ??
|
|
||||||
parsePositiveFrequencyNumber(frequency.frequency);
|
|
||||||
if (rank === null) { continue; }
|
|
||||||
const priorityRaw = dictionaryPriorityByName[dictionary];
|
|
||||||
const fallbackPriority =
|
|
||||||
typeof frequency.dictionaryIndex === 'number' && Number.isFinite(frequency.dictionaryIndex)
|
|
||||||
? Math.max(0, Math.floor(frequency.dictionaryIndex))
|
|
||||||
: Number.MAX_SAFE_INTEGER;
|
|
||||||
const priority =
|
|
||||||
typeof priorityRaw === 'number' && Number.isFinite(priorityRaw)
|
|
||||||
? Math.max(0, Math.floor(priorityRaw))
|
|
||||||
: fallbackPriority;
|
|
||||||
if (best === null || priority < best.priority || (priority === best.priority && rank < best.rank)) {
|
|
||||||
best = { priority, rank };
|
|
||||||
}
|
|
||||||
}
|
|
||||||
return best?.rank ?? null;
|
|
||||||
}
|
|
||||||
function hasExactSource(headword, token, requirePrimary) {
|
|
||||||
for (const src of headword.sources || []) {
|
|
||||||
if (src.originalText !== token) { continue; }
|
|
||||||
if (requirePrimary && !src.isPrimary) { continue; }
|
|
||||||
if (src.matchType !== 'exact') { continue; }
|
|
||||||
return true;
|
|
||||||
}
|
|
||||||
return false;
|
|
||||||
}
|
|
||||||
function collectExactHeadwordMatches(dictionaryEntries, token, requirePrimary) {
|
|
||||||
const matches = [];
|
|
||||||
for (const dictionaryEntry of dictionaryEntries || []) {
|
|
||||||
const headwords = Array.isArray(dictionaryEntry?.headwords) ? dictionaryEntry.headwords : [];
|
|
||||||
for (let headwordIndex = 0; headwordIndex < headwords.length; headwordIndex += 1) {
|
|
||||||
const headword = headwords[headwordIndex];
|
|
||||||
if (!hasExactSource(headword, token, requirePrimary)) { continue; }
|
|
||||||
matches.push({ dictionaryEntry, headword, headwordIndex });
|
|
||||||
}
|
|
||||||
}
|
|
||||||
return matches;
|
|
||||||
}
|
|
||||||
function sameHeadword(match, preferredMatch) {
|
|
||||||
if (!match || !preferredMatch) {
|
|
||||||
return false;
|
|
||||||
}
|
|
||||||
if (match.headword?.term !== preferredMatch.headword?.term) {
|
|
||||||
return false;
|
|
||||||
}
|
|
||||||
const matchReading = typeof match.headword?.reading === 'string' ? match.headword.reading : '';
|
|
||||||
const preferredReading =
|
|
||||||
typeof preferredMatch.headword?.reading === 'string' ? preferredMatch.headword.reading : '';
|
|
||||||
if (!matchReading || !preferredReading) {
|
|
||||||
return true;
|
|
||||||
}
|
|
||||||
return matchReading === preferredReading;
|
|
||||||
}
|
|
||||||
function getBestFrequencyRankForMatches(matches, dictionaryPriorityByName, dictionaryFrequencyModeByName) {
|
|
||||||
let best = null;
|
|
||||||
for (const match of matches) {
|
|
||||||
const rank = getBestFrequencyRank(
|
|
||||||
match.dictionaryEntry,
|
|
||||||
match.headwordIndex,
|
|
||||||
dictionaryPriorityByName,
|
|
||||||
dictionaryFrequencyModeByName
|
|
||||||
);
|
|
||||||
if (rank === null) { continue; }
|
|
||||||
if (best === null || rank < best) {
|
|
||||||
best = rank;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
return best;
|
|
||||||
}
|
|
||||||
function normalizeWordClasses(headword) {
|
|
||||||
if (!Array.isArray(headword?.wordClasses)) { return undefined; }
|
|
||||||
const classes = headword.wordClasses.filter((wordClass) => typeof wordClass === "string" && wordClass.trim().length > 0);
|
|
||||||
return classes.length > 0 ? classes : undefined;
|
|
||||||
}
|
|
||||||
function appendDictionaryNames(target, value) {
|
|
||||||
if (!value || typeof value !== 'object') {
|
|
||||||
return;
|
|
||||||
}
|
|
||||||
const candidates = [
|
|
||||||
value.dictionary,
|
|
||||||
value.dictionaryName,
|
|
||||||
value.name,
|
|
||||||
value.title,
|
|
||||||
value.dictionaryTitle,
|
|
||||||
value.dictionaryAlias
|
|
||||||
];
|
|
||||||
for (const candidate of candidates) {
|
|
||||||
if (typeof candidate === 'string' && candidate.trim().length > 0) {
|
|
||||||
target.push(candidate.trim());
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
function getDictionaryEntryNames(entry) {
|
|
||||||
const names = [];
|
|
||||||
appendDictionaryNames(names, entry);
|
|
||||||
for (const definition of entry?.definitions || []) {
|
|
||||||
appendDictionaryNames(names, definition);
|
|
||||||
}
|
|
||||||
for (const frequency of entry?.frequencies || []) {
|
|
||||||
appendDictionaryNames(names, frequency);
|
|
||||||
}
|
|
||||||
for (const pronunciation of entry?.pronunciations || []) {
|
|
||||||
appendDictionaryNames(names, pronunciation);
|
|
||||||
}
|
|
||||||
return names;
|
|
||||||
}
|
|
||||||
function isNameDictionaryEntry(entry) {
|
|
||||||
if (!includeNameMatchMetadata || !entry || typeof entry !== 'object') {
|
|
||||||
return false;
|
|
||||||
}
|
|
||||||
return getDictionaryEntryNames(entry).some((name) => name.startsWith(${JSON.stringify(CHARACTER_DICTIONARY_TITLE_PREFIX)}));
|
|
||||||
}
|
|
||||||
function parseSubMinerMediaIdFromString(value) {
|
|
||||||
const imageMatch = value.match(/\bimg\/m(\d+)-/i);
|
|
||||||
if (imageMatch) {
|
|
||||||
const parsed = Number.parseInt(imageMatch[1], 10);
|
|
||||||
if (Number.isSafeInteger(parsed) && parsed > 0) { return parsed; }
|
|
||||||
}
|
|
||||||
const titleMatch = value.match(/${CHARACTER_DICTIONARY_TITLE_PREFIX}[^\d]*(?:AniList\s*)?(\d+)/i);
|
|
||||||
if (titleMatch) {
|
|
||||||
const parsed = Number.parseInt(titleMatch[1], 10);
|
|
||||||
if (Number.isSafeInteger(parsed) && parsed > 0) { return parsed; }
|
|
||||||
}
|
|
||||||
return null;
|
|
||||||
}
|
|
||||||
function parseSubMinerMediaIdCandidate(value) {
|
|
||||||
if (typeof value === 'number' && Number.isSafeInteger(value) && value > 0) {
|
|
||||||
return value;
|
|
||||||
}
|
|
||||||
if (typeof value === 'string' && /^\d+$/.test(value.trim())) {
|
|
||||||
const parsed = Number.parseInt(value.trim(), 10);
|
|
||||||
if (Number.isSafeInteger(parsed) && parsed > 0) { return parsed; }
|
|
||||||
}
|
|
||||||
return null;
|
|
||||||
}
|
|
||||||
function collectSubMinerMediaIds(value, target) {
|
|
||||||
if (typeof value === 'string') {
|
|
||||||
const parsed = parseSubMinerMediaIdFromString(value);
|
|
||||||
if (parsed !== null) { target.add(parsed); }
|
|
||||||
return;
|
|
||||||
}
|
|
||||||
if (!value || typeof value !== 'object') {
|
|
||||||
return;
|
|
||||||
}
|
|
||||||
if (Array.isArray(value)) {
|
|
||||||
for (const item of value) { collectSubMinerMediaIds(item, target); }
|
|
||||||
return;
|
|
||||||
}
|
|
||||||
const mediaIdCandidates = [
|
|
||||||
value.subminerMediaId,
|
|
||||||
value.subMinerMediaId,
|
|
||||||
value.characterDictionaryMediaId,
|
|
||||||
value.data?.subminerMediaId,
|
|
||||||
value.data?.subMinerMediaId,
|
|
||||||
value.data?.characterDictionaryMediaId
|
|
||||||
];
|
|
||||||
for (const candidate of mediaIdCandidates) {
|
|
||||||
const parsed = parseSubMinerMediaIdCandidate(candidate);
|
|
||||||
if (parsed !== null) { target.add(parsed); }
|
|
||||||
}
|
|
||||||
for (const child of Object.values(value)) {
|
|
||||||
collectSubMinerMediaIds(child, target);
|
|
||||||
}
|
|
||||||
}
|
|
||||||
function getSubMinerMediaIds(entry) {
|
|
||||||
const mediaIds = new Set();
|
|
||||||
collectSubMinerMediaIds(entry, mediaIds);
|
|
||||||
return mediaIds;
|
|
||||||
}
|
|
||||||
function isCurrentMediaNameDictionaryEntry(entry) {
|
|
||||||
if (!isNameDictionaryEntry(entry)) {
|
|
||||||
return false;
|
|
||||||
}
|
|
||||||
if (currentCharacterDictionaryMediaId === null) {
|
|
||||||
return true;
|
|
||||||
}
|
|
||||||
const mediaIds = getSubMinerMediaIds(entry);
|
|
||||||
return mediaIds.size === 0 || mediaIds.has(currentCharacterDictionaryMediaId);
|
|
||||||
}
|
|
||||||
function findLongestNameMatch(dictionaryEntries, textWindow) {
|
|
||||||
let best = null;
|
|
||||||
for (const dictionaryEntry of dictionaryEntries || []) {
|
|
||||||
if (!isCurrentMediaNameDictionaryEntry(dictionaryEntry)) { continue; }
|
|
||||||
const headwords = Array.isArray(dictionaryEntry?.headwords) ? dictionaryEntry.headwords : [];
|
|
||||||
for (let headwordIndex = 0; headwordIndex < headwords.length; headwordIndex += 1) {
|
|
||||||
const headword = headwords[headwordIndex];
|
|
||||||
for (const src of headword?.sources || []) {
|
|
||||||
if (src.matchType !== 'exact' || src.isPrimary !== true) { continue; }
|
|
||||||
const originalText = typeof src.originalText === 'string' ? src.originalText : '';
|
|
||||||
if (!originalText || !textWindow.startsWith(originalText)) { continue; }
|
|
||||||
if (best === null || originalText.length > best.sourceLength) {
|
|
||||||
best = { dictionaryEntry, headword, headwordIndex, sourceLength: originalText.length };
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
return best;
|
|
||||||
}
|
|
||||||
function findLongestGenericMatchLength(dictionaryEntries, textWindow) {
|
|
||||||
let best = 0;
|
|
||||||
for (const dictionaryEntry of dictionaryEntries || []) {
|
|
||||||
if (isNameDictionaryEntry(dictionaryEntry)) { continue; }
|
|
||||||
const headwords = Array.isArray(dictionaryEntry?.headwords) ? dictionaryEntry.headwords : [];
|
|
||||||
for (const headword of headwords) {
|
|
||||||
for (const src of headword?.sources || []) {
|
|
||||||
if (src.matchType !== 'exact' || src.isPrimary !== true) { continue; }
|
|
||||||
const originalText = typeof src.originalText === 'string' ? src.originalText : '';
|
|
||||||
if (!originalText || !textWindow.startsWith(originalText)) { continue; }
|
|
||||||
if (originalText.length > best) { best = originalText.length; }
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
return best;
|
|
||||||
}
|
|
||||||
function getPreferredHeadword(dictionaryEntries, token, dictionaryPriorityByName, dictionaryFrequencyModeByName) {
|
|
||||||
const currentMediaDictionaryEntries =
|
|
||||||
currentCharacterDictionaryMediaId === null
|
|
||||||
? (dictionaryEntries || [])
|
|
||||||
: (dictionaryEntries || []).filter((entry) => {
|
|
||||||
if (!isNameDictionaryEntry(entry)) { return true; }
|
|
||||||
return isCurrentMediaNameDictionaryEntry(entry);
|
|
||||||
});
|
|
||||||
const exactPrimaryMatches = collectExactHeadwordMatches(currentMediaDictionaryEntries, token, true);
|
|
||||||
let matchedNameDictionary = false;
|
|
||||||
if (includeNameMatchMetadata) {
|
|
||||||
for (const dictionaryEntry of currentMediaDictionaryEntries || []) {
|
|
||||||
if (!isCurrentMediaNameDictionaryEntry(dictionaryEntry)) { continue; }
|
|
||||||
for (const match of exactPrimaryMatches) {
|
|
||||||
if (match.dictionaryEntry !== dictionaryEntry) { continue; }
|
|
||||||
matchedNameDictionary = true;
|
|
||||||
break;
|
|
||||||
}
|
|
||||||
if (matchedNameDictionary) { break; }
|
|
||||||
}
|
|
||||||
}
|
|
||||||
const preferredMatch = exactPrimaryMatches[0];
|
|
||||||
if (preferredMatch) {
|
|
||||||
const exactFrequencyMatches = collectExactHeadwordMatches(currentMediaDictionaryEntries, token, false)
|
|
||||||
.filter((match) => sameHeadword(match, preferredMatch));
|
|
||||||
return {
|
|
||||||
term: preferredMatch.headword.term,
|
|
||||||
reading: preferredMatch.headword.reading,
|
|
||||||
wordClasses: normalizeWordClasses(preferredMatch.headword),
|
|
||||||
isNameMatch:
|
|
||||||
matchedNameDictionary || isCurrentMediaNameDictionaryEntry(preferredMatch.dictionaryEntry),
|
|
||||||
frequencyRank: getBestFrequencyRankForMatches(
|
|
||||||
exactFrequencyMatches.length > 0 ? exactFrequencyMatches : exactPrimaryMatches,
|
|
||||||
dictionaryPriorityByName,
|
|
||||||
dictionaryFrequencyModeByName
|
|
||||||
)
|
|
||||||
};
|
|
||||||
}
|
|
||||||
return null;
|
|
||||||
}
|
|
||||||
`;
|
|
||||||
|
|
||||||
// Bump whenever the install script below changes so already-loaded parser
|
|
||||||
// windows re-install the new scan runtime instead of running the stale one.
|
|
||||||
const YOMITAN_SCAN_RUNTIME_VERSION = 1;
|
|
||||||
const YOMITAN_SCAN_RUNTIME_MISSING_SENTINEL = '__subminer-yomitan-scan-runtime-missing__';
|
|
||||||
|
|
||||||
interface YomitanScanRequestParams {
|
|
||||||
text: string;
|
|
||||||
profileIndex: number;
|
|
||||||
scanLength: number;
|
|
||||||
includeNameMatchMetadata: boolean;
|
|
||||||
greedyNameScanEnabled: boolean;
|
|
||||||
currentCharacterDictionaryMediaId: number | null;
|
|
||||||
dictionaryPriorityByName: Record<string, number>;
|
|
||||||
dictionaryFrequencyModeByName: Partial<Record<string, YomitanFrequencyMode>>;
|
|
||||||
cacheEpoch: number;
|
|
||||||
}
|
|
||||||
|
|
||||||
// Installed once per parser window (and re-installed after in-page reloads):
|
|
||||||
// keeps V8 from re-parsing the helper bundle on every subtitle line, and hosts
|
|
||||||
// the cross-line termsFind cache. Each subtitle line then only evaluates a tiny
|
|
||||||
// call into globalThis.__subminerYomitanScan.
|
|
||||||
const YOMITAN_SCAN_RUNTIME_INSTALL_SCRIPT = String.raw`
|
|
||||||
(() => {
|
|
||||||
if (globalThis.__subminerYomitanScanVersion === ${YOMITAN_SCAN_RUNTIME_VERSION}) {
|
|
||||||
return true;
|
|
||||||
}
|
|
||||||
const invoke = (action, params) =>
|
|
||||||
new Promise((resolve, reject) => {
|
|
||||||
chrome.runtime.sendMessage({ action, params }, (response) => {
|
|
||||||
if (chrome.runtime.lastError) {
|
|
||||||
reject(new Error(chrome.runtime.lastError.message));
|
|
||||||
return;
|
|
||||||
}
|
|
||||||
if (!response || typeof response !== "object") {
|
|
||||||
reject(new Error("Invalid response from Yomitan backend"));
|
|
||||||
return;
|
|
||||||
}
|
|
||||||
if (response.error) {
|
|
||||||
reject(new Error(response.error.message || "Yomitan backend error"));
|
|
||||||
return;
|
|
||||||
}
|
|
||||||
resolve(response.result);
|
|
||||||
});
|
|
||||||
});
|
|
||||||
// Cross-line termsFind LRU keyed by profile + substring: subtitle lines
|
|
||||||
// repeat particles and inflections constantly, so most lookups hit here.
|
|
||||||
// Entries hold in-flight promises so concurrent identical lookups dedupe.
|
|
||||||
const termsFindCache = new Map();
|
|
||||||
const TERMS_FIND_CACHE_LIMIT = 2000;
|
|
||||||
let termsFindCacheEpoch = -1;
|
|
||||||
const MAX_SHRINKING_WINDOW_RETRY_LOOKUPS = 4;
|
|
||||||
globalThis.__subminerYomitanScanVersion = ${YOMITAN_SCAN_RUNTIME_VERSION};
|
|
||||||
globalThis.__subminerYomitanScan = async (scanParams) => {
|
|
||||||
const {
|
|
||||||
text,
|
|
||||||
profileIndex,
|
|
||||||
scanLength,
|
|
||||||
includeNameMatchMetadata,
|
|
||||||
greedyNameScanEnabled,
|
|
||||||
currentCharacterDictionaryMediaId,
|
|
||||||
dictionaryPriorityByName,
|
|
||||||
dictionaryFrequencyModeByName,
|
|
||||||
cacheEpoch
|
|
||||||
} = scanParams;
|
|
||||||
if (cacheEpoch !== termsFindCacheEpoch) {
|
|
||||||
termsFindCache.clear();
|
|
||||||
termsFindCacheEpoch = cacheEpoch;
|
|
||||||
}
|
|
||||||
${YOMITAN_SCANNING_HELPERS}
|
|
||||||
const CAPTION_OPENING_BRACKETS = new Set(["(", "(", "[", "[", "{", "{", "「", "『", "【", "〈", "《", "≪", "<", "<"]);
|
|
||||||
function shouldEmitUnparsedRunAsToken(runText) {
|
|
||||||
if (!/[\p{L}\p{N}]/u.test(runText)) { return false; }
|
|
||||||
const firstChar = Array.from(runText.trim())[0];
|
|
||||||
return firstChar !== undefined && !CAPTION_OPENING_BRACKETS.has(firstChar);
|
|
||||||
}
|
|
||||||
function isLookupWorthyCodePoint(codePoint) {
|
|
||||||
if (isCodePointJapanese(codePoint)) { return true; }
|
|
||||||
return /[\p{L}\p{N}]/u.test(String.fromCodePoint(codePoint));
|
|
||||||
}
|
|
||||||
function isKanaOnlyRunText(runText) {
|
|
||||||
const chars = Array.from(runText);
|
|
||||||
return chars.length > 0 && chars.every((char) => isCodePointKana(char.codePointAt(0)));
|
|
||||||
}
|
|
||||||
const details = {matchType: "exact", deinflect: true};
|
|
||||||
const tokens = [];
|
|
||||||
async function termsFindAt(position, windowLength) {
|
|
||||||
const substring = text.substring(position, position + windowLength);
|
|
||||||
const cacheKey = profileIndex + " | |||||||