mirror of
https://github.com/ksyasuda/SubMiner.git
synced 2026-08-05 19:21:35 -07:00
afa66ee508
- Cap only blind shrinking-window retries (not backend-guided shrinks); a line that exhausts the cap escalates to one parseText fallback instead of dropping to raw text - Bound the cross-line termsFind cache by retained dictionary-entry weight, not just key count, and re-check it when a lookup resolves - Keep halfwidth katakana character names in the greedy pre-pass, and let a generic word beat a name it fully contains - Share one Han code-point table between the character dictionary and the scanner's name pre-pass; narrow the mob-disambiguator filter to the split letters, not every one-character term - Release the subtitle prefetch pause on a new onProcessingSettled signal instead of the tokenized emit, so duplicate/suppressed/failed lines no longer pause prefetch indefinitely - Split the Yomitan scan runtime's injected helper script into its own file
32 lines
1.5 KiB
TypeScript
32 lines
1.5 KiB
TypeScript
// Single source of truth for "this code point is a Han character", shared by
|
|
// the main-process character dictionary and the in-page Yomitan scan runtime.
|
|
// The two used to carry separate range lists, and they drifted: a name written
|
|
// with a supplementary-plane kanji could enter the generated dictionary while
|
|
// the scanner's greedy name pre-pass refused to probe the position.
|
|
//
|
|
// Ranges rather than \p{Script=Han}: the scan walk tests one code point per
|
|
// character of every subtitle line, where an integer compare beats building a
|
|
// string for a regex, and the script is injected as text into a page where a
|
|
// shared helper cannot be imported.
|
|
export const HAN_CODE_POINT_RANGES: ReadonlyArray<readonly [number, number]> = [
|
|
[0x3400, 0x4dbf], // Extension A
|
|
[0x4e00, 0x9fff], // CJK Unified Ideographs
|
|
[0xf900, 0xfaff], // Compatibility Ideographs
|
|
[0x20000, 0x2a6df], // Extension B
|
|
[0x2a700, 0x2ebef], // Extensions C-F
|
|
[0x2ebf0, 0x2ee5f], // Extension I
|
|
[0x2f800, 0x2fa1f], // Compatibility Ideographs Supplement
|
|
[0x30000, 0x3134f], // Extension G
|
|
[0x31350, 0x323af], // Extension H
|
|
[0x323b0, 0x33479], // Extension J (Unicode 17)
|
|
];
|
|
|
|
export function isHanCodePoint(codePoint: number): boolean {
|
|
return HAN_CODE_POINT_RANGES.some(([start, end]) => codePoint >= start && codePoint <= end);
|
|
}
|
|
|
|
/** The same ranges as a regular expression character class body (needs the `u` flag). */
|
|
export const HAN_REGEXP_CLASS_BODY = HAN_CODE_POINT_RANGES.map(
|
|
([start, end]) => `\\u{${start.toString(16)}}-\\u{${end.toString(16)}}`,
|
|
).join('');
|