perf(tokenizer): single-pass Yomitan scan with cross-line caching and prefetch fixes (#185)

This commit is contained in:
2026-08-06 21:44:09 -07:00
committed by GitHub
parent 441ecf3c04
commit dbdf578c68
38 changed files with 4073 additions and 1720 deletions
@@ -8,6 +8,7 @@ import {
isKanaChar,
isKanaOnlyText,
isTokenPos2Excluded,
normalizeKana,
} from './token-classification';
const POS1_EXCLUSIONS = new Set(['助詞']);
@@ -29,6 +30,26 @@ function makeNoun(surface: string): MergedToken {
};
}
test('kana normalization folds halfwidth kana, composing the voiced pairs', () => {
// カ + ゙ is two code points for one character: without composing them, a
// halfwidth word counts as longer than the reading that spells it, which
// disqualifies the reading from known-word matching.
assert.equal(normalizeKana('ガク'), normalizeKana('ガク'));
assert.equal(normalizeKana('パン'), normalizeKana('パン'));
assert.equal(normalizeKana('ミナト'), 'みなと');
assert.ok(isKanaOnlyText('ガク'));
});
test('kana normalization leaves characters other than halfwidth kana alone', () => {
// The composition is scoped to the halfwidth runs: applied to the whole
// string, NFKC would also rewrite these into something the dictionary, the
// known-word list, and the frequency data were never keyed on.
assert.equal(normalizeKana('①ガ'), '①が');
assert.equal(normalizeKana('Aガ'), 'Aが');
assert.equal(normalizeKana('㍑ガ'), '㍑が');
assert.equal(normalizeKana('fiガ'), 'fiが');
});
test('kana classification excludes the katakana-hiragana double hyphen', () => {
assert.equal(isKanaChar(''), false);
assert.equal(isKanaOnlyText(''), false);