mirror of
https://github.com/ksyasuda/SubMiner.git
synced 2026-08-14 01:55:58 -07:00
perf(tokenizer): single-pass Yomitan scan with cross-line caching and prefetch fixes (#185)
This commit is contained in:
@@ -8,6 +8,7 @@ import {
|
||||
isKanaChar,
|
||||
isKanaOnlyText,
|
||||
isTokenPos2Excluded,
|
||||
normalizeKana,
|
||||
} from './token-classification';
|
||||
|
||||
const POS1_EXCLUSIONS = new Set(['助詞']);
|
||||
@@ -29,6 +30,26 @@ function makeNoun(surface: string): MergedToken {
|
||||
};
|
||||
}
|
||||
|
||||
test('kana normalization folds halfwidth kana, composing the voiced pairs', () => {
|
||||
// カ + ゙ is two code points for one character: without composing them, a
|
||||
// halfwidth word counts as longer than the reading that spells it, which
|
||||
// disqualifies the reading from known-word matching.
|
||||
assert.equal(normalizeKana('ガク'), normalizeKana('ガク'));
|
||||
assert.equal(normalizeKana('パン'), normalizeKana('パン'));
|
||||
assert.equal(normalizeKana('ミナト'), 'みなと');
|
||||
assert.ok(isKanaOnlyText('ガク'));
|
||||
});
|
||||
|
||||
test('kana normalization leaves characters other than halfwidth kana alone', () => {
|
||||
// The composition is scoped to the halfwidth runs: applied to the whole
|
||||
// string, NFKC would also rewrite these into something the dictionary, the
|
||||
// known-word list, and the frequency data were never keyed on.
|
||||
assert.equal(normalizeKana('①ガ'), '①が');
|
||||
assert.equal(normalizeKana('Aガ'), 'Aが');
|
||||
assert.equal(normalizeKana('㍑ガ'), '㍑が');
|
||||
assert.equal(normalizeKana('fiガ'), 'fiが');
|
||||
});
|
||||
|
||||
test('kana classification excludes the katakana-hiragana double hyphen', () => {
|
||||
assert.equal(isKanaChar('゠'), false);
|
||||
assert.equal(isKanaOnlyText('゠'), false);
|
||||
|
||||
Reference in New Issue
Block a user