diff --git a/__test__/main/core/entries.test.ts b/__test__/main/core/entries.test.ts index da1b563..68fc52f 100644 --- a/__test__/main/core/entries.test.ts +++ b/__test__/main/core/entries.test.ts @@ -472,4 +472,17 @@ test("Test Japanese words", () => { expect(createLookupWordsJa("走った")).toEqual(expect.arrayContaining(["走る"])); expect(createLookupWordsJa("おいた")).toEqual(expect.arrayContaining(["おく", "おいる"])); expect(createLookupWordsJa("19az")).toEqual(expect.arrayContaining(["19az"])); + // Halfwidth katakana must also reach fullwidth headwords (manga/UI text), + // keeping the usual prefix decomposition on both spellings. + expect(createLookupWordsJa("テレビ")).toEqual(expect.arrayContaining(["テレビ", "テレビ", "テレ", "テ", "テレヒ", "テレ", "テ"])); + expect(createLookupWordsJa("ガス")).toEqual(expect.arrayContaining(["ガス", "ガス", "ガ", "カ", "ガ"])); + // Conversion is scoped to halfwidth katakana: other characters must not get + // compatibility-folded (NFKC over the whole string would turn ① into "1"). + const mixed = createLookupWordsJa("①テレビ"); + expect(mixed).toEqual(expect.arrayContaining(["①テレビ", "①テレビ", "①テレ", "①テレ"])); + expect(mixed).not.toContain("1テレビ"); + // Fullwidth katakana runs keep cutting into all prefixes (show-all-candidates design) + expect(createLookupWordsJa("ソフトウェアエンジニア")).toEqual( + expect.arrayContaining(["ソフトウェアエンジニア", "ソフトウェア", "ソフト", "ソ"]), + ); }); diff --git a/src/main/core/entry/ja.ts b/src/main/core/entry/ja.ts index 1545e17..10c7b26 100644 --- a/src/main/core/entry/ja.ts +++ b/src/main/core/entry/ja.ts @@ -10,6 +10,17 @@ import rule from "../rule"; const RE_ALPHABETS_NUMBERS = /[A-Za-z0-9]/g; const FULLWIDTH_OFFSET = 0xfee0; +// Halfwidth katakana (U+FF61-FF9F, common in manga/UI text) is converted to +// its fullwidth form so テレビ looks up the same headwords as テレビ. +// Only maximal halfwidth-katakana runs are normalized: NFKC over the whole +// string would fold unrelated characters (e.g. ① -> 1). Normalizing each run +// lets NFKC compose voiced marks with the preceding letter (カ + ゙ -> ガ) +// without touching anything outside the run. +const RE_HALFWIDTH_KATAKANA_RUN = /[\uFF61-\uFF9F]+/g; + +const convertHalfwidthKatakana = (s: string): string => + s.replace(RE_HALFWIDTH_KATAKANA_RUN, (run) => run.normalize("NFKC")); + const createLookupWordsJa = (sourceStr: string): string[] => { const str = sourceStr .substring(0, 40) @@ -20,13 +31,25 @@ const createLookupWordsJa = (sourceStr: string): string[] => { result.push(sourceStr); // Add the original word - for (let i = str.length; i >= 1; i--) { - const part = str.substring(0, i); - result.push(part); + // For halfwidth input, keep both chains: every candidate retains its usual + // prefix decomposition (show-all-plausible-candidates design), e.g. + // テレヒ -> テレヒ/テレ/テ plus テレビ/テレ/テ. Without halfwidth katakana the + // replace is a no-op, so no second chain is added. + const chains = [str]; + const converted = convertHalfwidthKatakana(str); + if (converted !== str) { + chains.push(converted); + } + + for (const chain of chains) { + for (let i = chain.length; i >= 1; i--) { + const part = chain.substring(0, i); + result.push(part); - if (i >= 2) { - const deinedWords = rule.doJa(part); - result.merge(deinedWords ?? []); + if (i >= 2) { + const deinedWords = rule.doJa(part); + result.merge(deinedWords ?? []); + } } } return result.toArray();