From 8e9eddd7117dae26392ac567610823710288625b Mon Sep 17 00:00:00 2001 From: Abdulaziz Al Malki Date: Sat, 3 Oct 2026 16:57:23 +0300 Subject: [PATCH 1/3] Fix katakana lookup: normalize halfwidth forms to fullwidth MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Halfwidth katakana (U+FF61-FF9F, common in manga/UI text) never matched fullwidth headwords: hovering テレビ found nothing for テレビ. When the hovered text contains halfwidth katakana, the NFKC-normalized form is added as an additional candidate chain, so every prefix decomposition is generated for both spellings (テレヒ -> テレ/テ and テレ/テ). NFKC also folds voiced/semi-voiced combining sequences (カ + ゙ -> ガ). Prefix generation for fullwidth katakana runs is unchanged: all candidates are kept, per the show-all-plausible-candidates design. --- __test__/main/core/entries.test.ts | 8 ++++++++ src/main/core/entry/ja.ts | 30 ++++++++++++++++++++++++------ 2 files changed, 32 insertions(+), 6 deletions(-) diff --git a/__test__/main/core/entries.test.ts b/__test__/main/core/entries.test.ts index da1b563f..0812b3b2 100644 --- a/__test__/main/core/entries.test.ts +++ b/__test__/main/core/entries.test.ts @@ -472,4 +472,12 @@ test("Test Japanese words", () => { expect(createLookupWordsJa("走った")).toEqual(expect.arrayContaining(["走る"])); expect(createLookupWordsJa("おいた")).toEqual(expect.arrayContaining(["おく", "おいる"])); expect(createLookupWordsJa("19az")).toEqual(expect.arrayContaining(["19az"])); + // Halfwidth katakana must also reach fullwidth headwords (manga/UI text), + // keeping the usual prefix decomposition on both spellings. + expect(createLookupWordsJa("テレビ")).toEqual(expect.arrayContaining(["テレビ", "テレビ", "テレ", "テ", "テレヒ", "テレ", "テ"])); + expect(createLookupWordsJa("ガス")).toEqual(expect.arrayContaining(["ガス", "ガス", "ガ", "カ", "ガ"])); + // Fullwidth katakana runs keep cutting into all prefixes (show-all-candidates design) + expect(createLookupWordsJa("ソフトウェアエンジニア")).toEqual( + expect.arrayContaining(["ソフトウェアエンジニア", "ソフトウェア", "ソフト", "ソ"]), + ); }); diff --git a/src/main/core/entry/ja.ts b/src/main/core/entry/ja.ts index 1545e17d..95c80ee8 100644 --- a/src/main/core/entry/ja.ts +++ b/src/main/core/entry/ja.ts @@ -10,6 +10,11 @@ import rule from "../rule"; const RE_ALPHABETS_NUMBERS = /[A-Za-z0-9]/g; const FULLWIDTH_OFFSET = 0xfee0; +// Halfwidth katakana (U+FF61-FF9F, common in manga/UI text) is normalized to +// its fullwidth form so テレビ looks up the same headwords as テレビ. +// NFKC also folds the voiced/semi-voiced combining sequences (カ + ゙ -> ガ). +const RE_HALFWIDTH_KATAKANA = /[\uFF61-\uFF9F]/; + const createLookupWordsJa = (sourceStr: string): string[] => { const str = sourceStr .substring(0, 40) @@ -20,13 +25,26 @@ const createLookupWordsJa = (sourceStr: string): string[] => { result.push(sourceStr); // Add the original word - for (let i = str.length; i >= 1; i--) { - const part = str.substring(0, i); - result.push(part); + // For halfwidth input, keep both chains: every candidate retains its usual + // prefix decomposition (show-all-plausible-candidates design), e.g. + // テレヒ -> テレヒ/テレ/テ plus テレビ/テレ/テ. + const chains = [str]; + if (RE_HALFWIDTH_KATAKANA.test(str)) { + const normalized = str.normalize("NFKC"); + if (normalized !== str) { + chains.push(normalized); + } + } + + for (const chain of chains) { + for (let i = chain.length; i >= 1; i--) { + const part = chain.substring(0, i); + result.push(part); - if (i >= 2) { - const deinedWords = rule.doJa(part); - result.merge(deinedWords ?? []); + if (i >= 2) { + const deinedWords = rule.doJa(part); + result.merge(deinedWords ?? []); + } } } return result.toArray(); From 15a0ddc5130c442fd22222923ee698e7699b545b Mon Sep 17 00:00:00 2001 From: Abdulaziz Al Malki Date: Sat, 3 Oct 2026 18:13:13 +0300 Subject: [PATCH 2/3] Scope the conversion to halfwidth katakana characters only MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Whole-string NFKC folds unrelated characters (circled digits -> ASCII, ligatures, fullwidth latin). Convert each halfwidth-katakana character (U+FF61-FF9F) individually, then a canonical-composition NFC pass to combine the converted voiced marks with the preceding letter (カ + ゙ -> ガ). NFC performs no compatibility folds, so nothing outside the halfwidth katakana range can change. --- __test__/main/core/entries.test.ts | 5 +++++ src/main/core/entry/ja.ts | 18 +++++++++++++----- 2 files changed, 18 insertions(+), 5 deletions(-) diff --git a/__test__/main/core/entries.test.ts b/__test__/main/core/entries.test.ts index 0812b3b2..68fc52ff 100644 --- a/__test__/main/core/entries.test.ts +++ b/__test__/main/core/entries.test.ts @@ -476,6 +476,11 @@ test("Test Japanese words", () => { // keeping the usual prefix decomposition on both spellings. expect(createLookupWordsJa("テレビ")).toEqual(expect.arrayContaining(["テレビ", "テレビ", "テレ", "テ", "テレヒ", "テレ", "テ"])); expect(createLookupWordsJa("ガス")).toEqual(expect.arrayContaining(["ガス", "ガス", "ガ", "カ", "ガ"])); + // Conversion is scoped to halfwidth katakana: other characters must not get + // compatibility-folded (NFKC over the whole string would turn ① into "1"). + const mixed = createLookupWordsJa("①テレビ"); + expect(mixed).toEqual(expect.arrayContaining(["①テレビ", "①テレビ", "①テレ", "①テレ"])); + expect(mixed).not.toContain("1テレビ"); // Fullwidth katakana runs keep cutting into all prefixes (show-all-candidates design) expect(createLookupWordsJa("ソフトウェアエンジニア")).toEqual( expect.arrayContaining(["ソフトウェアエンジニア", "ソフトウェア", "ソフト", "ソ"]), diff --git a/src/main/core/entry/ja.ts b/src/main/core/entry/ja.ts index 95c80ee8..fb52c394 100644 --- a/src/main/core/entry/ja.ts +++ b/src/main/core/entry/ja.ts @@ -10,10 +10,18 @@ import rule from "../rule"; const RE_ALPHABETS_NUMBERS = /[A-Za-z0-9]/g; const FULLWIDTH_OFFSET = 0xfee0; -// Halfwidth katakana (U+FF61-FF9F, common in manga/UI text) is normalized to +// Halfwidth katakana (U+FF61-FF9F, common in manga/UI text) is converted to // its fullwidth form so テレビ looks up the same headwords as テレビ. -// NFKC also folds the voiced/semi-voiced combining sequences (カ + ゙ -> ガ). +// The conversion is applied per character inside the halfwidth-katakana range +// only: NFKC over the whole string would fold unrelated characters too +// (circled digits, fullwidth latin, ligatures...). The trailing NFC pass is a +// canonical-composition pass, needed to combine the converted voiced marks +// with the preceding letter (カ + ゙ -> ガ); it performs no compatibility folds. const RE_HALFWIDTH_KATAKANA = /[\uFF61-\uFF9F]/; +const RE_HALFWIDTH_KATAKANA_G = /[\uFF61-\uFF9F]/g; + +const convertHalfwidthKatakana = (s: string): string => + s.replace(RE_HALFWIDTH_KATAKANA_G, (c) => c.normalize("NFKC")).normalize("NFC"); const createLookupWordsJa = (sourceStr: string): string[] => { const str = sourceStr @@ -30,9 +38,9 @@ const createLookupWordsJa = (sourceStr: string): string[] => { // テレヒ -> テレヒ/テレ/テ plus テレビ/テレ/テ. const chains = [str]; if (RE_HALFWIDTH_KATAKANA.test(str)) { - const normalized = str.normalize("NFKC"); - if (normalized !== str) { - chains.push(normalized); + const converted = convertHalfwidthKatakana(str); + if (converted !== str) { + chains.push(converted); } } From 4fe574f85c916bd45da3c48568a83d94b7fc00a0 Mon Sep 17 00:00:00 2001 From: Abdulaziz Al Malki Date: Sat, 3 Oct 2026 19:52:22 +0300 Subject: [PATCH 3/3] Convert halfwidth katakana per run instead of per character MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Follow-up on review: one regex (runs, U+FF61-FF9F+), one NFKC call per maximal run. Running NFKC per character needed a separate NFC pass to compose voiced marks; NFKC over the whole run composes them directly (カ + ゙ -> ガ). No test()/replace() duplication: a string without halfwidth katakana comes back unchanged, so the chain guard is the already-present converted !== str check. --- src/main/core/entry/ja.ts | 25 +++++++++++-------------- 1 file changed, 11 insertions(+), 14 deletions(-) diff --git a/src/main/core/entry/ja.ts b/src/main/core/entry/ja.ts index fb52c394..10c7b267 100644 --- a/src/main/core/entry/ja.ts +++ b/src/main/core/entry/ja.ts @@ -12,16 +12,14 @@ const FULLWIDTH_OFFSET = 0xfee0; // Halfwidth katakana (U+FF61-FF9F, common in manga/UI text) is converted to // its fullwidth form so テレビ looks up the same headwords as テレビ. -// The conversion is applied per character inside the halfwidth-katakana range -// only: NFKC over the whole string would fold unrelated characters too -// (circled digits, fullwidth latin, ligatures...). The trailing NFC pass is a -// canonical-composition pass, needed to combine the converted voiced marks -// with the preceding letter (カ + ゙ -> ガ); it performs no compatibility folds. -const RE_HALFWIDTH_KATAKANA = /[\uFF61-\uFF9F]/; -const RE_HALFWIDTH_KATAKANA_G = /[\uFF61-\uFF9F]/g; +// Only maximal halfwidth-katakana runs are normalized: NFKC over the whole +// string would fold unrelated characters (e.g. ① -> 1). Normalizing each run +// lets NFKC compose voiced marks with the preceding letter (カ + ゙ -> ガ) +// without touching anything outside the run. +const RE_HALFWIDTH_KATAKANA_RUN = /[\uFF61-\uFF9F]+/g; const convertHalfwidthKatakana = (s: string): string => - s.replace(RE_HALFWIDTH_KATAKANA_G, (c) => c.normalize("NFKC")).normalize("NFC"); + s.replace(RE_HALFWIDTH_KATAKANA_RUN, (run) => run.normalize("NFKC")); const createLookupWordsJa = (sourceStr: string): string[] => { const str = sourceStr @@ -35,13 +33,12 @@ const createLookupWordsJa = (sourceStr: string): string[] => { // For halfwidth input, keep both chains: every candidate retains its usual // prefix decomposition (show-all-plausible-candidates design), e.g. - // テレヒ -> テレヒ/テレ/テ plus テレビ/テレ/テ. + // テレヒ -> テレヒ/テレ/テ plus テレビ/テレ/テ. Without halfwidth katakana the + // replace is a no-op, so no second chain is added. const chains = [str]; - if (RE_HALFWIDTH_KATAKANA.test(str)) { - const converted = convertHalfwidthKatakana(str); - if (converted !== str) { - chains.push(converted); - } + const converted = convertHalfwidthKatakana(str); + if (converted !== str) { + chains.push(converted); } for (const chain of chains) {