Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
13 changes: 13 additions & 0 deletions __test__/main/core/entries.test.ts
Original file line number Diff line number Diff line change
Expand Up @@ -472,4 +472,17 @@ test("Test Japanese words", () => {
expect(createLookupWordsJa("走った")).toEqual(expect.arrayContaining(["走る"]));
expect(createLookupWordsJa("おいた")).toEqual(expect.arrayContaining(["おく", "おいる"]));
expect(createLookupWordsJa("19az")).toEqual(expect.arrayContaining(["19az"]));
// Halfwidth katakana must also reach fullwidth headwords (manga/UI text),
// keeping the usual prefix decomposition on both spellings.
expect(createLookupWordsJa("テレビ")).toEqual(expect.arrayContaining(["テレビ", "テレビ", "テレ", "テ", "テレヒ", "テレ", "テ"]));
expect(createLookupWordsJa("ガス")).toEqual(expect.arrayContaining(["ガス", "ガス", "ガ", "カ", "ガ"]));
// Conversion is scoped to halfwidth katakana: other characters must not get
// compatibility-folded (NFKC over the whole string would turn ① into "1").
const mixed = createLookupWordsJa("①テレビ");
expect(mixed).toEqual(expect.arrayContaining(["①テレビ", "①テレビ", "①テレ", "①テレ"]));
expect(mixed).not.toContain("1テレビ");
// Fullwidth katakana runs keep cutting into all prefixes (show-all-candidates design)
expect(createLookupWordsJa("ソフトウェアエンジニア")).toEqual(
expect.arrayContaining(["ソフトウェアエンジニア", "ソフトウェア", "ソフト", "ソ"]),
);
});
35 changes: 29 additions & 6 deletions src/main/core/entry/ja.ts
Original file line number Diff line number Diff line change
Expand Up @@ -10,6 +10,17 @@ import rule from "../rule";
const RE_ALPHABETS_NUMBERS = /[A-Za-z0-9]/g;
const FULLWIDTH_OFFSET = 0xfee0;

// Halfwidth katakana (U+FF61-FF9F, common in manga/UI text) is converted to
// its fullwidth form so テレビ looks up the same headwords as テレビ.
// Only maximal halfwidth-katakana runs are normalized: NFKC over the whole
// string would fold unrelated characters (e.g. ① -> 1). Normalizing each run
// lets NFKC compose voiced marks with the preceding letter (カ + ゙ -> ガ)
// without touching anything outside the run.
const RE_HALFWIDTH_KATAKANA_RUN = /[\uFF61-\uFF9F]+/g;

const convertHalfwidthKatakana = (s: string): string =>
s.replace(RE_HALFWIDTH_KATAKANA_RUN, (run) => run.normalize("NFKC"));

const createLookupWordsJa = (sourceStr: string): string[] => {
const str = sourceStr
.substring(0, 40)
Expand All @@ -20,13 +31,25 @@ const createLookupWordsJa = (sourceStr: string): string[] => {

result.push(sourceStr); // Add the original word

for (let i = str.length; i >= 1; i--) {
const part = str.substring(0, i);
result.push(part);
// For halfwidth input, keep both chains: every candidate retains its usual
// prefix decomposition (show-all-plausible-candidates design), e.g.
// テレヒ -> テレヒ/テレ/テ plus テレビ/テレ/テ. Without halfwidth katakana the
// replace is a no-op, so no second chain is added.
const chains = [str];
const converted = convertHalfwidthKatakana(str);
if (converted !== str) {
chains.push(converted);
}

for (const chain of chains) {
for (let i = chain.length; i >= 1; i--) {
const part = chain.substring(0, i);
result.push(part);

if (i >= 2) {
const deinedWords = rule.doJa(part);
result.merge(deinedWords ?? []);
if (i >= 2) {
const deinedWords = rule.doJa(part);
result.merge(deinedWords ?? []);
}
}
}
return result.toArray();
Expand Down
Loading