From 8116d098b34070230375efd026a5b20879f4e5cb Mon Sep 17 00:00:00 2001 From: v4nn4 <1796093+v4nn4@users.noreply.github.com> Date: Tue, 25 Aug 2026 10:24:28 +0200 Subject: [PATCH] =?UTF-8?q?Add=20Esperanto,=20Scottish=20Gaelic,=20Guaran?= =?UTF-8?q?=C3=AD,=20Hawaiian=20engines;=20park=20Navajo?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Four new verb engines, each 100% against a single kaikki/UniMorph oracle (Beta tier), taking the live count 59 → 63: - Esperanto (epo): pure suffixation off the -i stem, zero irregulars. 139,993/139,993 forms, 2,891 lemmas, 100% rule-derived. - Scottish Gaelic (gla): broad/slender synthetic core (future/past/ conditional/imperative/relative/passive), lenition normalisation, ~10 suppletives mined; no synthetic present (periphrastic, out of scope). 5,396/5,396 forms, 794 lemmas. - Guaraní (grn): areal/aireal/stative classes + nasal harmony read from the kaikki gug-conj template names; 5 voices. 4,964/4,964 forms, 68 lemmas, ~99.75% rule-derived (16 soro coactive overrides). - Hawaiian (haw): derivational engine — hoʻo- causative (onset-driven allomorphy), full reduplication, -ʻia passive — verified against kaikki lemma-linked derived terms. Periphrastic TAM (ua/ke…nei/e…ana) is syntax, out of scope. 316/316 forms, 252 lemmas, 94.9% rule-derived. Navajo (nav): parked (docs/nav/oracles.md). A measured build attempt found only 13.9% of forms rule-derivable from UniMorph's surface data — templatic position-class polysynthesis with cross-slot morphophonemic fusion is a different architecture (Athabaskan FST + segmented lexicon) than ablaut's principal-parts-plus-rules model. High Valyrian/Klingon (constructed) skipped. Registration: Lang::{Epo,Gla,Grn,Haw} at ordinals 59–62, reverse.rs arrays →63, from_code aliases, Conjugation variants + build dispatch, PyO3 + wasm, Cargo includes, correctness.py NAMES, CI golden gates. Verified in main: 4 gates 100%, 376 lib tests, clippy -D warnings clean, fmt clean, wasm + python features compile. Co-Authored-By: Claude Opus 4.8 (1M context) --- .github/workflows/ci.yml | 46 ++ .gitignore | 1 + Cargo.toml | 8 + data/epo/overrides.tsv | 2 + data/epo/parts.tsv | 5 + data/gla/parts.tsv | 7 + data/gla/verbs.tsv | 795 ++++++++++++++++++++++++++++++++++ data/grn/overrides.tsv | 21 + data/grn/parts.tsv | 73 ++++ data/haw/overrides.tsv | 24 + data/haw/parts.tsv | 254 +++++++++++ docs/epo/adjudications.tsv | 2 + docs/gla/adjudications.tsv | 1 + docs/gla/oracles.md | 52 +++ docs/grn/adjudications.tsv | 1 + docs/haw/adjudications.tsv | 1 + docs/nav/oracles.md | 35 ++ scripts/correctness.py | 2 +- scripts/epo/fetch_kaikki.sh | 11 + scripts/epo/kaikki_to_tsv.py | 87 ++++ scripts/gla/fetch_kaikki.sh | 14 + scripts/gla/kaikki_to_tsv.py | 209 +++++++++ scripts/grn/build.py | 107 +++++ scripts/grn/fetch_kaikki.sh | 8 + scripts/haw/fetch_kaikki.sh | 13 + scripts/haw/kaikki_to_tsv.py | 78 ++++ scripts/haw/mine_overrides.py | 98 +++++ scripts/nav/fetch_unimorph.sh | 7 + src/bin/golden_epo.rs | 56 +++ src/bin/golden_gla.rs | 79 ++++ src/bin/golden_grn.rs | 59 +++ src/bin/golden_haw.rs | 55 +++ src/epo.rs | 324 ++++++++++++++ src/gla.rs | 406 +++++++++++++++++ src/grn.rs | 593 +++++++++++++++++++++++++ src/haw.rs | 286 ++++++++++++ src/lib.rs | 40 ++ src/python.rs | 175 ++++++++ src/reverse.rs | 92 +++- src/wasm.rs | 20 + 40 files changed, 4144 insertions(+), 3 deletions(-) create mode 100644 data/epo/overrides.tsv create mode 100644 data/epo/parts.tsv create mode 100644 data/gla/parts.tsv create mode 100644 data/gla/verbs.tsv create mode 100644 data/grn/overrides.tsv create mode 100644 data/grn/parts.tsv create mode 100644 data/haw/overrides.tsv create mode 100644 data/haw/parts.tsv create mode 100644 docs/epo/adjudications.tsv create mode 100644 docs/gla/adjudications.tsv create mode 100644 docs/gla/oracles.md create mode 100644 docs/grn/adjudications.tsv create mode 100644 docs/haw/adjudications.tsv create mode 100644 docs/nav/oracles.md create mode 100755 scripts/epo/fetch_kaikki.sh create mode 100755 scripts/epo/kaikki_to_tsv.py create mode 100755 scripts/gla/fetch_kaikki.sh create mode 100644 scripts/gla/kaikki_to_tsv.py create mode 100644 scripts/grn/build.py create mode 100755 scripts/grn/fetch_kaikki.sh create mode 100755 scripts/haw/fetch_kaikki.sh create mode 100644 scripts/haw/kaikki_to_tsv.py create mode 100644 scripts/haw/mine_overrides.py create mode 100755 scripts/nav/fetch_unimorph.sh create mode 100644 src/bin/golden_epo.rs create mode 100644 src/bin/golden_gla.rs create mode 100644 src/bin/golden_grn.rs create mode 100644 src/bin/golden_haw.rs create mode 100644 src/epo.rs create mode 100644 src/gla.rs create mode 100644 src/grn.rs create mode 100644 src/haw.rs diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index 68cfe2cf..d2f9aced 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -981,5 +981,51 @@ jobs: run: ./scripts/zul/fetch_kaikki.sh - name: Zulu golden harness with two-oracle regression gate run: cargo run --release --bin golden_zul -- data/zul/unimorph.tsv data/zul/kaikki.tsv --check + - name: Cache kaikki epo + id: kaikki-epo + uses: actions/cache@v5 + with: + path: data/epo/kaikki.tsv + key: kaikki-epo-v1 + - name: Fetch kaikki epo + if: steps.kaikki-epo.outputs.cache-hit != 'true' + run: ./scripts/epo/fetch_kaikki.sh + - name: Esperanto golden harness with single-oracle regression gate (Beta) + run: cargo run --release --bin golden_epo -- data/epo/kaikki.tsv --check + - name: Cache kaikki gla + id: kaikki-gla + uses: actions/cache@v5 + with: + path: data/gla/kaikki.tsv + key: kaikki-gla-v1 + - name: Fetch kaikki gla + if: steps.kaikki-gla.outputs.cache-hit != 'true' + run: ./scripts/gla/fetch_kaikki.sh + - name: Scottish Gaelic golden harness with single-oracle regression gate (Beta) + run: cargo run --release --bin golden_gla -- data/gla/kaikki.tsv --check + - name: Cache kaikki grn + id: kaikki-grn + uses: actions/cache@v5 + with: + path: data/grn/kaikki-raw.jsonl + key: kaikki-grn-v1 + - name: Fetch kaikki grn + if: steps.kaikki-grn.outputs.cache-hit != 'true' + run: ./scripts/grn/fetch_kaikki.sh + - name: Build grn gold + run: python3 scripts/grn/build.py + - name: Guaraní golden harness with single-oracle regression gate (Beta) + run: cargo run --release --bin golden_grn -- data/grn/kaikki.tsv --check + - name: Cache kaikki haw + id: kaikki-haw + uses: actions/cache@v5 + with: + path: data/haw/kaikki.tsv + key: kaikki-haw-v1 + - name: Fetch kaikki haw + if: steps.kaikki-haw.outputs.cache-hit != 'true' + run: ./scripts/haw/fetch_kaikki.sh + - name: Hawaiian golden harness with single-oracle regression gate (Beta) + run: cargo run --release --bin golden_haw -- data/haw/kaikki.tsv --check - name: Reverse-lookup gate (fra, spa, eng; deu needs the kaikki dump) run: cargo run --release --bin reverse_gate -- --check --langs fra,spa,eng diff --git a/.gitignore b/.gitignore index 3b820104..cca99585 100644 --- a/.gitignore +++ b/.gitignore @@ -9,5 +9,6 @@ Cargo.lock __pycache__/ !/data/*/parts.tsv !/data/*/aux.tsv +!/data/*/overrides.tsv .claude/worktrees/ errors.log diff --git a/Cargo.toml b/Cargo.toml index 90299f45..d255fd34 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -72,6 +72,14 @@ include = [ "data/ind/overrides.tsv", "data/zul/parts.tsv", "data/zul/overrides.tsv", + "data/epo/parts.tsv", + "data/epo/overrides.tsv", + "data/gla/verbs.tsv", + "data/gla/parts.tsv", + "data/grn/parts.tsv", + "data/grn/overrides.tsv", + "data/haw/parts.tsv", + "data/haw/overrides.tsv", "data/por/classes.tsv", "data/por/verbs.tsv", "data/ron/classes.tsv", diff --git a/data/epo/overrides.tsv b/data/epo/overrides.tsv new file mode 100644 index 00000000..d352611c --- /dev/null +++ b/data/epo/overrides.tsv @@ -0,0 +1,2 @@ +# Per-cell overrides: none. Esperanto has no irregular verbs. +# lemma features form diff --git a/data/epo/parts.tsv b/data/epo/parts.tsv new file mode 100644 index 00000000..fd82d9bd --- /dev/null +++ b/data/epo/parts.tsv @@ -0,0 +1,5 @@ +# Esperanto is perfectly regular: no irregular verbs, no principal +# parts to mine. This file exists only so the reverse-index include +# (src/reverse.rs) and Cargo package include resolve. Column 1 (lemma) +# is intentionally empty. +# lemma diff --git a/data/gla/parts.tsv b/data/gla/parts.tsv new file mode 100644 index 00000000..596ae150 --- /dev/null +++ b/data/gla/parts.tsv @@ -0,0 +1,7 @@ +# Suppletive / irregular overrides that the mined principal parts +# cannot express. Columns: +# lemma past future vn ptcp fut-dep pst-dep cond-stem +# "-" = keep the mined/derived value. cond-stem, when set, replaces the +# lemma as the base for the derived relative future, conditional and +# imperative forms (thoir → bheir-/thug-). +thoir - - - - - - beir diff --git a/data/gla/verbs.tsv b/data/gla/verbs.tsv new file mode 100644 index 00000000..1ddf6394 --- /dev/null +++ b/data/gla/verbs.tsv @@ -0,0 +1,795 @@ +# lemma past future vn ptcp (mined principal parts, delenited past; "-" = derive) +abair tuirt their ràdh ràite +adhbharaich adhbharaich adhbharaichidh adhbharachadh adhbharaichte +adhlac adhlac adhlacaidh adhlacadh adhlacte +adhlaic adhlaic adhlaicidh adhlacadh - +ag ag agidh agadh - +agair agair agairidh agairt - +aidich aidich aidichidh aideachadh aidichte +ainlean ainlean ainleanaidh ainleanmhainn ainleanta +ainmich ainmich ainmichidh ainmeachadh ainmichte +ainneartaich ainneartaich ainneartaichidh ainneartachadh ainneartaichte +aisling aisling aislingidh aisling aislingte +aithnich aithnich aithnichidh aithneachadh aithnichte +aithris aithris aithrisidh aithris aithriste +altraim altraim altraimidh altram altraimte +amail amail amailidh amal amailte +amais amais amaisidh amas amaiste +amhairc amhairc amhaircidh amharc amhaircte +aodaich aodaich aodaichidh aodachadh aodaichte +aonaich aonaich aonaichidh aonachadh aonaichte +aontaich aontaich aontaichidh aontachadh aontaichte +armaich armaich armaichidh armachadh armaichte +at at ataidh at athte +ath-agair ath-agair ath-agairidh ath-agairt ath-agairte +ath-ainmich ath-ainmich ath-ainmichidh ath-ainmeachadh ath-ainmichte +ath-aithris ath-aithris ath-aithrisidh ath-aithris ath-aithriste +ath-bheachdaich ath-bheachdaich ath-bheachdaichidh ath-bheachdachadh ath-bheachdaichte +ath-bheothaich ath-bheothaich ath-bheothaichidh ath-bheothachadh ath-bheothaichte +ath-bheòthaich ath-bheòthaich ath-bheòthaichidh ath-bheòthachadh ath-bheòthaichte +ath-bhreithnich ath-bhreithnich ath-bhreithnichidh ath-bhreithneachadh ath-bhreithnichte +ath-bhrod - - ath-bhrodadh - +ath-bhrosnaich ath-bhrosnaich ath-bhrosnaichidh ath-bhrosnachadh ath-bhrosnaichte +ath-bhuannaich ath-bhuannaich ath-bhuannaichidh ath-bhuannachadh ath-bhuannaichte +ath-bhuidhinn ath-bhuidhinn ath-bhuidhnidh ath-bhuidhinn ath-bhuidhinnte +ath-chagainn ath-chagainn ath-chagnaidh ath-chagnadh ath-chagainnte +ath-cheannaich ath-cheannaich ath-cheannaichidh ath-cheannach ath-cheannaichte +ath-cheannsaich ath-cheannsaich ath-cheannsaichidh ath-cheannsachadh ath-cheannsaichte +ath-cheasnaich ath-cheasnaich ath-cheasnaichidh ath-cheasnachadh ath-cheasnaichte +ath-chlaon ath-chlaon ath-chlaonaidh ath-chlaonadh ath-chlaonte +ath-choisich ath-choisich ath-choisichidh ath-choiseachd ath-choisichte +ath-chruinnich ath-chruinnich ath-chruinnichidh ath-chruinneachadh ath-chruinnichte +ath-chruthaich ath-chruthaich ath-chruthaichidh ath-chruthachadh ath-chruthaichte +ath-chuairtich ath-chuairtich ath-chuairtichidh ath-chuairteachadh ath-chuairtichte +ath-chuir ath-chuir ath-chuiridh ath-chur ath-chuirte +ath-dhealbh ath-dhealbh ath-dhealbhaidh ath-dhealbhadh ath-dhealbhta +ath-dhèan ath-rinn ath-nì ath-dhèanamh ath-dhèanta +ath-dhìol ath-dhìol ath-dhìolaidh ath-dhìoladh ath-dhìolte +ath-dhùblaich ath-dhùblaich ath-dhùblaichidh ath-dhùblachadh ath-dhùblaichte +ath-dhùisg ath-dhùisg ath-dhùisgidh ath-dhùsgadh ath-dhùisgte +ath-eagaraich ath-eagaraich ath-eagaraichidh ath-eagrachadh ath-eagaraichte +ath-fhasdaich ath-fhasdaich ath-fhasdaichidh ath-fhasdachadh ath-fhasdaichte +ath-ghabh ath-ghabh ath-ghabhaidh ath-ghabhail ath-ghabhte +ath-ghin ath-ghin ath-ghinidh ath-ghintinn ath-ghinte +ath-ghlac ath-ghlac ath-ghlacaidh ath-ghlacadh ath-ghlacte +ath-innis ath-innis ath-innsidh ath-innse ath-inniste +ath-lathaich ath-lathaich ath-lathaichidh ath-lathachadh ath-lathaichte +ath-leasaich ath-leasaich ath-leasaichidh ath-leasachadh ath-leasaichte +ath-leum ath-leum ath-leumaidh ath-leum ath-leumte +ath-lorgaich ath-lorgaich ath-lorgaichidh ath-lorgachadh ath-lorgaichte +ath-lìon ath-lìon ath-lìonaidh ath-lìonadh ath-lìonte +ath-neartaich ath-neartaich ath-neartaichidh ath-neartachadh ath-neartaichte +ath-nochd ath-nochd ath-nochdaidh ath-nochdadh ath-nochdte +ath-nuadhaich ath-nuadhaich ath-nuadhaichidh ath-nuadhachadh ath-nuadhaichte +ath-roinn ath-roinn ath-roinnidh ath-roinn ath-roinnte +ath-ruith ath-ruith ath-ruithidh ath-ruith ath-ruithte +ath-rèitich ath-rèitich ath-rèitichidh ath-rèiteachadh ath-rèitichte +ath-sgrìobh ath-sgrìobh ath-sgrìobhaidh ath-sgrìobhadh ath-sgrìobhte +ath-sgrùd ath-sgrùd ath-sgrùdaidh ath-sgrùdadh ath-sgrùdte +ath-shealbhaich ath-shealbhaich ath-shealbhaichidh ath-shealbhachadh ath-shealbhaichte +ath-sheall ath-sheall ath-sheallaidh ath-shealltainn ath-sheallte +ath-shuidhich ath-shuidhich ath-shuidhichidh ath-shuidheachadh ath-shuidhichte +ath-smaoinich ath-smaoinich ath-smaoinichidh ath-smaoineachadh ath-smaoinichte +ath-thagair ath-thagair ath-thagraidh ath-thagairt ath-thagairte +ath-thagh ath-thagh ath-thaghaidh ath-thagadh ath-thaghte +ath-thog ath-thog ath-thogaidh ath-thogail ath-thogte +ath-thuislich ath-thuislich ath-thuislichidh ath-thuisleachadh ath-thuislichte +ath-thuit ath-thuit ath-thuitidh ath-thuiteam ath-thuite +ath-thòisich ath-thòisich ath-thòisichidh ath-thòiseachadh ath-thòisichte +ath-ùraich ath-ùraich ath-ùraichidh ath-ùrachadh ath-ùraichte +atharraich atharraich atharraichidh atharrachadh atharraichte +bac bac bacaidh bacadh bacte +bagair bagair bagairidh bagairt bagairte +baist baist baistidh baisteadh baiste +barantaich barantaich barantaichidh barantachd barantaichte +barrantaich barrantaich barrantaichidh barrantachd barrantaichte +beadraich beadraich beadraichidh beadradh beadraichte +beannaich beannaich beannaichidh beannachadh beannaichte +beartaich beartaich beartaichidh beartachadh beartaichte +beir rug beiridh breith beirte +beothaich beothaich beothaichidh beothachadh beothaichte +beuc beuc beucaidh beucadh beucte +beàrr beàrr bearraidh bearradh beàrrte +beò-ghlac beò-ghlac beò-ghlacaidh beò-ghlacadh beò-ghlacte +beòthaich beòthaich beòthaichidh beòthachadh - +bhòt bhòt bhòtaidh bhòtadh - +blais blais blaisidh blasad blaiste +bleadraig bleadraig bleadraigidh bleadraigeadh bleadraigte +bleoghainn bleoghainn bleoghainnidh bleoghann bleoghainnte +blàthaich blàthaich blàthaichidh blàthachadh blàthaichte +boc boc bocaidh bocadh bocte +bocsaig bocsaig bocsaigidh bocsaigeadh bocsaigte +brabhsaich - - brabhsadh - +breab breab breabaidh breabadh breabte +breac breac breacaidh breacadh breacte +breislich breislich breislichidh breisleachadh breislichte +breithnich breithnich breithnichidh breithneachadh breithnichte +breun breun breunaidh breunad breunte +briog briog briogaidh briogadh briogte +bris bris brisidh briseadh briste +brod brod brodaidh brodadh brodte +bronn bronn bronnaidh bronnadh bronnta +brosnaich brosnaich brosnaichidh brosnachadh - +bruadair bruadair bruadairidh bruadar bruadairte +bruich bruich bruichidh bruicheadh bruichte +bruidhinn bruidhinn bruidhnidh bruidhinn bruidhnte +bruis bruis bruisidh bruiseadh bruiste +brunndail brunndail brunndailidh brunndail brunndailte +brùnndail brùnndail brùnndailidh brùnndail brùnndailte +buail buail buailidh bualadh buailte +buain buain buainidh buain buainte +buair buair buairidh buaireadh buairte +buannaich buannaich buannaichidh buannachadh buannaichte +builg builg builgidh builgeadh builgte +buin buin buinidh buntainn buinte +buinnig buinnig buinnigidh buinnigeadh - +bàsaich bàsaich bàsaichidh bàsachadh bàsaichte +bàth bàth bàthaidh bàthadh bàthte +bòc bòc bòcaidh bòcadh bòcte +bòrd bòrd bòrdaidh bòrdadh bòrdte +cac cac cacaidh cac cacte +cadail - - cadal - +cagainn cagainn cagnaidh cagnadh cagainnte +cagair cagair cagairidh cagar cagairte +caidil caidil caidlidh cadal caidilte +caill caill caillidh call caillte +caisg caisg caisgidh casg caisgte +caith caith caithidh caitheamh caithte +callaich callaich callaichidh callachadh callaichte +can can canaidh cantainn cante +caochail caochail caochailidh caochladh caochailte +caog caog caogaidh caogadh caogte +caoidh caoidh caoidhidh caoidh caoidhte +caoin caoin caoinidh caoineadh caointe +caomhainn caomhainn caoimhnidh caomhnadh caomhainnte +caraich caraich caraichidh carachadh caraichte +ceangail ceangail ceanglaidh ceangal ceangailte +ceannaich ceannaich ceannaichidh ceannach ceannaichte +ceannsaich ceannsaich ceannsaichidh ceannsachadh ceannsaichte +ceileir ceileir ceileiridh ceileireadh ceileirte +ceumnaich ceumnaich ceumnaichidh ceumnachadh ceumnaichte +ciallaich ciallaich ciallaichidh ciallachadh ciallaichte +cinn cinn cinnidh cinntinn cinnte +ciùrr ciùrr ciùrraidh ciùrradh ciùrrte +clach clach clachaidh clachadh clachte +cladh cladh cladhaidh cladh cladhte +cladhaich cladhaich cladhaichidh cladhach cladhaichte +claon claon claonaidh claonadh claonte +cleachd cleachd cleachdaidh cleachdadh cleachdte +cleasaich cleasaich cleasaichidh cleasachd cleasaichte +cliath cliath cliathaidh cliathadh cliathte +cliog cliog cliogaidh cliogadh cliogte +cliop cliop cliopaidh cliopadh - +clisg clisg clisgidh clisgeadh clisgte +cluich cluich cluichidh cluich cluichte +cluinn cuala cluinnidh cluinntinn cluinnte +clàraich clàraich clàraichidh clàrachadh clàraichte +clìoraig clìoraig clìoraigidh clìoraigeadh - +clò-bhuail clò-bhuail clò-bhuailidh clò-bhualadh clò-bhuailte +clò-sgrìobh clò-sgrìobh clò-sgrìobhaidh clò-sgrìobhadh clò-sgrìobhte +cnag cnag cnagaidh cnagadh cnagte +cnuasaich cnuasaich cnuasaichidh cnuasachadh cnuasaichte +cnàmh cnàmh cnàmhaidh cnàmh cnàmhte +cnàmh-loisg cnàmh-loisg cnàmh-loisgidh cnàmh-losgadh cnàmh-loisgte +co-aonaich co-aonaich co-aonaichidh co-aonachadh co-aonaichte +co-aontaich co-aontaich co-aontaichidh co-aontachadh co-aontaichte +co-cheangail co-cheangail co-cheanglaidh co-cheangal co-cheangailte +co-naisg co-naisg co-naisgidh co-nasgadh co-naisgte +co-roinn co-roinn co-roinnidh co-roinn co-roinnte +co-òrdaich co-òrdaich co-òrdaichidh co-òrdachadh co-òrdaichte +co-òrdanaich co-òrdanaich co-òrdanaichidh co-òrdanachadh co-òrdanaichte +cofhurtaich cofhurtaich cofhurtaichidh cofhurtachadh cofhurtaichte +cog cog cogaidh cogadh cogte +coilean coilean coileanaidh coileanadh coileante +coimeas coimeas coimeasaidh coimeas coimeaste +coimhead coimhead coimheadaidh coimhead coimheadte +coinnich coinnich coinnichidh coinneachadh coinnichte +coisich coisich coisichidh coiseachd coisichte +coisinn coisinn coisinnidh cosnadh coisinnte +coisrig coisrig coisrigidh coisrigeadh coisrigte +coitich coitich coitichidh coiteachadh coitichte +comh-shnìomh comh-shnìomh comh-shnìomhaidh comh-shnìomh comh-shnìomhte +comhairlich comhairlich comhairlichidh comhairleachadh comhairlichte +comharraich comharraich comharraichidh comharrachadh comharraichte +comhartaich comhartaich comhartaichidh comhartaich comhartaichte +connsaich connsaich connsaichidh connsachadh connsaichte +cop cop copidh copadh copte +corraich corraich corraichidh corrachadh corraichte +cosg cosg cosgaidh cosg cosgte +craobh craobh craobhaidh craobhadh craobhta +craol craol craolaidh - - +crath crath crathaidh crathadh crathte +creach creach creachaidh creachadh creachte +creid creid creididh creidsinn creidte +criom criom criomaidh criomadh criomte +crith crith crithidh crith crithte +croch croch crochaidh crochadh crochte +crom crom cromaidh cromadh cromte +cronaich cronaich cronaichidh cronachadh cronaichte +cruach cruach cruachaidh cruachadh cruachte +crudhaich crudhaich crudhaichidh crudhachadh crudhaichte +cruinnich cruinnich cruinnichidh cruinneachadh cruinnichte +cruthaich cruthaich cruthaichidh cruthachadh cruthaichte +cràidh cràidh cràidhidh cràdh cràidhte +crìochnaich crìochnaich crìochnaichidh crìochnachadh crìochnaichte +crìon crìon crìonaidh crìonadh crìonte +crùb crùb crùbaidh crùbadh crùbte +crùn crùn crùnaidh crùnadh crùinte +cuairtich cuairtich cuairtichidh cuairteachadh cuairtichte +cuartaich cuartaich cuartaichidh cuartachadh cuartaichte +cuibhrich cuibhrich cuibhrichidh cuibhreachadh cuibhrichte +cuidich cuidich cuidichidh cuideachadh cuidichte +cuimhnich cuimhnich cuimhnichidh cuimhneachadh cuimhnichte +cuip cuip cuipidh cuipeadh cuipte +cuir cuir cuiridh cur cuirte +cuir-ris cuir-ris cuiridh-ris cur-ris - +cum cum cumaidh cumadh cumta +cunnt cunnt cunntaidh cunntadh cunnte +càin càin càinidh càineadh càinte +càirich càirich càirichidh càradh càirichte +cèir cèir cèiridh cèireadh cèirte +cèirich cèirich cèirichidh cèireachadh cèirichte +cìl cìl cìlidh cìleadh cìlte +cìr cìr cìridh cìreadh cìrte +còmhdaich còmhdaich còmhdaichidh còmhdachadh còmhdaichte +còmhnaich còmhnaich còmhnaichidh còmhnaidh còmhnaichte +còrd còrd còrdaidh còrdadh còrdte +cùm cùm cumaidh cumail cumta +cùnnt cùnnt cùnntaidh cùnntadh cùnnte +daingnich daingnich daingnichidh daingneachadh daingnichte +dall dall dallaidh dalladh dallta +danns danns dannsaidh dannsadh dannste +dealaich dealaich dealaichidh dealachadh dealaichte +dealbh dealbh dealbhaidh dealbhadh dealbhta +dealuich dealuich dealuichidh dealuicheadh dealuichte +dearbh dearbh dearbhaidh dearbhadh dearbhte +dearc dearc dearcaidh dearcadh dearcte +dearg dearg deargaidh deargadh deargte +dearrs dearrs dearrsaidh dearrsadh dearrste +dearrsaich dearrsaich dearrasaichidh dearrsadh dearrasaichte +deasaich deasaich deasaichidh deasachadh deasaichte +deasbair deasbair deasbairidh deasbaireachd deasbairte +deoc deoc deocaidh deocadh deocte +deoghail deoghail deoghailidh deoghal deoghailte +deàrrs deàrrs deàrrsaidh deàrrsadh deàrrste +dian-amhairc - - dian-amharc - +diùlt diùlt diùltaidh diùltadh diùlte +doimhnich doimhnich doimhnichidh doimhneachadh doimhnichte +dorchaich dorchaich dorchaichidh dorchadh dorchaichte +draibh draibh draibhidh draibheadh draibhte +dreachd dreachd dreachdaidh dreachdadh dreachdta +druid druid druididh druideadh druidte +drùidh drùidh drùidhidh drùdhadh - +drùin drùin drùinidh drùineadh drùinte +dual dual dualidh dualadh dualte +dubh dubh dubhaidh dubhadh dubhte +dustaig dustaig dustaigidh dustaigeadh dustaigte +dàir dàir dàiridh dàir dàirte +dèan rinn nì dèanamh dèanta +dèilig dèilig dèiligidh dèiligeadh dèiligte +dì-armaich dì-armaich dì-armaichidh dì-armachadh dì-armaichte +dì-cheann dì-cheann dì-cheannaidh dì-cheannadh dì-cheannte +dì-cheannaich dì-cheannaich dì-cheannaichidh dì-cheannach dì-cheannaichte +dì-luchdaich dì-luchdaich dì-luchdaichidh dì-luchdachadh dì-luchdaichte +dì-làraich dì-làraich dì-làraichidh dì-làrachadh dì-làraichte +dìobair dìobair dìobraidh dìobradh dìobairte +dìochuimhnich dìochuimhnich dìochuimhnichidh dìochuimhneachadh dìochuimhnichte +dìoghail dìoghail dìoghlaidh dìoghladh dìoghailte +dìol dìol dìolaidh dìoladh dìolte +dìon dìon - dìon dìonte +dìrich dìrich dìrichidh dìreadh dìrichte +dòirt dòirt dòirtidh dòrtadh dòirte +dòth dòth dòthaidh dòthadh dòthte +dùin dùin dùinidh dùnadh dùinte +dùisg dùisg dùisgidh dùsgadh dùisgte +eadar-cheangail eadar-cheangail eadar-cheanglaidh eadar-cheangal eadar-cheangailte +eadar-chuir eadar-chuir eadar-chuiridh eadar-chur eadar-chuirte +eadar-dhealaich eadar-dhealaich eadar-dhealaichidh eadar-dhealachadh eadar-dhealaichte +eadar-dhuilleagaich eadar-dhuilleagaich eadar-dhuilleagaichidh eadar-dhuilleagachadh eadar-dhuilleagaichte +eadar-dhuillich eadar-dhuillich eadar-dhuillichidh eadar-dhuilleachadh eadar-dhuillichte +eadar-fhigh eadar-fhigh eadar-fhighidh eadar-fhìgheadh eadar-fhighte +eadar-fhill eadar-fhill eadar-fhillidh eadar-fhilleadh eadar-fhillte +eadar-mhìnich eadar-mhìnich eadar-mhìnichidh eadar-mhìneachadh eadar-mhìnichte +eadar-phongaich eadar-phongaich eadar-phongaichidh eadar-phongachadh eadar-phongaichte +eadar-phòs eadar-phòs eadar-phòsaidh eadar-phòsadh eadar-phòsta +eadar-sgaoil eadar-sgaoil eadar-sgaoilidh eadar-sgaoileadh eadar-sgaoilte +eadar-sgap eadar-sgap eadar-sgapaidh eadar-sgapadh eadar-sgapte +eadar-sgrìobh eadar-sgrìobh eadar-sgrìobhaidh eadar-sgrìobhadh eadar-sgrìobhta +eadar-shreathaich eadar-shreathaich eadar-shreathaichidh eadar-shreathadh eadar-shreathte +eadar-theangaich eadar-theangaich eadar-theangaichidh eadar-theangachadh eadar-theangaichte +eadar-thoinn eadar-thoinn eadar-thoinnidh eadar-thoinneadh eadar-thoinnte +earb earb earbaidh earbsadh earbte +eas-aontaich eas-aontaich eas-aontaichidh eas-aontachadh eas-aontaichte +eas-onoraich eas-onoraich eas-onoraichidh eas-onorachadh eas-onoraichte +eas-urramaich eas-urramaich eas-urramaichidh eas-urramachadh eas-urramaichte +eas-òrdaich eas-òrdaich eas-òrdaichidh eas-òrdachadh eas-òrdaichte +eas-ùmhlaich eas-ùmhlaich eas-ùmhlaichidh eas-ùmhlachadh eas-ùmhlaichte +eiridnich eiridnich eiridnichidh eiridneachadh eiridnichte +eug eug eugaidh eugadh eugte +faic cunnaic chì faicinn faicte +faigh fuair gheibh faighinn faighte +faighnich faighnich faighnichidh faighneachd faighnichte +failc failc failcidh failceadh failcte +fairich fairich fairichidh faireachdainn fairichte +falaich falaich falaichidh falach falaichte +falamhaich falamhaich falamhaichidh falamhachadh falamhaichte +falbh falbh falbhaidh falbh falbhte +fan fan fanaidh fantail - +fannaich fannaich fannaichidh fannachadh fannaichte +farraid farraid farraididh farraid farraidte +fasdaich fasdaich fasdaichidh fasdachadh fasdaichte +fast fast fastaidh fastadh - +fastaich fastaich fastaichidh fastachadh fastaichte +fastaidh fastaidh fastaidhidh fastadh fastaidhte +fead fead feadaidh feadail feadte +feall feall feallaidh fealladh - +feamainn feamainn feamainnidh feamnadh feamainnte +feann feann feannaidh feannadh - +feith feith feithidh feitheamh feithte +feuch feuch feuchaidh feuchainn feuchte +feur feur feuraidh feuradh feurte +fiar fiar fiaraidh fiaradh fiarte +fiaraich fiaraich fiaraichidh fiarachadh fiaraichte +figh figh fighidh fighe fighte +fill fill fillidh filleadh fillte +fionn fionn fionnaidh fionnadh fionnte +fionnaraich fionnaraich fionnaraichidh fionnarachadh fionnaraichte +fliuch fliuch fliuchaidh fliuchadh fliuchte +fo-roinn fo-roinn fo-roinnidh fo-roinn fo-roinnte +fo-sgrìobh fo-sgrìobh fo-sgrìobhaidh fo-sgrìobhadh fo-sgrìobhte +foghain foghain fòghnaidh fòghnadh foghainte +foghlaim foghlaim foghlaimidh foghlam foghlaimte +foillsich foillsich follsichidh foillseachadh foillsichte +folaich folaich folaichidh folach folaichte +fosgail fosgail fosglaidh fosgladh fosgailte +fraidhig fraidhig fraidhigidh fraidhigeadh fraidhigte +fras fras frasaidh frasadh fraste +freagair freagair freagairidh freagairt freagairte +freumhaich freumhaich freumhaichidh freumhachadh freumhaichte +frith-ainmich frith-ainmich frith-ainmichidh frith-ainmeachadh frith-ainmichte +frith-bheart frith-bheart frith-bheartaidh frith-bheartadh frith-bhearte +frith-bhuail frith-bhuail frith-bhuailidh frith-bhualadh frith-bhuailte +frith-leum frith-leum frith-leumaidh frith-leum frith-leumte +fritheil fritheil fritheilidh frithealadh fritheilte +fuaigheil fuaigheil fuaighlidh fuaigheal fuaigheilte +fuaimnich fuaimnich fuaimnnichidh fuaimneachadh fuaimnichte +fuaraich fuaraich fuaraichidh fuarachadh fuaraichte +fuasgail fuasgail fuasgailidh fuasgladh fuasgailte +fuathaich fuathaich fuathaichidh fuathachadh fuathaichte +fuiling fuiling fuilingidh fulang fuilingte +fuin fuin fuinidh fuine fuinte +fuirich fuirich fuirichidh fuireach fuirichte +fàg fàg fàgaidh fàgail fàgte +fàillig fàillig fàilligidh fàilligeadh fàilligte +fàillinnich fàillinnich fàillinnichidh fàillinneachadh fàillinnichte +fàillnich fàillnich fàillnichidh fàillneachadh fàillnichte +fàisnich fàisnich fàisnichidh fàisneachadh fàisnichte +fàs fàs fàsaidh fàs fàsta +fàsaich fàsaich fàsaichidh fàsachadh fàsaichte +fògair fògair fògairidh fògradh fògairte +fòn fòn fònaidh fònadh fònte +gabh gabh gabhaidh gabhail gabhte +gail gail gailidh gal gailte +gairm gairm gairmidh gairm gairmte +gar gar garaidh garadh garte +geall geall geallaidh gealltainn geallte +geamhraich geamhraich geamhraichidh geamhrachadh geamhraichte +gearain gearain gearainidh gearan gearainte +gearr gearr gearraidh gearradh gearrta +geàrr geàrr gearraidh gearradh - +gin gin ginidh gintinn ginte +giorraich giorraich giorraichidh giorrachadh giorraichte +giùlain giùlain giùlainidh giùlan giùlainte +glac glac glacaidh glacadh glacte +glais glais glaisidh - - +glan glan glanaidh glanadh glante +glaodh glaodh glaodhaidh glaodhadh glaodhte +glas glas glasaidh - - +gleus gleus gleusaidh gleusadh gleuste +gluais gluais gluaisidh gluasad gluaiste +glèidh glèidh glèidhidh glèidheadh glèidhte +gléidh gléidh gléidhidh gléidheadh gléidhte +gnog gnog gnogaidh gnogadh gnogte +gnìomh gnìomh gnìomhaidh gnìomhadh gnìomhte +gnìomhaich gnìomhaich gnìomhaichidh gnìomhachadh gnìomhaichte +goid goid goididh goid goidte +goil goil goilidh goil goilte +gon gon gonaidh gonadh gonte +greas greas greasaidh greasadh greaste +greim greim greimidh greimeadh - +grunnaich grunnaich grunnaichidh grunnachadh grunnaichte +gràdhaich gràdhaich gràdhaichidh gràdhachadh gràdhaichte +grèim grèim greimidh greimeadh grèimte +grìos grìos grìosaidh grìosadh grìoste +grìosaich grìosaich grìosaichidh grìosachadh grìosaichte +guidh guidh guidhidh guidhe guidhte +guil guil guilidh gul guilte +guir guir guiridh gur guirte +gàir gàir gàiridh gàireachdainn gàirte +gèill gèill gèillidh gèilleadh gèillte +iadh iadh iadhaidh iadhadh iadhte +iarr iarr iarraidh iarraidh iarrte +iasgaich iasgaich iasgaichidh iasgach iasgaichte +iath iath iathaidh iathadh iadhte +ilmich ilmich ilmichidh ilmeach ilmichte +imich imich imichidh imeachd imichte +imlich imlich imlichidh imlich imlichte +imrich imrich imrichidh imrich imrichte +innis innis innsidh innse inniste +inntrinn inntrinn inntrinnidh inntreadh inntrinnte +iom-sheòl iom-sheòl iom-sheòlaidh iom-sheòladh iom-sheòlta +ioma-ghlac ioma-ghlac ioma-ghlacaidh ioma-ghlacadh ioma-ghlacte +iomair iomair iomairidh iomradh iomairte +iompaich iompaich iompaichidh iompachadh iompaichte +ionndrainn ionndrainn ionndrainnidh ionndrainn ionndrainnte +ionnlaid ionnlaid ionnlaididh ionnlaid ionnlaidte +ionnsaich ionnsaich ionnsaichidh ionnsachadh ionnsaichte +ioraim ioraim ioraimidh iormadh ioraimte +itealaich itealaich itealaichidh itealaich itealaichte +ith ith ithidh ithe ithte +labhair labhair labhraidh labhairt - +lagaich lagaich lagaichidh lagachadh lagaichte +laghaich laghaich laghaichidh laghachadh laghaichte +laigh laigh laighidh laighe laighte +leagh leagh leaghaidh leaghadh leaghte +lean lean leanaidh leantainn leanta +leasaich leasaich leasaichidh leasachadh leasaichte +leig leig leigidh leigeil leigte +leighis leighis leighisidh leigheas leighiste +leudaich leudaich leudaichidh leudachadh leudaichte +leugh leugh leughaidh leughadh leughte +leum leum leumaidh leum leumte +leòn leòn leònaidh leònadh leònta +ligh ligh lighidh lì lighte +litrich litrich litrichidh litreachadh litrichte +lobh lobh lobhaidh lobhadh lobhte +loisg loisg loisgidh losgadh loisgte +lorg lorg lorgaidh lorg lorgte +lot lot lotaidh lot - +luaidh luaidh luaidhidh luaidh luaidhte +luchdaich luchdaich luchdaichidh luchdadh luchdaichte +làimhsich làimhsich làimhsichidh làimhseachadh - +lèim lèim lèimidh lèim lèimte +lìbhrig lìbhrig lìbhrigidh lìbhrigeadh lìbhrigte +lìomh lìomh lìomhaidh lìomhadh lìomhte +lìon lìon lìonaidh lìonadh lìonte +lùb lùb lùbaidh lùbadh lùbte +mag mag magaidh magadh magte +mair mair mairidh mairsinn mairte +maistir maistir maistridh maistreadh - +maith maith maithidh mathadh maithte +malairt malairt malairtidh malairt malairte +malairtich malairtich malairtichidh malairteachadh malairtichte +mallaich mallaich mallaichidh mallachadh mallaichte +maoidh maoidh maoidhidh maoidheadh maoidhte +marbh marbh marbhaidh marbhadh marbhta +masg masg masgaidh masgadh - +maslaich maslaich maslaichidh maslachadh maslaichte +math math mathaidh mathadh mathte +meal meal mealaidh mealadh mealte +meall meall meallaidh mealladh meallta +meas meas measaidh measadh measte +measg measg measgaidh measgadh measgte +measgaich measgaich measgaichidh measgachadh measgaichte +meil meil meilidh meileadh meilte +meirg meirg meirgidh meirg meirgte +meòmhraich meòmhraich meòmhraichidh meòmhrachadh meòmhraichte +meòraich meòraich meòraichidh meòrachadh meòraichte +miannaich miannaich miannaichidh miannach miannaichte +mill mill millidh milleadh millte +misnich misnich misnichidh misneachadh misnichte +mol mol molaidh moladh molta +mort mort mortaidh mort morte +mothaich mothaich mothaichidh mothachadh mothaichte +muin muin muinidh muineadh muinte +murt murt murtaidh murt murte +màirnealaich màirnealaich màirnealaichidh màirnealachadh màirnealaichte +mèinn mèinn mèinnidh mèinneadh mèinnte +mì-bhreithnich mì-bhreithnich mì-bhreithnichidh mì-bhreithneachadh mì-bhreithnichte +mì-bhuilich mì-bhuilich mì-bhuilichidh mì-bhuileachadh mì-bhuilichte +mì-chaomhainn mì-chaomhainn mì-chaoimhnidh mì-chaomhnadh mì-chaomhainnte +mì-chleachd mì-chleachd mì-chleachdaidh mì-chleachdadh mì-chleachte +mì-chliùitich mì-chliùitich mì-chliùitichidh mì-chliùiteachadh mì-chliùitichte +mì-choisrig mì-choisrig mì-choisrigidh mì-choisrigeadh mì-choisrigte +mì-chomhairlich mì-chomhairlich mì-chomhairlichidh mì-chomhairleachadh mì-chomhairlichte +mì-chreid mì-chreid mì-chreididh mì-chreidsinn mì-chreidte +mì-chunnt mì-chunnt mi-chunntaidh mì-chunntadh mì-chunnte +mì-chàirich mì-chàirich mì-chàirichidh mì-chàireachadh mì-chàirichte +mì-chòrd mì-chòrd mi-chòrdaidh mì-chòrdadh mì-chordte +mì-dhealbh mì-dhealbh mì-dhealbhaidh mì-dhealbh mì-dhealbhte +mì-dhealbhaich mì-dhealbhaich mì-dhealbhaichidh mì-dhealbhachadh mì-dhealbhaichte +mì-dhreachaich mì-dhreachaich mì-dhreachaichidh mì-dhreachachadh mì-dhreachaichte +mì-gheuraich mì-gheuraich mì-gheuraichidh mì-gheurachadh mì-gheuraichte +mì-ghnàthaich mì-ghnàthaich mì-ghnàthaichidh mì-ghnàthachadh mì-ghnàthaichte +mì-laghaich mì-laghaich mì-laghaichidh mì-laghachadh mì-laghaichte +mì-mhisnich mì-mhisnich mì-mhisnichidh mì-mhisneachadh mì-mhisnichte +mì-mhol mì-mhol mì-mholaidh mì-mholadh mì-mholte +mì-mhìnich mì-mhìnich mì-mhìnichidh mì-mhìneachadh mì-mhìnichte +mì-naomhaich mì-naomhaich mì-naomhaichidh mì-naomhachadh mì-naomhaichte +mì-onoirich mì-onoirich mì-onoirichidh mì-onoireachadh mì-onoirichte +mì-sgeadaich mì-sgeadaich mì-sgeadaichidh mì-sgeadachadh mì-sgeadaichte +mì-sheòl mì-sheòl mì-sheòlaidh mì-sheòladh mì-sheòlte +mì-shnasaich mì-shnasaich mì-shnasaichidh mì-shnasachadh mì-shnasaichte +mì-stiùir mì-stiùir mì-stiùiridh mì-stiùireadh mì-stiùirte +mì-thaitinn mì-thaitinn mì-thaitnidh mì-thaitneadh mì-thaitinnte +mì-thaitnich mì-thaitnich mì-thaitnichidh mì-thaitneachadh mì-thaitnichte +mì-thoilich mì-thoilich mì-thoilichidh mì-thoileachadh mì-thoilichte +mì-threòirich mì-threòirich mì-threòirichidh mì-threòireachadh mì-threòirichte +mì-thràthaich mì-thràthaich mì-thràthaichidh mì-thràthachadh mì-thràthaichte +mì-thuig mì-thuig mì-thuigidh mì-thuigsinn mì-thuigte +mì-urramaich mì-urramaich mì-urramaichidh mì-urramachadh mì-urramaichte +mìnich mìnich mìnichidh mìneachadh mìnichte +mùch mùch mùchaidh mùchadh mùchte +mùin mùin mùinidh mùin mùinte +mùth mùth mùthaidh mùthadh mùthte +naisg naisg naisgidh nasgadh naisgte +naomhaich naomhaich naomhaichidh naomhachadh naomhaichte +neadaich neadaich neadaichidh neadachadh neadaichte +neo-dhèan neo-rinn neo-nì neo-dhèanamh neo-dhèanta +neulaich neulaich neulaichidh neulachadh te +nigh nigh nighidh nighe nighte +nochd nochd nochdaidh nochdadh nochdte +notaich notaich notaichidh notachadh notaichte +nuadhaich nuadhaich nuadhaichidh nuadhachadh nuadhaichte +nàraich nàraich nàraichidh nàrachadh nàraichte +obraich obraich obraichidh obrachadh obraichte +oibrich oibrich oibrichidh oibreachadh oibrichte +or or oridh oradh orte +pasaig pasaig pasaigidh pasaigeadh pasaigte +peacaich peacaich peacaichidh peacachadh peacaichte +peall peall peallaidh pealladh peallte +peanasaich peanasaich peanasaichidh peanasachadh peanasaichte +peant peant peandaidh peantadh peante +pian pian pianaidh pianadh piante +plaosg plaosg plaosgaidh plaosgadh plaosgte +plosg plosg plosgaidh plosgadh plosgte +poidsig poidsig poidsigidh poidseadh poidsigte +poll poll pollaidh polladh pollta +post post postaidh postadh poste +postaich postaich postaichidh postachadh postaichte +praidhig praidhig praidhigidh praidhigeadh praidhigte +preas preas preasaidh preasadh preaste +priob priob priobaidh priobadh priobte +pronn pronn pronnaidh pronnadh pronnte +put put putaidh putadh pute +pàigh pàigh pàighidh pàigheadh pàighte +pòg pòg pògaidh pògadh pògte +pòs pòs pòsaidh pòsadh pòsta +rach caidh thèid dol rachte +rannsaich rannsaich rannsaichidh rannsachadh rannsaichte +reic reic reicidh reic reicte +reoth reoth reothaidh reothadh reòthte +reòdh reòdh reòdhaidh reòdhadh reòdhte +reòth reòth reòthaidh reòthadh reòta +riaghail riaghail riaghailidh riaghailt riaghailte +riamh riamh riamhaidh riamhadh riamhte +riaraich riaraich riaraichidh riarachadh riaraichte +rib rib ribidh ribeadh ribte +ridhil ridhil ridhlidh ridhleadh - +riochdaich riochdaich riochdaichidh riochdachadh riochdaichte +riof riof riofaidh riofadh riofte +ro-shuidhich ro-shuidhich ro-shuidhichidh ro-shuidheachadh ro-shuidhichte +roghnaich roghnaich roghnaichidh roghnachadh roghnaichte +roilig roilig roiligidh roiligeadh roiligte +roinn roinn roinnidh roinn roinnte +rolaig rolaig rolaigidh rolaigeadh rolaigte +ruidhil ruidhil ruidhlidh ruidhleadh - +ruig ràinig ruigidh ruigsinn ruigte +ruith ruith ruithidh ruith ruithte +ràc ràc ràcaidh ràcadh ràcte +ràn ràn rànaidh rànail rànte +rèitich rèitich rèitichidh rèiteachadh rèitichte +ròst ròst ròstaidh ròstadh ròsta +rùdh rùdh rùdhaidh rùdhadh - +rùisg rùisg rùisgidh rùsgadh rùisgte +rùraich rùraich rùraichidh rùrach rùraichte +rùsg rùsg rùsgaidh rùsgadh rùsgte +sabaid sabaid sabaididh sabaid sabaidte +saill saill saillidh sailleadh saillte +samhlaich samhlaich samhlaichidh samhlachadh samhlaichte +saoil saoil saoilidh saoilsinn saoilte +saor saor saoraidh saoradh saorte +saothraich saothraich saothraichidh saothrachadh saothraichte +seachain seachain seachainidh seachnadh seachainte +seachainn seachainn seachainnidh seachnadh seachainnte +sealbhaich sealbhaich sealbhaichidh sealbhachadh sealbhaichte +sealg sealg sealgaidh sealg sealgte +seall seall seallaidh sealltainn seallte +searg searg seargaidh seargadh seargte +searmonaich searmonaich searmonaichidh searmonachadh searmonaichte +seas seas seasaidh seasamh seaste +seilbhich seilbhich seilbhichidh seilbheachadh seilbhichte +seinn seinn seinnidh seinn seinnte +seirm seirm seirmidh seirm seirmte +seun seun seunaidh seunad seunte +seòl seòl seòlaidh seòladh seòlta +seòrsaich seòrsaich seòrsaichidh seòrsachadh seòrsaichte +sgal sgal sgalaidh sgaladh sgalte +sgall sgall sgallaidh sgalladh sgallta +sgaoil sgaoil sgaoilidh sgaoileadh sgaoilte +sgap sgap sgapaidh sgapadh sgapte +sgar sgar sgaraidh sgaradh sgarte +sgeadaich sgeadaich sgeadaichidh sgeadachadh sgeadaichte +sgeith sgeith sgeithidh sgeitheadh sgeithte +sgol sgol sgolaidh sgoladh sgolte +sgrath sgrath sgrathaidh sgrathadh sgrathte +sgreamhaich sgreamhaich sgreamhaichidh sgreamhachadh sgreamhaichte +sgreuch sgreuch sgreuchaidh sgreuchail sgreuchte +sgrios sgrios sgriosaidh sgriosadh sgriosta +sgrog sgrog sgrogaidh sgrogadh - +sgrìob sgrìob sgrìobaidh sgrìobadh sgrìobte +sgrìobh sgrìobh sgrìobhaidh sgrìobhadh sgrìobhte +sgròb sgròb sgròbaidh sgròbadh sgròbte +sgrùdaich sgrùdaich sgrùdaichidh sgrùdachadh sgrùdaichte +sguab sguab sguabaidh sguabadh sguabte +sguir sguir sguiridh sgur sguirte +sgàin sgàin sgàinidh sgàineadh sgàinte +sgèith sgèith sgèithidh sgèitheadh sgèithte +sil sil silidh sileadh silte +sir sir siridh sireadh sirte +sitrich sitrich sitrichidh sitrich - +siubhail siubhail siubhlaidh siubhal siubhailte +slac slac slacaidh slacadh slacte +slaod slaod slaodaidh slaodadh slaodte +slisnich slisnich slisnichidh slisneachadh slisnichte +sloisir sloisir sloisridh sloisreadh sloisirte +sluig sluig sluigidh slugadh sluigte +slànaich slànaich slànaichidh slànachadh slànaichte +slìob slìob slìobidh slìobadh slìobte +slìog slìog slìogaidh slìogadh slìogte +smaoinich smaoinich smaoinichidh smaoineachadh smaoinichte +smaointich smaointich smaointichidh smaointeachadh smaointichte +smoc smoc smocaidh smocadh smocte +smuainich smuainich smuainichidh smuaineachadh smuainichte +smuaintich smuaintich smuaintichidh smuainteachadh smuaintichte +smàil smàil smàilidh smàladh smàilte +smàl smàl smàlaidh smàladh smàlte +smèid smèid smèididh smèideadh smèidte +smùid smùid smùididh smùideadh smùidte +snap snap snapaidh snapadh snapta +snàig snàig snàigidh snàgail snàigte +snàmh snàmh snàmhaidh snàmh snàmhte +snìomh snìomh snìomhaidh snìomh snìomhte +soidhn soidhn soidhnidh soidhneadh soidhnte +soilleirich soilleirich soilleirichidh soilleireachadh soilleirichte +soirbhich soirbhich soirbhichidh soirbheachadh soirbhichte +sorchaich sorchaich sorchaichidh sorchachadh sorchaichte +spad spad spadaidh spadadh spadte +speal speal spealaidh spealadh spealte +spealg spealg spealgaidh spealgadh spealgte +spleuchd spleuchd spleuchdaidh spleuchdadh spleuchdte +sponsairich sponsairich sponsairichidh sponsaradh sponsairichte +spoth spoth spothaidh spoth spothte +spàrr spàrr sparraidh sparradh sparrte +spìosraich spìosraich spìosraichidh spìosrachadh spìosraichte +srac - - sracadh - +srann srann srannaidh srannail srannte +sreap sreap sreapaidh sreap sreapte +sreoth sreoth sreothaidh sreothart - +srian srian srianaidh srianadh srianta +sruth sruth sruthaidh sruthadh sruthte +sràc sràc sràcaidh sràcadh sràcte +stad stad stadaidh stadadh stadte +steall steall steallaidh stealladh steallte +stiubh stiubh stiubhaidh stiubhadh stiubhte +stiubhaig stiubhaig stiubhaigidh stiubhaigeadh stiubhaigte +stiùir stiùir stiùiridh stiùireadh stiùirte +stiùirich stiùirich stiùirichidh stiùireadh stiùirte +stob stob stobaidh stobadh stobte +stoirmich stoirmich stoirmichidh stoirmeachadh stoirmichte +streap streap streapaidh streap streapte +stràc stràc stràcaidh stràcadh stràcte +strì strì strìthidh strì strìte +stuadh stuadh stuadhaidh stuadhadh - +stuagh stuagh stuaghaidh stuaghadh - +stèidh stèidh stèidhidh stèidheadh stèidhte +stèidhich stèidhich stèidhichidh stèidheachadh stèidhichte +stòir stòir stòiridh stòireadh stòirte +stòl - - stòladh - +stòr stòr stòraidh stòradh stòrte +suidh suidh suidhidh suidhe suidhte +suidhich suidhich suidhichidh suidheachadh suidhichte +sàbh sàbh sàbhaidh sàbhadh sàbhte +sàbhail sàbhail sàbhailidh sàbhaladh sàbhailte +sàil sàil sàilidh sàileadh sàilte +sàraich sàraich sàraichidh sàrachadh sàraichte +sàs sàs sàsaidh sàsadh sàsta +sàsaich sàsaich sàsaichidh sàsachadh sàsaichte +sèid sèid sèididh sèideadh sèidte +sèimhich sèimhich sèimhichidh sèimheachadh sèimhichte +séid séid séididh séideadh séidte +sìn sìn sìnidh sìneadh sìnte +sìolaich sìolaich sìolaichidh sìolachadh sìolaichte +sìolaidh sìolaidh sìolaidhidh sìoladh sìolaidhte +sòr sòr sòraidh sòradh sòrte +sùgair sùgair sùgairidh sùgradh sùgairte +sùgh sùgh sùghaidh sùghadh sùghte +sùith sùith sùithidh sùitheadh sùithte +tabhainn tabhainn tabhainnidh tabhann tabhainnte +tabhair tug bheir toirt tugta +tabhannaich tabhannaich tabhannaichidh tabhannaich tabhannaichte +tachair tachair tachraidh tachairt tachairte +tachd tachd tachdaidh tachdadh tachdte +tadhail tadhail tadhailidh tadhal tadhailte +tagair tagair tagraidh tagairt tagairte +tagh tagh taghaidh taghadh taghte +taidhp taidhp taidhpidh taidhpeadh taidhpte +tairg tairg tairgidh tairgse tairgte +taisg taisg taisgidh tasgadh taisgte +taitinn taitinn taitnidh taitneadh taitinnte +talaich talaich talaichidh talach talaichte +taobh taobh taobhaidh taobhadh taobhte +taobhaich taobhaich taobhaichidh taobhachadh taobhaichte +tar-aisig tar-aisig tar-aisigidh tar-aiseag tar-aisigte +tarraing tarraing tàirnidh tarraing tarraingte +tathainn tathainn tathainnidh tathainn tathainnte +teagaisg teagaisg teagaisgidh teagasg teagaisgte +teasaich teasaich teasaichidh teasachadh teasaichte +teich teich teichidh teiche teichte +teirig teirig teirigidh teireachdainn - +teàrn teàrn teàrnaidh teàrnadh teàrnte +thig thàinig thig tighinn tigte +thoir thug bheir toirt tugta +thoir-air-falbh - - toirt-air-falbh - +tilg tilg tilgidh tilgeil tilgthe +till till tillidh tilleadh tillte +tionndaidh tionndaidh tionndaidhidh tionndadh tionndaidhte +tog tog togaidh togail togta +togair togair tograidh togradh togairte +toilich toilich toilichidh toileachadh toilichte +toinn toinn toinnidh toinneamh toinnte +toll toll tollaidh tolladh tollte +tomh tomh tomhaidh tomhadh tomhte +tomhais tomhais tomhaisidh tomhas tomhaiste +toraich toraich toraichidh torachadh toraichte +torraich torraich torraichidh torrachadh torraichte +traisg traisg traisgidh trasgadh traisgte +treabh treabh treabhaidh treabhadh treabhta +treòirich treòirich treòirichidh treòireachadh treòirichte +triall triall triallaidh triall triallta +triantanaich triantanaich triantanaichidh triantanachadh triantanaichte +troid troid troididh trod troidte +truaill truaill truaillidh truailleadh truaillte +truis truis truisidh truiseadh truiste +trus trus trusaidh trusadh trusta +tràigh tràigh tràighidh tràghadh tràighte +trèan trèan trèanaidh trèanadh trèante +trèig trèig trèigidh trèigsinn trèigte +tréig tréig tréigidh tréigsinn tréigte +tugh tugh tughaidh tughadh tughta +tuig tuig tuigidh tuigsinn tuigte +tuinich tuinich tuinichidh tuineachadh tuinichte +tuisealaich tuisealaich tuisealaichidh tuisealachadh tuisealaichte +tuit tuit tuitidh tuiteam tuite +tum tum tumaidh tumadh tumta +turchair turchair turchairidh turchairt - +tàlaidh tàlaidh tàlaidhidh tàladh tàlaidhte +tàmh tàmh tàmhaidh tàmh tàmhte +tèarainn tèarainn tèarnaidh tèarnadh tèarainte +tòisich tòisich tòisichidh tòiseachadh tòisichte +ubagaich ubagaich ubagaichidh ubagachadh ubagaichte +uchd-mhacaich uchd-mhacaich uchd-mhacaichidh uchd-mhacachadh uchd-mhacaichte +uidheamaich uidheamaich uidheamaichidh uidheamachadh uidheamaichte +uillnich uillnich uillnichidh uillneachadh uillnichte +ullaich ullaich ullaichidh ullachadh ullaichte +urramaich urramaich urramaichidh urramachadh urramaichte +watchaig - - watchaigeadh - +àicheidh àicheidh àicheidhidh àicheadh àicheidhte +àireamh àireamh àireamhaidh àireamh àireamhta +àithn àithn àithnidh àithneadh àithnte +àitich àitich àitichidh àiteach àitichte +àraich àraich àraichidh àrach àraichte +àrdaich àrdaich àrdaichidh àrdachadh àrdaichte +èalaidh èalaidh èalaidhidh èaladh èalaidhte +èigh èigh èighidh èigheachd èighte +èighich èighich èighichidh èigheach èighichte +èignich èignich èignichidh èigneachadh èignichte +èirich èirich èiridh èirigh èirichte +èisd èisd èisdidh èisdeachd èisdte +èist èist èistidh èisteachd èiste +éirich éirich éiridh éirigh éirichte +éisd éisd éisdidh éisdeachd éisdte +éist éist éistidh éisteachd éiste +òl òl òlaidh òl òlta +òrdaich òrdaich òrdaichidh òrdachadh òrdaichte +ùrlaraich ùrlaraich ùrlaraichidh ùrlarachadh ùrlaraichte diff --git a/data/grn/overrides.tsv b/data/grn/overrides.tsv new file mode 100644 index 00000000..25f7a223 --- /dev/null +++ b/data/grn/overrides.tsv @@ -0,0 +1,21 @@ +# Guarani per-cell overrides: lemma ⇥ features ⇥ form. +# Consulted before the productive rules in src/grn.rs. +# soro (nasoral) forms its coactive causative with the nasal-mutated +# stem mo+ndoro (s→nd), not the regular mbo+soro. It is the only +# nasoral verb attested, so this nasal mutation is listed, not ruled. +soro V;COACTIVE;HORTATIVE;1PL.EXCL toromondoro +soro V;COACTIVE;HORTATIVE;1PL.INCL tañamondoro +soro V;COACTIVE;HORTATIVE;1SG tamondoro +soro V;COACTIVE;HORTATIVE;2PL tapemondoro +soro V;COACTIVE;HORTATIVE;2SG teremondoro +soro V;COACTIVE;HORTATIVE;3PL tomondoro +soro V;COACTIVE;HORTATIVE;3SG tomondoro +soro V;COACTIVE;IMPERATIVE;2PL pemondoro +soro V;COACTIVE;IMPERATIVE;2SG emondoro +soro V;COACTIVE;INDICATIVE;1PL.EXCL romondoro +soro V;COACTIVE;INDICATIVE;1PL.INCL ñamondoro +soro V;COACTIVE;INDICATIVE;1SG amondoro +soro V;COACTIVE;INDICATIVE;2PL pemondoro +soro V;COACTIVE;INDICATIVE;2SG remondoro +soro V;COACTIVE;INDICATIVE;3PL omondoro +soro V;COACTIVE;INDICATIVE;3SG omondoro diff --git a/data/grn/parts.tsv b/data/grn/parts.tsv new file mode 100644 index 00000000..7e870be6 --- /dev/null +++ b/data/grn/parts.tsv @@ -0,0 +1,73 @@ +# Guarani principal parts: lemma ⇥ conjugation class. +# The class is read verbatim from the kaikki gug-conj-* inflection +# template (see scripts/grn/build.py); the engine in src/grn.rs +# derives the whole paradigm from it. gn-conj-* is folded into gug-. +lemma class +'a h +'u h +guata areal-oral +gueraha areal-oral +g̃uahe areal-nasal +hai areal-oral +hecha areal-oral +hechaga'u areal-oral +heka areal-oral +hekuavo areal-oral +hendu areal-nasal +hesape areal-oral +hupi areal-oral +hupyty areal-oral +ime areal-nasal +jaho'i areal-oral +jajái areal-oral +japo areal-oral +jehekýi areal-oral +jehu areal-oral +jepy'amongeta areal-oral +jeroky areal-oral +jerure areal-oral +jogua areal-oral +johéi areal-oral +jopy areal-oral +juka areal-oral +jupi areal-oral +juvy areal-oral +karu areal-oral +kañy areal-nasal +ke aireal-oral +kuaa aireal-oral +kuave'ẽ areal-oral +kytĩ aireal-nasal +ma'ẽ areal-nasal +mano areal-nasal +mbo'e areal-nasal +mbohovái areal-nasal +mbohéra areal-nasal +mboja'o areal-nasal +mbopiro'y areal-nasal +mbopu areal-nasal +mboty areal-nasal +mbotyryry areal-nasal +mbyepoti areal-nasal +monde areal-nasal +muña areal-nasal +myenyhẽ areal-nasal +pe'a aireal-oral +poi areal-oral +porandu areal-oral +puka areal-oral +purahéi areal-oral +puru aireal-oral +pytyvõ areal-oral +rambosa areal-nasal +reko areal-oral +soro areal-nasoral +sãmbyhy areal-nasal +timbo areal-nasal +veve areal-oral +y'u h +ñani areal-nasal +ñapytĩ areal-nasal +ñe'ẽ areal-nasal +ñembo'y areal-nasal +ñembosarái areal-nasal diff --git a/data/haw/overrides.tsv b/data/haw/overrides.tsv new file mode 100644 index 00000000..3b2344f6 --- /dev/null +++ b/data/haw/overrides.tsv @@ -0,0 +1,24 @@ +# Hawaiian derivational overrides: lemma ⇥ feature ⇥ form. +# Lexicalised forms the productive rules in src/haw.rs do not +# predict — causative+reduplication fusions, prefix-triggered +# vowel lengthening, ʻokina/long-vowel edge cases. Mined by +# scripts/haw/mine_overrides.py from the kaikki oracle. +hele V;CAUS hoʻohelehele +hoʻi V;CAUS hoʻihoʻi +hoʻokolo V;CAUS hoʻokolohua +hoʻokolo V;CAUS hoʻokolokolo +hoʻoluli V;CAUS hoʻoluliluli +kāhili V;CAUS hoʻokahili +lōʻihi V;CAUS hoʻoloʻihi +mahele V;CAUS hoʻomāhele +malū V;CAUS hoʻomālū +pio V;CAUS hoʻopiopio +pono V;CAUS hoʻoponopono +ā V;CAUS hōʻā +ʻaʻā V;CAUS hoʻaʻā +ʻāhewa V;CAUS hoʻāhewa +ʻēʻē V;CAUS hoʻēʻē +ʻō V;CAUS hoʻō +ʻōiwi V;CAUS hoʻōiwi +ʻōlapa V;CAUS hoʻōlapa +ʻōnaha V;CAUS hoʻōnaha diff --git a/data/haw/parts.tsv b/data/haw/parts.tsv new file mode 100644 index 00000000..bb490042 --- /dev/null +++ b/data/haw/parts.tsv @@ -0,0 +1,254 @@ +# Hawaiian verb-lemma inventory (col 1): the reverse-lookup +# lexicon. Mined from the kaikki verb oracle. +aka +akaaka +akua +ala +ana +ani +ea +emi +eʻe +hahai +haka +hakakā +hala +hamo +hanini +hapa +hau +haukaʻe +haʻa +haʻahaʻa +haʻalele +haʻihaʻi +hea +heheʻe +hele +hemahema +hewa +heʻe +hili +hina +hinu +hio +hiu +hiʻi +hoe +holo +holoholo +hoʻi +hoʻokolo +hoʻoluli +hula +huli +hunāhunā +hāliu +hū +hūnā +ikaika +ili +kaha +kahakaha +kahi +kala +kali +kau +kaumaha +kaʻawale +kaʻi +kaʻo +kaʻukaʻu +kea +keha +kepa +keu +keʻa +kiawe +kiaʻi +kiko +kikī +kila +kino +kipi +kiʻekiʻe +kiʻi +koho +kokoke +kolo +kolohe +komo +kono +konā +koʻo +kuene +kuhi +kukule +kulu +kupu +kuʻi +kuʻu +kāhili +kāhāhā +kē +kī +kīkī +kō +kū +kūʻonoʻono +laka +laukōnā +laulā +lawe +laʻa +lelele +lemu +leo +lewa +leʻa +limalima +lohe +loli +luaʻi +luhi +luli +luʻu +lōʻihi +mahele +makala +makanahele +makemake +mala +malolo +malū +mamao +maʻa +maʻalahi +maʻemaʻe +mihi +miki +miko +mikomiko +mio +moe +momona +mona +moni +moʻa +muku +māino +mākaukau +mākaʻikaʻi +mākonā +mākū +mālie +māloʻeloʻe +mālō +māmā +māuna +māʻeʻele +māʻona +naku +nakulu +nalo +nauki +naʻanaʻa +niu +noho +noi +noʻonoʻo +nā +nākuʻi +oki +ola +olo +pae +paheʻe +pahu +paio +pala +pale +palekana +pana +pani +panoa +pau +pauaho +paʻapū +paʻi +pelu +peʻa +pili +pilikia +pio +piʻi +poina +poni +pono +puka +pule +pumehana +puni +punihei +pākuʻi +pāpaʻa +pāʻani +pīhoihoi +pīʻao +pōhae +pōʻai +pūhili +pūnono +pūpū +uaua +uhi +ukiuki +uli +uluhua +ulukū +wae +wahine +wawā +wiki +wā +wīwī +ā +ākea +ū +ʻae +ʻai +ʻaiā +ʻakaʻaka +ʻaki +ʻale +ʻaleʻale +ʻalo +ʻaloʻalo +ʻalu +ʻaluʻalu +ʻau +ʻaui +ʻauʻau +ʻaʻā +ʻeleu +ʻike +ʻino +ʻinoʻino +ʻoi +ʻona +ʻonaʻona +ʻonipaʻa +ʻono +ʻopi +ʻoʻoleʻa +ʻume +ʻuʻuku +ʻāhewa +ʻāʻā +ʻē +ʻēʻē +ʻō +ʻōiwi +ʻōlapa +ʻōnaha +ʻōpā diff --git a/docs/epo/adjudications.tsv b/docs/epo/adjudications.tsv new file mode 100644 index 00000000..e23b7af1 --- /dev/null +++ b/docs/epo/adjudications.tsv @@ -0,0 +1,2 @@ +# Adjudications (lemma features ruling): none needed — the engine +# reproduces every kaikki cell by rule. diff --git a/docs/gla/adjudications.tsv b/docs/gla/adjudications.tsv new file mode 100644 index 00000000..47687450 --- /dev/null +++ b/docs/gla/adjudications.tsv @@ -0,0 +1 @@ +# lemma features ruling diff --git a/docs/gla/oracles.md b/docs/gla/oracles.md new file mode 100644 index 00000000..dc48c225 --- /dev/null +++ b/docs/gla/oracles.md @@ -0,0 +1,52 @@ +# Scottish Gaelic gold-data oracle + +**kaikki.org Scottish Gaelic** (en.wiktionary via Wiktextract; CC BY-SA): +`scripts/gla/fetch_kaikki.sh` downloads the verb dump and +`scripts/gla/kaikki_to_tsv.py` derives two files: + +- `data/gla/kaikki.tsv` — the golden gold (lemma ⇥ form ⇥ features). +- `data/gla/verbs.tsv` — mined principal parts (lemma, past, future, + verbal noun, verbal adjective) that the engine embeds. + +Single oracle ⇒ **Beta** tier. 1,402 verbs; 794 contribute at least one +scored form (164 have full `{{gd-conj}}` tables, the rest carry +principal parts only). + +## Normalization + +kaikki prints forms with their initial mutations and particles baked in +(`ghlan`, `chuir`, `dh'òl`, `dh'fhosgail`). The past, conditional and +relative future are lenited in the independent column; the converter +de-lenites them to the unmutated citation form (the shared oracle +convention), and the engine's `lenite` re-applies the display mutation. +Future, imperative and non-finite forms are already citation forms and +are left untouched. De-lenition is lemma-aware so roots that genuinely +begin consonant+`h` (`bhòt`, `thig`) are not mis-stripped. + +Rows tagged `error-unrecognized-form` (the relative future) are kept; +`table-tags`/`inflection-template` noise, multi-word analytic forms +(`chuireadh sinn`), the emphatic/negative/interrogative columns and the +periphrastic present are dropped. + +## Schema + +`V;VN` (verbal noun), `V.PTCP` (verbal adjective), and +`V;{PST,FUT,COND,IMP};{IND,DEP,IMPRS,1SG,1PL,2SG,2PL,3,REL}` — the +synthetic slots. Scottish Gaelic has **no synthetic present** (the +present/progressive is periphrastic with *bi*) so no present row is +generated. + +## Out of scope + +The copula / substantive verb *bi*/*is* and their impersonal/negative +paradigm lemmas (`thathar`, `nach`, …) are wholly periphrastic and are +excluded from the gold (`EXCLUDE` in the converter). *thoir* is the one +suppletive verb whose conditional stem (*bheir-*) is supplied through +`data/gla/parts.tsv`; the remaining suppletives (*rach, thig, abair, +faigh, dèan, beir, cluinn, ruig, can*) have only principal parts, which +are mined directly. + +## Result + +- 5,396 scored forms, **100.00%**; lemma coverage 794/794 (100.00%). +- Gate: `min_form_pct` 99.8, `min_lemma_coverage_pct` 99.0. diff --git a/docs/grn/adjudications.tsv b/docs/grn/adjudications.tsv new file mode 100644 index 00000000..9af0364d --- /dev/null +++ b/docs/grn/adjudications.tsv @@ -0,0 +1 @@ +# lemma features chosen note diff --git a/docs/haw/adjudications.tsv b/docs/haw/adjudications.tsv new file mode 100644 index 00000000..9af0364d --- /dev/null +++ b/docs/haw/adjudications.tsv @@ -0,0 +1 @@ +# lemma features chosen note diff --git a/docs/nav/oracles.md b/docs/nav/oracles.md new file mode 100644 index 00000000..6cdefa70 --- /dev/null +++ b/docs/nav/oracles.md @@ -0,0 +1,35 @@ +# Navajo (nav): parked — templatic polysynthesis, no productive path + +UniMorph `nav` is actually clean: 10,544 verb triples over **493 verbs** in a +uniform subject-agreement × mode/aspect grid (5 modes × person 1/2/3/4 × +number, up to 50 cells). The blocker is not data — it is **architecture**. + +A quantitative build attempt tested every productive route ablaut's +"principal parts + rules" contract relies on, over the full corpus: + +| Productive rule | Best-case coverage | +|---|---| +| Subject agreement (3sg → 1sg within a mode) | 13.1% (137 distinct swap rules) | +| Distributive plural (`da-`) | 25.0% | +| 4th person from 3rd (`ji-`/`j-`) | 19.8% | +| Iterative from imperfective/perfective (`ná-`) | 4.0% / 0.0% | +| Optative from imperfective | 6.0% | +| **Union of all best-case rules** | **13.9% derivable → 86.1% stored** | + +Navajo verbs fuse ~5–8 ordered prefix-complex morphemes (disjunct/thematic + +`da`-distributive + deictic 4th-person + iterative + mode conjugation-marker + +subject + classifier) plus a per-mode stem, with heavy morphophonemic fusion +(vowel contraction, tone, d-/l-effect). `da-` is an infix whose slot depends +on each verb's disjunct material (`iichįʼ` → `daʼiichįʼ`); the 3sg→1sg +transform needs 137 prefix-swaps because the classifier fuses (ł→sh for 1sg, +l→s …); and the perfective conjugation class (ø/yi/ni/si) is lexically +idiosyncratic. None of this is recoverable from the aligned surface forms +UniMorph provides. + +A correct generator is a full Athabaskan position-class FST over a +morpheme-segmented lexicon (a digitized Young & Morgan), which no available +oracle supplies and which is a different architecture from every ablaut +engine. A storage-backed table would gate ~100% only against the data it +stores — a circular, meaningless check. Beta was the ceiling anyway (kaikki +`nav` is Wiktionary-lineage, not an independent second oracle). Revisit only +alongside a bespoke FST + segmented lexicon. diff --git a/scripts/correctness.py b/scripts/correctness.py index 0470fcd6..5ffee4c4 100644 --- a/scripts/correctness.py +++ b/scripts/correctness.py @@ -37,7 +37,7 @@ "hin": "Hindi", "swa": "Swahili", "tam": "Tamil", "tel": "Telugu", "tgl": "Tagalog", "pes": "Persian", "kan": "Kannada", "guj": "Gujarati", "urd": "Urdu", "ben": "Bengali", "mar": "Marathi", - "mkd": "Macedonian", "afr": "Afrikaans", "bul": "Bulgarian", "ell": "Greek", "sqi": "Albanian", "pol": "Polish", "aze": "Azerbaijani", "uzb": "Uzbek", "tuk": "Turkmen", "bel": "Belarusian", "cym": "Welsh", "fao": "Faroese", "glg": "Galician", "kaz": "Kazakh", "lat": "Latin", "ltz": "Luxembourgish", "oci": "Occitan", "tat": "Tatar", "ydd": "Yiddish", "ara": "Arabic", "heb": "Hebrew", "amh": "Amharic", "ind": "Indonesian", "zul": "Zulu", + "mkd": "Macedonian", "afr": "Afrikaans", "bul": "Bulgarian", "ell": "Greek", "sqi": "Albanian", "pol": "Polish", "aze": "Azerbaijani", "uzb": "Uzbek", "tuk": "Turkmen", "bel": "Belarusian", "cym": "Welsh", "fao": "Faroese", "glg": "Galician", "kaz": "Kazakh", "lat": "Latin", "ltz": "Luxembourgish", "oci": "Occitan", "tat": "Tatar", "ydd": "Yiddish", "ara": "Arabic", "heb": "Hebrew", "amh": "Amharic", "ind": "Indonesian", "zul": "Zulu", "epo": "Esperanto", "gla": "Scottish Gaelic", "grn": "Guarani", "haw": "Hawaiian", } diff --git a/scripts/epo/fetch_kaikki.sh b/scripts/epo/fetch_kaikki.sh new file mode 100755 index 00000000..443f7037 --- /dev/null +++ b/scripts/epo/fetch_kaikki.sh @@ -0,0 +1,11 @@ +#!/bin/sh +# Fetch the kaikki.org (Wiktextract) Esperanto verb extraction (CC BY-SA) +# and convert it for the Esperanto golden harness. This is the single +# oracle (Beta tier): Esperanto is perfectly regular, so one clean source +# fully pins the paradigm. +set -e +mkdir -p data/epo +curl -sL "https://kaikki.org/dictionary/Esperanto/pos-verb/kaikki.org-dictionary-Esperanto-by-pos-verb.jsonl" \ + -o data/epo/kaikki-verbs.jsonl +python3 scripts/epo/kaikki_to_tsv.py data/epo/kaikki-verbs.jsonl > data/epo/kaikki.tsv +wc -l data/epo/kaikki.tsv diff --git a/scripts/epo/kaikki_to_tsv.py b/scripts/epo/kaikki_to_tsv.py new file mode 100755 index 00000000..9551c010 --- /dev/null +++ b/scripts/epo/kaikki_to_tsv.py @@ -0,0 +1,87 @@ +#!/usr/bin/env python3 +"""Convert the kaikki.org (Wiktextract) Esperanto verb extraction to the +shared `lemma ⇥ form ⇥ features` TSV. + +Esperanto is perfectly regular: every verb is cited by its `-i` +infinitive and conjugates by pure suffixation off the invariant stem +(infinitive minus `-i`). kaikki keys each entry on that infinitive and +lists the full `eo-conj` table. Only the entries that actually carry the +conjugation template are emitted (2894 of them; the rest are non-verb +homographs Wiktextract mislabels `pos: verb` — borrowings, Latin/Malay +look-alikes — that have no inflected forms and so contribute no rows). + +The kaikki tag sets are mapped onto canonical bundles the golden harness +also generates: + + present / past / future / conditional / volitive / infinitive + → V;PRS / V;PST / V;FUT / V;COND / V;VOL / V;NFIN + + participle × voice(active|passive) × tense(present|past|future) + × form(adjective|noun-from-verb|adverbial) + × [adjective/noun: number(sg|pl) × case(nom|acc)] + → V.PTCP;;;;; + V.PTCP;;;ADV + +Usage: python3 scripts/epo/kaikki_to_tsv.py data/epo/kaikki-verbs.jsonl +""" + +import json +import sys + + +def canonical(tags): + """Map a kaikki tag set to a canonical bundle, or None to skip.""" + t = set(tags) + if t & {"table-tags", "inflection-template", "alternative"}: + return None + if "participle" in t: + voice = "ACT" if "active" in t else "PASS" if "passive" in t else None + if voice is None: + return None + tense = "PST" if "past" in t else "FUT" if "future" in t else "PRS" + if "adverbial" in t: + return f"V.PTCP;{voice};{tense};ADV" + form = "N" if "noun-from-verb" in t else "ADJ" + num = "PL" if "plural" in t else "SG" + case = "ACC" if "accusative" in t else "NOM" + return f"V.PTCP;{voice};{tense};{form};{num};{case}" + # Finite / non-finite. The noisy secondary table double-tags some + # cells (infinitive+plural, volitive as imperative+past); the + # priority order below collapses them onto the right bundle, and + # because they carry the same surface form it is harmless anyway. + if "infinitive" in t: + return "V;NFIN" + if "volitive" in t or "imperative" in t: + return "V;VOL" + if "conditional" in t: + return "V;COND" + if "future" in t: + return "V;FUT" + if "past" in t: + return "V;PST" + if "present" in t: + return "V;PRS" + return None + + +def main(path): + rows = set() + for line in open(path, encoding="utf-8"): + d = json.loads(line) + lemma = d.get("word", "") + if not lemma or " " in lemma or not lemma.endswith("i"): + continue + for fm in d.get("forms", []): + form = fm.get("form", "") + if not form or " " in form or form.startswith("-"): + continue + feat = canonical(fm.get("tags", [])) + if feat: + rows.add((lemma, form, feat)) + out = sys.stdout + for lemma, form, feat in sorted(rows): + out.write(f"{lemma}\t{form}\t{feat}\n") + + +if __name__ == "__main__": + main(sys.argv[1]) diff --git a/scripts/gla/fetch_kaikki.sh b/scripts/gla/fetch_kaikki.sh new file mode 100755 index 00000000..934452ba --- /dev/null +++ b/scripts/gla/fetch_kaikki.sh @@ -0,0 +1,14 @@ +#!/bin/sh +# Fetch the kaikki.org (Wiktextract) Scottish Gaelic verb extraction +# (CC BY-SA) and convert it into the golden-harness gold +# (data/gla/kaikki.tsv) plus the mined principal parts +# (data/gla/verbs.tsv). The single kaikki oracle places Scottish Gaelic +# in the Beta tier. Forms carry their mutations/particles baked in +# (ghlan, chuir, dh'òl); the converter normalizes them to unmutated +# citation forms, exactly as the Irish pipeline does. +set -e +mkdir -p data/gla +curl -sL "https://kaikki.org/dictionary/Scottish%20Gaelic/pos-verb/kaikki.org-dictionary-ScottishGaelic-by-pos-verb.jsonl" \ + -o data/gla/kaikki.jsonl +python3 scripts/gla/kaikki_to_tsv.py data/gla/kaikki.jsonl +wc -l data/gla/kaikki.tsv diff --git a/scripts/gla/kaikki_to_tsv.py b/scripts/gla/kaikki_to_tsv.py new file mode 100644 index 00000000..11469559 --- /dev/null +++ b/scripts/gla/kaikki_to_tsv.py @@ -0,0 +1,209 @@ +#!/usr/bin/env python3 +"""Derive Scottish Gaelic gold (kaikki.tsv) and mined principal parts +(verbs.tsv) from the kaikki.org Scottish Gaelic verb dump. + +kaikki prints forms with their initial mutations and particles baked +in (ghlan, chuir, dh'òl, ag ràdh). The past, conditional and relative +future are lenited in the independent column; we de-lenite them to the +unmutated citation form (the engine's `lenite` re-applies the display +mutation), exactly as the Irish pipeline does. Future, imperative and +non-finite forms are already citation forms and are left untouched. +""" +import json +import sys +from collections import defaultdict + +SRC = sys.argv[1] if len(sys.argv) > 1 else "data/gla/kaikki.jsonl" +GOLD = "data/gla/kaikki.tsv" +VERBS = "data/gla/verbs.tsv" + +LENITABLE = set("bcdfgmpst") + +# The copula / substantive verb (*bi*, *is*) and its impersonal/negative +# paradigm rows: wholly periphrastic and out of scope (the present is +# `tha mi …`, the conditional `bhithinn`). kaikki lists them as verbs; +# they leak only a couple of *bi*-suppletive conditional cells. +EXCLUDE = {"bi", "is", "thathar", "nach", "bhathar", "robh", "rabhar"} + + +def delenite(f, lemma): + """Strip a leading dh'/d' particle and undo initial lenition. + + Lenition inserts an `h` after the stem's initial consonant; we only + strip it when the lemma's own initial is unlenited (glan → ghlan → + glan), never when the root genuinely begins consonant+h (bhòt, + thig) and the surface shares that same initial consonant.""" + for pre in ("dh'", "d'", "dh’", "d’"): + if f.startswith(pre): + f = f[len(pre):] + break + if len(f) >= 2 and f[0] in LENITABLE and f[1] == "h": + root_has_h = len(lemma) >= 2 and lemma[0] == f[0] and lemma[1] == "h" + if not root_has_h: + f = f[0] + f[2:] + return f + + +def bad(form): + return ( + not form + or form == "-" + or " " in form + or "{" in form + or "}" in form + or "\t" in form + ) + + +def classify(tags, form): + """Map a kaikki tag-set to (feature, delenite?) or None to skip. + + Returns a list of (feature, delenite) because a couple of raw tag + bundles are disambiguated by the form's own suffix (the imperative + 1pl/2pl/3 collapse onto identical tags in kaikki).""" + T = set(tags) + if T & {"inflection-template", "table-tags", "mutation", "mutation-radical"}: + return [] + # Copula / substantive-verb periphrasis and non-synthetic moods. + if T & {"present", "negative", "interrogative", "affirmative", "emphatic", + "alternative"}: + return [] + # Spurious imperative rows duplicating the non-finite forms. + if "imperative" in T and ("participle" in T or "noun-from-verb" in T): + return [] + + # Non-finite principal parts. + if T == {"noun-from-verb"}: + return [("V;VN", False)] + if T == {"participle", "past"}: + return [("V.PTCP", False)] + + # Bare principal parts (verbs without a full table). + if T == {"past"}: + return [("V;PST;IND", True)] + if T == {"future"}: + return [("V;FUT;IND", False)] + + # Past. + if T == {"independent", "indicative", "past", "personal"}: + return [("V;PST;IND", True)] + if T == {"impersonal", "independent", "indicative", "past"}: + return [("V;PST;IMPRS", True)] + if T == {"dependent", "past", "personal"}: + return [("V;PST;DEP", True)] + if T == {"dependent", "impersonal", "past"}: + return [("V;PST;IMPRS", True)] + + # Future. + if T == {"future", "independent", "indicative", "personal"}: + return [("V;FUT;IND", False)] + if T == {"future", "impersonal", "independent", "indicative"}: + return [("V;FUT;IMPRS", False)] + if T == {"dependent", "future", "personal"}: + return [("V;FUT;DEP", False)] + if T == {"dependent", "future", "impersonal"}: + return [("V;FUT;IMPRS", False)] + # Relative future (kaikki mis-tags it error-unrecognized-form). + if T == {"error-unrecognized-form", "independent", "indicative", "personal"}: + return [("V;FUT;REL", True)] + + # Conditional (independent). + if T == {"conditional", "first-person", "independent", "singular"}: + return [("V;COND;1SG", True)] + if T == {"conditional", "first-person", "independent", "plural"}: + return [("V;COND;1PL", True)] + if T == {"conditional", "impersonal", "independent"}: + return [("V;COND;IMPRS", True)] + if T == {"conditional", "error-unrecognized-form", "independent", "personal"}: + return [("V;COND;3", True)] + # Conditional (dependent) — not lenited. + if T == {"dependent", "first-person", "singular"}: + return [("V;COND;1SG", False)] + if T == {"dependent", "first-person", "plural"}: + return [("V;COND;1PL", False)] + if T == {"dependent", "impersonal"}: + return [("V;COND;IMPRS", False)] + if T == {"dependent", "error-unrecognized-form", "personal"}: + return [("V;COND;3", False)] + + # Imperative. + if T == {"first-person", "imperative", "independent", "singular"}: + return [("V;IMP;1SG", False)] + if T == {"imperative", "independent", "second-person", "singular"}: + return [("V;IMP;2SG", False)] + if T == {"imperative", "independent", "plural", "third-person"}: + return [("V;IMP;3", False)] + if T == {"dependent", "imperative", "impersonal", "independent"}: + return [("V;IMP;IMPRS", False)] + if T == {"imperative", "impersonal", "independent"}: + # 1pl (-amaid), 2pl (-aibh/-ibh) and 3 (-adh/-eadh) share tags. + if form.endswith("amaid"): + return [("V;IMP;1PL", False)] + if form.endswith("aibh") or form.endswith("ibh"): + return [("V;IMP;2PL", False)] + return [("V;IMP;3", False)] + + return [] + + +def main(): + lines = [json.loads(l) for l in open(SRC)] + # gold[lemma][feature] = set(forms) + gold = defaultdict(lambda: defaultdict(set)) + # principal parts + parts = {} + for entry in lines: + lemma = entry["word"].strip() + if bad(lemma) or "-" == lemma or lemma in EXCLUDE: + continue + pp = {} + for f in entry.get("forms", []): + form = f.get("form", "") + tags = f.get("tags", []) + for feature, dl in classify(tags, form): + surface = delenite(form, lemma) if dl else form + if bad(surface): + continue + gold[lemma][feature].add(surface) + # collect principal parts + if feature == "V;PST;IND": + pp.setdefault("past", surface) + elif feature == "V;FUT;IND": + pp.setdefault("future", surface) + elif feature == "V;VN": + pp.setdefault("vn", surface) + elif feature == "V.PTCP": + pp.setdefault("ptcp", surface) + if pp: + parts[lemma] = pp + + # Write gold. + rows = 0 + with open(GOLD, "w") as fh: + for lemma in sorted(gold): + for feature in sorted(gold[lemma]): + for form in sorted(gold[lemma][feature]): + fh.write(f"{lemma}\t{form}\t{feature}\n") + rows += 1 + + # Write mined principal parts. + with open(VERBS, "w") as fh: + fh.write("# lemma\tpast\tfuture\tvn\tptcp (mined principal parts, " + "delenited past; \"-\" = derive)\n") + for lemma in sorted(parts): + pp = parts[lemma] + fh.write("\t".join([ + lemma, + pp.get("past", "-"), + pp.get("future", "-"), + pp.get("vn", "-"), + pp.get("ptcp", "-"), + ]) + "\n") + + print(f"lemmas in gold: {len(gold)}", file=sys.stderr) + print(f"gold rows: {rows}", file=sys.stderr) + print(f"principal-part rows: {len(parts)}", file=sys.stderr) + + +if __name__ == "__main__": + main() diff --git a/scripts/grn/build.py b/scripts/grn/build.py new file mode 100644 index 00000000..c842bc85 --- /dev/null +++ b/scripts/grn/build.py @@ -0,0 +1,107 @@ +#!/usr/bin/env python3 +"""Build data/grn/{kaikki.tsv,parts.tsv} from the kaikki.org Paraguayan +Guarani dump. See src/grn.rs for the engine these gold files gate. + +Usage: python3 scripts/grn/build.py (run from repo root) +Reads data/grn/kaikki-raw.jsonl (fetch via scripts/grn/fetch_kaikki.sh). +Writes data/grn/kaikki.tsv and data/grn/parts.tsv. +""" +import json +import os +from collections import Counter, defaultdict + +ROOT = os.path.join(os.path.dirname(__file__), "..", "..") +RAW = os.path.join(ROOT, "data", "grn", "kaikki-raw.jsonl") +PRON = {"che", "nde", "ha'e", "ha'ekuéra", "ñande", "ore", "peẽ"} + + +def norm(form): + return " ".join(t for t in form.split(" ") if t not in PRON).strip() + + +def person(s): + if "first-person" in s: + if "singular" in s: + return "1SG" + if "inclusive" in s: + return "1PL.INCL" + if "exclusive" in s: + return "1PL.EXCL" + if "second-person" in s: + return "2SG" if "singular" in s else "2PL" + if "third-person" in s: + return "3SG" if "singular" in s else "3PL" + return None + + +def voice(s): + for v in ("passive", "reciprocal", "coactive", "objective"): + if v in s: + return v.upper() + return "ACT" + + +def mood(s): + for m in ("indicative", "hortative", "imperative"): + if m in s: + return m.upper() + return None + + +def cls_token(name): + name = name.replace("gn-conj-", "gug-conj-") + body = name[len("gug-conj-"):] + return "h" if body == "h" else body # areal-oral, aireal-nasal, ... + + +def main(): + gold = defaultdict(lambda: defaultdict(set)) # lemma -> feat -> {forms} + parts = {} # lemma -> class token + for line in open(RAW, encoding="utf-8"): + d = json.loads(line) + if d.get("pos") != "verb": + continue + tmpls = [t.get("name") for t in d.get("inflection_templates", [])] + if not tmpls: + continue + tok = cls_token(tmpls[0]) + if tok.startswith("stative"): + continue # 1 verb, unverified voice matrix + w = d["word"] + parts[w] = tok + for fo in d.get("forms", []): + tg = set(fo.get("tags", [])) + form = fo.get("form", "") + if {"inflection-template", "table-tags", "error-unrecognized-form"} & tg: + continue + if form in ("-", "", "hikuái", "no-table-tags"): + continue + p, m = person(tg), mood(tg) + if not p or not m: + continue + n = norm(form) + if n in ("", "hikuái"): + continue + gold[w][f"V;{voice(tg)};{m};{p}"].add(n) + + with open(os.path.join(ROOT, "data", "grn", "kaikki.tsv"), "w", encoding="utf-8") as f: + for lemma in sorted(gold): + for feat in sorted(gold[lemma]): + for form in sorted(gold[lemma][feat]): + f.write(f"{lemma}\t{form}\t{feat}\n") + + with open(os.path.join(ROOT, "data", "grn", "parts.tsv"), "w", encoding="utf-8") as f: + f.write("# Guarani principal parts: lemma ⇥ conjugation class.\n") + f.write("# The class is read verbatim from the kaikki gug-conj-* inflection\n") + f.write("# template (see scripts/grn/build.py); the engine in src/grn.rs\n") + f.write("# derives the whole paradigm from it. gn-conj-* is folded into gug-.\n") + f.write("lemma\tclass\n") + for lemma in sorted(parts): + f.write(f"{lemma}\t{parts[lemma]}\n") + + print("lemmas:", len(gold), "classes:", Counter(parts.values())) + print("gold rows:", sum(len(v) for feats in gold.values() for v in feats.values())) + + +if __name__ == "__main__": + main() diff --git a/scripts/grn/fetch_kaikki.sh b/scripts/grn/fetch_kaikki.sh new file mode 100755 index 00000000..d95f2396 --- /dev/null +++ b/scripts/grn/fetch_kaikki.sh @@ -0,0 +1,8 @@ +#!/usr/bin/env bash +# Fetch the kaikki.org Paraguayan Guarani dump used to build the gold +# files. Run from the repo root, then `python3 scripts/grn/build.py`. +set -euo pipefail +mkdir -p data/grn +URL="https://kaikki.org/dictionary/Paraguayan%20Guarani/kaikki.org-dictionary-ParaguayanGuarani.jsonl" +curl -fSL "$URL" -o data/grn/kaikki-raw.jsonl +wc -l data/grn/kaikki-raw.jsonl diff --git a/scripts/haw/fetch_kaikki.sh b/scripts/haw/fetch_kaikki.sh new file mode 100755 index 00000000..2a54df6a --- /dev/null +++ b/scripts/haw/fetch_kaikki.sh @@ -0,0 +1,13 @@ +#!/bin/sh +# Fetch the kaikki.org (Wiktextract) Hawaiian verb extraction (CC BY-SA) and +# convert it into the Hawaiian golden oracle. Hawaiian's only bound verbal +# morphology is derivational (the hoʻo- causative, full reduplication, the +# -ʻia passive); TAM is periphrastic and out of scope. The oracle is the +# lemma-linked "Derived terms" and passive `forms` Wiktionary records — the +# single independent source the engine is gated against (Beta tier). +set -e +mkdir -p data/haw +curl -sL "https://kaikki.org/dictionary/Hawaiian/pos-verb/kaikki.org-dictionary-Hawaiian-by-pos-verb.jsonl" \ + -o data/haw/kaikki-verbs.jsonl +python3 scripts/haw/kaikki_to_tsv.py data/haw/kaikki-verbs.jsonl > data/haw/kaikki.tsv +wc -l data/haw/kaikki.tsv diff --git a/scripts/haw/kaikki_to_tsv.py b/scripts/haw/kaikki_to_tsv.py new file mode 100644 index 00000000..4a81d34b --- /dev/null +++ b/scripts/haw/kaikki_to_tsv.py @@ -0,0 +1,78 @@ +#!/usr/bin/env python3 +"""Turn the kaikki.org (Wiktextract) Hawaiian verb dump into the golden +oracle TSV for the Hawaiian *derivational* engine. + +Hawaiian marks tense/aspect/mood with free preverbal particles (ua V, ke V +nei, e V ana, …) while the verb stem stays invariant — that is periphrasis, +out of scope here exactly as analytic TAM is treated elsewhere in the repo. +The only *bound* verbal morphology is derivational, and Wiktionary records it +as human-curated "Derived terms" and passive-tagged `forms` hanging off each +verb lemma. This script mines three productive, lemma-linked paradigms: + + * V;CAUS — the causative/simulative prefix hoʻo- (allomorphs hō-/hoʻā-…), + taken from every single-word derived term of shape hoʻo…/hō…; + * V;RDP — full reduplication (base+base), the productive + plural/intensive/frequentative stem; + * V;PASS — the -ʻia passive and its lexical -a/-na allomorphs, from + `forms` tagged "passive". + +Emit `lemma form features`, the format the shared golden harness +consumes. Each row is an independent Wiktionary attestation, never an engine +output — the engine is scored against these, not the reverse. +""" +import json +import sys + +CAUS_PREFIXES = ("hoʻo", "hoʻā", "hō", "hoʻ") + + +def is_word(w: str) -> bool: + return bool(w) and " " not in w + + +def main() -> None: + path = sys.argv[1] + rows = [] + with open(path, encoding="utf-8") as f: + for line in f: + line = line.strip() + if not line: + continue + d = json.loads(line) + if d.get("pos") != "verb": + continue + base = d.get("word", "") + if not is_word(base): + continue + + # V;CAUS: single-word hoʻo-/hō- derived terms. + seen = set() + for der in d.get("derived", []): + w = der.get("word", "") + if w in seen or w == base or not is_word(w): + continue + if any(w.startswith(p) for p in CAUS_PREFIXES): + rows.append((base, w, "V;CAUS")) + seen.add(w) + # V;RDP: full reduplication base+base. + if w == base + base: + rows.append((base, w, "V;RDP")) + + # V;PASS: forms tagged "passive" (the lexical -a/-na/-ʻia residue). + for fo in d.get("forms", []): + if "passive" in fo.get("tags", []): + pf = fo.get("form", "") + if is_word(pf) and pf != base: + rows.append((base, pf, "V;PASS")) + + seen = set() + for lemma, form, feat in sorted(set(rows)): + key = (lemma, form, feat) + if key in seen: + continue + seen.add(key) + print(f"{lemma}\t{form}\t{feat}") + + +if __name__ == "__main__": + main() diff --git a/scripts/haw/mine_overrides.py b/scripts/haw/mine_overrides.py new file mode 100644 index 00000000..5071ef6e --- /dev/null +++ b/scripts/haw/mine_overrides.py @@ -0,0 +1,98 @@ +#!/usr/bin/env python3 +"""Mine the Hawaiian override + principal-parts layers from the oracle. + +The engine derives the causative, reduplicated and passive stems by rule +(see src/haw.rs). A handful of derived terms are lexicalised — causatives +fused with reduplication (hele→hoʻohelehele), stems whose vowel lengthens +under prefixation (mahele→hoʻomāhele), ʻokina/long-vowel edge cases — that +no productive rule should be forced to predict. This script replays the +Rust rules in Python, and writes: + + * data/haw/overrides.tsv — lemma ⇥ feature ⇥ form, one row per oracle + form the rules miss (consulted before the rules in the engine); + * data/haw/parts.tsv — the distinct verb-lemma inventory (col 1), + the lexicon the reverse index is built from. + +Run after scripts/haw/fetch_kaikki.sh. Idempotent. +""" +import sys + +VOWELS = set("aeiou") +LONG = set("āēīōū") + + +def causative_candidates(base: str) -> list: + """Mirror of haw::causative — the candidate hoʻo- surface forms.""" + if not base: + return [] + cands = ["hoʻo" + base] + v0, rest = base[0], base[1:] + if v0 == "ʻ": + cands.append("hō" + base) + elif v0 == "a": + cands += ["hoʻā" + rest, "hō" + base] + elif v0 == "o": + cands.append("hoʻō" + rest) + elif v0 == "e": + cands += ["hoʻē" + rest, "hōʻe" + rest] + elif v0 == "i": + cands.append("hoʻī" + rest) + elif v0 == "u": + cands.append("hoʻū" + rest) + elif v0 in LONG: + cands += ["hō" + base, "hoʻ" + base] + return cands + + +def reduplication_candidates(base: str) -> list: + return [base + base] + + +def passive_candidates(base: str) -> list: + return [base + s for s in ("ʻia", "a", "na", "hia", "lia", "mia", "ʻana")] + + +CANDS = { + "V;CAUS": causative_candidates, + "V;RDP": reduplication_candidates, + "V;PASS": passive_candidates, +} + + +def main() -> None: + oracle = sys.argv[1] + lemmas = set() + overrides = [] + with open(oracle, encoding="utf-8") as f: + for line in f: + parts = line.rstrip("\n").split("\t") + if len(parts) != 3: + continue + lemma, form, feat = parts + lemmas.add(lemma) + gen = CANDS.get(feat) + if gen is None: + continue + if form not in gen(lemma): + overrides.append((lemma, feat, form)) + + with open("data/haw/overrides.tsv", "w", encoding="utf-8") as out: + out.write("# Hawaiian derivational overrides: lemma ⇥ feature ⇥ form.\n") + out.write("# Lexicalised forms the productive rules in src/haw.rs do not\n") + out.write("# predict — causative+reduplication fusions, prefix-triggered\n") + out.write("# vowel lengthening, ʻokina/long-vowel edge cases. Mined by\n") + out.write("# scripts/haw/mine_overrides.py from the kaikki oracle.\n") + for lemma, feat, form in sorted(set(overrides)): + out.write(f"{lemma}\t{feat}\t{form}\n") + + with open("data/haw/parts.tsv", "w", encoding="utf-8") as out: + out.write("# Hawaiian verb-lemma inventory (col 1): the reverse-lookup\n") + out.write("# lexicon. Mined from the kaikki verb oracle.\n") + for lemma in sorted(lemmas): + out.write(f"{lemma}\n") + + print(f"overrides: {len(set(overrides))} lemmas: {len(lemmas)}") + + +if __name__ == "__main__": + main() diff --git a/scripts/nav/fetch_unimorph.sh b/scripts/nav/fetch_unimorph.sh new file mode 100755 index 00000000..340f0370 --- /dev/null +++ b/scripts/nav/fetch_unimorph.sh @@ -0,0 +1,7 @@ +#!/usr/bin/env bash +# Fetch UniMorph Navajo and keep only the verb paradigm (V;...), the scoped oracle. +set -euo pipefail +DIR="$(cd "$(dirname "$0")/../../data/nav" && pwd)" +URL="https://raw.githubusercontent.com/unimorph/nav/master/nav" +curl -sL "$URL" | awk -F'\t' '$3 ~ /^V/' | sort -u > "$DIR/unimorph.tsv" +echo "wrote $DIR/unimorph.tsv ($(wc -l < "$DIR/unimorph.tsv") verb triples)" diff --git a/src/bin/golden_epo.rs b/src/bin/golden_epo.rs new file mode 100644 index 00000000..58a27446 --- /dev/null +++ b/src/bin/golden_epo.rs @@ -0,0 +1,56 @@ +//! Esperanto golden-test harness: diff the engine against the kaikki.org +//! (Wiktextract) Esperanto verb extraction. +//! +//! Usage: cargo run --release --bin golden_epo [gold.tsv ...] [--check] +//! (default: data/epo/kaikki.tsv — see scripts/epo/fetch_kaikki.sh) +//! +//! Esperanto is perfectly regular, so a single clean oracle (kaikki) +//! suffices (Beta tier). The engine and the oracle both build every cell +//! by pure suffixation, so the gate is a full 100.00%. + +use ablaut::epo::Verb; +use ablaut::harness::{run, Spec}; + +const CATEGORIES: [&str; 6] = [ + "infinitive", + "present", + "past", + "future", + "conditional", + "participle", +]; + +fn category(features: &str) -> &'static str { + if features.starts_with("V.PTCP") { + "participle" + } else if features == "V;NFIN" { + "infinitive" + } else if features == "V;PST" { + "past" + } else if features == "V;FUT" { + "future" + } else if features == "V;COND" { + "conditional" + } else { + // V;PRS and V;VOL + "present" + } +} + +fn main() { + run(Spec { + lang: "epo", + // Single oracle (Beta): the second path is an empty placeholder, + // so kaikki is scored directly. + default_paths: ["data/epo/kaikki.tsv", "data/epo/_no_second_oracle.tsv"], + adjudications: "docs/epo/adjudications.tsv", + mismatches: "target/golden_epo_mismatches.tsv", + categories: &CATEGORIES, + min_form_pct: 99.8, + min_lemma_coverage_pct: 99.5, + carry_features: &[], + parse: |lemma| Verb::from_infinitive(lemma).ok(), + generate: |verb, features| Some(vec![verb.form(features)?]), + category, + }); +} diff --git a/src/bin/golden_gla.rs b/src/bin/golden_gla.rs new file mode 100644 index 00000000..31042b19 --- /dev/null +++ b/src/bin/golden_gla.rs @@ -0,0 +1,79 @@ +//! Scottish Gaelic golden harness: diff the engine against the +//! kaikki.org Scottish Gaelic gold (`data/gla/kaikki.tsv`). +//! +//! Usage: cargo run --release --bin golden_gla [gold.tsv ...] [--check] +//! (default: data/gla/kaikki.tsv — see scripts/gla/kaikki_to_tsv.py) + +use ablaut::gla::{Slot, Tense, Verb}; +use ablaut::harness::{run, Spec}; + +const CATEGORIES: [&str; 6] = [ + "nonfinite", + "past", + "future", + "conditional", + "imperative", + "relative-future", +]; + +/// Map a feature bundle from the gold TSV to the engine's output +/// (None means unsupported bundle). +fn generate(verb: &Verb, features: &str) -> Option> { + let f: Vec<&str> = features.split(';').collect(); + let one = |t: Tense, s: Slot| verb.form(t, s).map(|f| vec![f]); + match f.as_slice() { + ["V", "VN"] => Some(vec![verb.verbal_noun()]), + ["V.PTCP"] => Some(vec![verb.verbal_adjective()]), + ["V", "PST", "IND"] => one(Tense::Past, Slot::Independent), + ["V", "PST", "DEP"] => one(Tense::Past, Slot::Dependent), + ["V", "PST", "IMPRS"] => one(Tense::Past, Slot::Impersonal), + ["V", "FUT", "IND"] => one(Tense::Future, Slot::Independent), + ["V", "FUT", "DEP"] => one(Tense::Future, Slot::Dependent), + ["V", "FUT", "IMPRS"] => one(Tense::Future, Slot::Impersonal), + ["V", "FUT", "REL"] => one(Tense::RelativeFuture, Slot::Independent), + ["V", "COND", "3"] => one(Tense::Conditional, Slot::Third), + ["V", "COND", "1SG"] => one(Tense::Conditional, Slot::FirstSingular), + ["V", "COND", "1PL"] => one(Tense::Conditional, Slot::FirstPlural), + ["V", "COND", "IMPRS"] => one(Tense::Conditional, Slot::Impersonal), + ["V", "IMP", "2SG"] => one(Tense::Imperative, Slot::SecondSingular), + ["V", "IMP", "1SG"] => one(Tense::Imperative, Slot::FirstSingular), + ["V", "IMP", "1PL"] => one(Tense::Imperative, Slot::FirstPlural), + ["V", "IMP", "2PL"] => one(Tense::Imperative, Slot::SecondPlural), + ["V", "IMP", "3"] => one(Tense::Imperative, Slot::Third), + ["V", "IMP", "IMPRS"] => one(Tense::Imperative, Slot::Impersonal), + _ => None, + } +} + +/// Coarse category for the per-slot breakdown. +fn category(features: &str) -> &'static str { + if features == "V;VN" || features == "V.PTCP" { + "nonfinite" + } else if features.starts_with("V;PST") { + "past" + } else if features == "V;FUT;REL" { + "relative-future" + } else if features.starts_with("V;FUT") { + "future" + } else if features.starts_with("V;COND") { + "conditional" + } else { + "imperative" + } +} + +fn main() { + run(Spec { + lang: "gla", + default_paths: ["data/gla/kaikki.tsv", "data/gla/.no-second-oracle"], + adjudications: "docs/gla/adjudications.tsv", + mismatches: "target/golden_gla_mismatches.tsv", + categories: &CATEGORIES, + min_form_pct: 99.8, + min_lemma_coverage_pct: 99.0, + carry_features: &[], + parse: |lemma| Verb::from_infinitive(lemma).ok(), + generate, + category, + }); +} diff --git a/src/bin/golden_grn.rs b/src/bin/golden_grn.rs new file mode 100644 index 00000000..6302716c --- /dev/null +++ b/src/bin/golden_grn.rs @@ -0,0 +1,59 @@ +//! Guaraní golden-test harness: diff the engine against the kaikki.org +//! Paraguayan Guaraní conjugation tables. +//! +//! Usage: cargo run --release --bin golden_grn [gold.tsv] [--check] +//! (default: data/grn/kaikki.tsv — see scripts/grn/fetch_kaikki.sh +//! and scripts/grn/build.py) +//! +//! Single-oracle Beta gate. The gold is every cleanly-tagged cell of the +//! 68 verbs carrying a `gug-conj-*` template — active plus the passive, +//! reciprocal, coactive and objective voices, across the indicative, +//! hortative and imperative, stripped of the optional subject pronoun. +//! Excluded and documented: the `error-unrecognized-form` cells kaikki +//! itself flags (poro-/mba'e- incorporations) and the single stative +//! (chendal) verb, whose possessive-marked paradigm is unverifiable from +//! one attestation. + +use ablaut::grn::Verb; +use ablaut::harness::{run, Spec}; + +const CATEGORIES: [&str; 6] = [ + "active", + "passive", + "reciprocal", + "coactive", + "objective", + "other", +]; + +fn category(features: &str) -> &'static str { + if features.starts_with("V;ACT;") { + "active" + } else if features.starts_with("V;PASSIVE;") { + "passive" + } else if features.starts_with("V;RECIPROCAL;") { + "reciprocal" + } else if features.starts_with("V;COACTIVE;") { + "coactive" + } else if features.starts_with("V;OBJECTIVE;") { + "objective" + } else { + "other" + } +} + +fn main() { + run(Spec { + lang: "grn", + default_paths: ["data/grn/kaikki.tsv", ""], + adjudications: "docs/grn/adjudications.tsv", + mismatches: "target/golden_grn_mismatches.tsv", + categories: &CATEGORIES, + min_form_pct: 99.8, + min_lemma_coverage_pct: 99.5, + carry_features: &[], + parse: |lemma| Verb::from_lemma(lemma).ok(), + generate: |verb, features| verb.generate(features), + category, + }); +} diff --git a/src/bin/golden_haw.rs b/src/bin/golden_haw.rs new file mode 100644 index 00000000..0b193a23 --- /dev/null +++ b/src/bin/golden_haw.rs @@ -0,0 +1,55 @@ +//! Hawaiian golden-test harness: diff the engine against the single +//! Hawaiian oracle (kaikki.org Hawaiian — the lemma-linked "Derived terms" +//! and passive `forms` from Wiktionary) — Beta tier. Hawaiian marks TAM with +//! free particles (periphrasis, out of scope); the bound morphology is +//! derivational, so the scored slots are the causative (hoʻo-), full +//! reduplication and the -ʻia passive. There is no second oracle (UniMorph +//! has no `haw`), so kaikki is scored directly. +//! +//! Usage: cargo run --release --bin golden_haw [gold.tsv ...] [--check] +//! (default: data/haw/kaikki.tsv) + +use ablaut::harness::{run, Spec}; +use ablaut::haw::Verb; + +const CATEGORIES: [&str; 3] = ["causative", "reduplicated", "passive"]; + +fn category(features: &str) -> &'static str { + let has = |t: &str| features.split(';').any(|x| x == t); + if has("CAUS") { + "causative" + } else if has("RDP") { + "reduplicated" + } else if has("PASS") { + "passive" + } else { + "causative" + } +} + +fn generate(verb: &Verb, features: &str) -> Option> { + let forms = verb.forms(features); + if forms.is_empty() { + None + } else { + Some(forms) + } +} + +fn main() { + run(Spec { + lang: "haw", + // Single oracle (Beta): the second path is an empty placeholder, so + // kaikki is scored directly. + default_paths: ["data/haw/kaikki.tsv", "data/haw/_no_second_oracle.tsv"], + adjudications: "docs/haw/adjudications.tsv", + mismatches: "target/golden_haw_mismatches.tsv", + categories: &CATEGORIES, + min_form_pct: 99.5, + min_lemma_coverage_pct: 99.5, + carry_features: &[], + parse: |lemma| Verb::from_lemma(lemma).ok(), + generate, + category, + }); +} diff --git a/src/epo.rs b/src/epo.rs new file mode 100644 index 00000000..b10075d6 --- /dev/null +++ b/src/epo.rs @@ -0,0 +1,324 @@ +//! Esperanto conjugation. Esperanto is the regular-conjugation limit: +//! every verb is cited by its `-i` infinitive, and the whole paradigm is +//! pure suffixation off the invariant stem (the infinitive minus `-i`). +//! There is not a single irregular verb — even `esti` "to be" is regular +//! (`estas`, `estis`, `estos`, `estus`, `estu`). So there is nothing to +//! mine: `data/epo/parts.tsv` and `data/epo/overrides.tsv` stay empty and +//! every scored form is rule-derived. +//! +//! The finite/non-finite endings are: `-as` present, `-is` past, `-os` +//! future, `-us` conditional, `-u` volitive (jussive/imperative), `-i` +//! infinitive. +//! +//! The participles are a clean cross-product: voice (active `-ant/-int/ +//! -ont-`, passive `-at/-it/-ot-`) × tense (present/past/future) × the +//! closing category — adjective `-a`, noun `-o`, adverb `-e`. The +//! adjectival and nominal forms further inflect for number (`-j`) and the +//! accusative (`-n`): amanta, amantaj, amantan, amantajn. That gives +//! 6 finite/non-finite + 6 participle stems × 9 closings = 60 forms, all +//! generated here without exception. + +/// A finite / non-finite slot. +#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash)] +pub enum Slot { + Present, + Past, + Future, + Conditional, + Volitive, + Infinitive, +} + +/// The voice of a participle. +#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash)] +pub enum Voice { + Active, + Passive, +} + +/// The tense of a participle. +#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash)] +pub enum Tense { + Present, + Past, + Future, +} + +/// Why an input cannot be conjugated as an Esperanto verb. +#[derive(Debug, Clone, PartialEq, Eq)] +pub enum Error { + /// The input does not look like an Esperanto `-i` infinitive. + NotAVerb, +} + +impl std::fmt::Display for Error { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + write!(f, "not an Esperanto infinitive") + } +} + +/// The participle stem for a voice/tense pair: active `-ant/-int/-ont-`, +/// passive `-at/-it/-ot-`. +fn participle_infix(voice: Voice, tense: Tense) -> &'static str { + match (voice, tense) { + (Voice::Active, Tense::Present) => "ant", + (Voice::Active, Tense::Past) => "int", + (Voice::Active, Tense::Future) => "ont", + (Voice::Passive, Tense::Present) => "at", + (Voice::Passive, Tense::Past) => "it", + (Voice::Passive, Tense::Future) => "ot", + } +} + +/// The nine closings of a participle stem, in a fixed order: the four +/// adjectival (`-a/-aj/-an/-ajn`), the four nominal (`-o/-oj/-on/-ojn`) +/// and the invariant adverb (`-e`). +const CLOSINGS: [&str; 9] = ["a", "aj", "an", "ajn", "o", "oj", "on", "ojn", "e"]; + +/// A conjugatable Esperanto verb: the infinitive and its invariant stem. +#[derive(Debug, Clone)] +pub struct Verb { + infinitive: String, + stem: String, +} + +impl Verb { + /// Build a verb from its `-i` infinitive. The infinitive marker is the + /// citation form itself, so no stripping is needed; the trailing `-i` + /// is removed to expose the invariant stem. + pub fn from_infinitive(infinitive: &str) -> Result { + let inf = infinitive.trim().to_lowercase(); + let Some(stem) = inf.strip_suffix('i') else { + return Err(Error::NotAVerb); + }; + if stem.is_empty() + || inf.contains(char::is_whitespace) + || !inf.chars().all(|c| c.is_alphabetic() || c == '-') + { + return Err(Error::NotAVerb); + } + Ok(Self { + infinitive: inf.clone(), + stem: stem.to_string(), + }) + } + + /// The infinitive (citation form). + #[must_use] + pub fn infinitive(&self) -> &str { + &self.infinitive + } + + /// The invariant stem (infinitive minus `-i`). + #[must_use] + pub fn stem(&self) -> &str { + &self.stem + } + + /// A finite / non-finite form. + #[must_use] + pub fn finite(&self, slot: Slot) -> String { + match slot { + Slot::Present => format!("{}as", self.stem), + Slot::Past => format!("{}is", self.stem), + Slot::Future => format!("{}os", self.stem), + Slot::Conditional => format!("{}us", self.stem), + Slot::Volitive => format!("{}u", self.stem), + Slot::Infinitive => self.infinitive.clone(), + } + } + + /// One participle form, addressed by voice, tense and a closing index + /// into [`CLOSINGS`] (0..9). + #[must_use] + fn participle_at(&self, voice: Voice, tense: Tense, closing: usize) -> String { + format!( + "{}{}{}", + self.stem, + participle_infix(voice, tense), + CLOSINGS[closing] + ) + } + + /// The nine forms of one participle stem, in [`CLOSINGS`] order. + #[must_use] + fn participles(&self, voice: Voice, tense: Tense) -> Vec { + (0..CLOSINGS.len()) + .map(|i| self.participle_at(voice, tense, i)) + .collect() + } + + /// The surface form for a canonical feature bundle, or `None` if the + /// bundle is not a slot of the paradigm. The bundle vocabulary is the + /// one the golden harness scores against (see `src/bin/golden_epo.rs`). + #[must_use] + pub fn form(&self, feature: &str) -> Option { + match feature { + "V;PRS" => Some(self.finite(Slot::Present)), + "V;PST" => Some(self.finite(Slot::Past)), + "V;FUT" => Some(self.finite(Slot::Future)), + "V;COND" => Some(self.finite(Slot::Conditional)), + "V;VOL" => Some(self.finite(Slot::Volitive)), + "V;NFIN" => Some(self.finite(Slot::Infinitive)), + _ => self.participle_form(feature), + } + } + + /// Parse a `V.PTCP;;;[;;]` bundle + /// and generate its form. + fn participle_form(&self, feature: &str) -> Option { + let rest = feature.strip_prefix("V.PTCP;")?; + let mut it = rest.split(';'); + let voice = match it.next()? { + "ACT" => Voice::Active, + "PASS" => Voice::Passive, + _ => return None, + }; + let tense = match it.next()? { + "PRS" => Tense::Present, + "PST" => Tense::Past, + "FUT" => Tense::Future, + _ => return None, + }; + let closing = match it.next()? { + "ADV" => "e", + cat @ ("ADJ" | "N") => { + let vowel = if cat == "ADJ" { "a" } else { "o" }; + let plural = matches!(it.next()?, "PL"); + let accusative = matches!(it.next()?, "ACC"); + return Some(format!( + "{}{}{}{}{}", + self.stem, + participle_infix(voice, tense), + vowel, + if plural { "j" } else { "" }, + if accusative { "n" } else { "" }, + )); + } + _ => return None, + }; + Some(format!( + "{}{}{}", + self.stem, + participle_infix(voice, tense), + closing + )) + } +} + +/// The full conjugation table of an Esperanto verb — shared by the +/// WebAssembly and Python bindings. The finite/non-finite cells are +/// scalars; each participle group carries its nine closings in +/// [`CLOSINGS`] order (adjective nom/acc × sg/pl, noun nom/acc × sg/pl, +/// adverb). +#[cfg_attr(feature = "serde", derive(serde::Serialize))] +#[cfg_attr(feature = "serde", serde(rename_all = "camelCase"))] +pub struct Table { + pub infinitive: String, + pub present: String, + pub past: String, + pub future: String, + pub conditional: String, + pub volitive: String, + pub active_present: Vec, + pub active_past: Vec, + pub active_future: Vec, + pub passive_present: Vec, + pub passive_past: Vec, + pub passive_future: Vec, +} + +impl Table { + #[must_use] + pub fn build(v: &Verb) -> Self { + Self { + infinitive: v.infinitive().to_string(), + present: v.finite(Slot::Present), + past: v.finite(Slot::Past), + future: v.finite(Slot::Future), + conditional: v.finite(Slot::Conditional), + volitive: v.finite(Slot::Volitive), + active_present: v.participles(Voice::Active, Tense::Present), + active_past: v.participles(Voice::Active, Tense::Past), + active_future: v.participles(Voice::Active, Tense::Future), + passive_present: v.participles(Voice::Passive, Tense::Present), + passive_past: v.participles(Voice::Passive, Tense::Past), + passive_future: v.participles(Voice::Passive, Tense::Future), + } + } +} + +#[cfg(test)] +mod tests { + use super::*; + + fn v(inf: &str) -> Verb { + Verb::from_infinitive(inf).unwrap() + } + + #[test] + fn finite_endings() { + let a = v("ami"); + assert_eq!(a.finite(Slot::Present), "amas"); + assert_eq!(a.finite(Slot::Past), "amis"); + assert_eq!(a.finite(Slot::Future), "amos"); + assert_eq!(a.finite(Slot::Conditional), "amus"); + assert_eq!(a.finite(Slot::Volitive), "amu"); + assert_eq!(a.finite(Slot::Infinitive), "ami"); + } + + #[test] + fn esti_is_regular() { + let e = v("esti"); + assert_eq!(e.finite(Slot::Present), "estas"); + assert_eq!(e.finite(Slot::Past), "estis"); + assert_eq!(e.finite(Slot::Future), "estos"); + assert_eq!(e.finite(Slot::Volitive), "estu"); + } + + #[test] + fn active_participles() { + let a = v("ami"); + assert_eq!(a.form("V.PTCP;ACT;PRS;ADJ;SG;NOM").unwrap(), "amanta"); + assert_eq!(a.form("V.PTCP;ACT;PRS;ADJ;PL;NOM").unwrap(), "amantaj"); + assert_eq!(a.form("V.PTCP;ACT;PRS;ADJ;SG;ACC").unwrap(), "amantan"); + assert_eq!(a.form("V.PTCP;ACT;PRS;ADJ;PL;ACC").unwrap(), "amantajn"); + assert_eq!(a.form("V.PTCP;ACT;PST;ADJ;SG;NOM").unwrap(), "aminta"); + assert_eq!(a.form("V.PTCP;ACT;FUT;ADJ;SG;NOM").unwrap(), "amonta"); + assert_eq!(a.form("V.PTCP;ACT;PRS;N;SG;NOM").unwrap(), "amanto"); + assert_eq!(a.form("V.PTCP;ACT;PRS;ADV").unwrap(), "amante"); + } + + #[test] + fn passive_participles() { + let a = v("ami"); + assert_eq!(a.form("V.PTCP;PASS;PRS;ADJ;SG;NOM").unwrap(), "amata"); + assert_eq!(a.form("V.PTCP;PASS;PST;ADJ;SG;NOM").unwrap(), "amita"); + assert_eq!(a.form("V.PTCP;PASS;FUT;ADJ;SG;NOM").unwrap(), "amota"); + assert_eq!(a.form("V.PTCP;PASS;PRS;N;PL;ACC").unwrap(), "amatojn"); + assert_eq!(a.form("V.PTCP;PASS;PST;ADV").unwrap(), "amite"); + } + + #[test] + fn table_groups() { + let t = Table::build(&v("ami")); + assert_eq!(t.present, "amas"); + // CLOSINGS order: a, aj, an, ajn, o, oj, on, ojn, e + assert_eq!( + t.active_present, + vec![ + "amanta", "amantaj", "amantan", "amantajn", "amanto", "amantoj", "amanton", + "amantojn", "amante" + ] + ); + assert_eq!(t.passive_future[0], "amota"); + } + + #[test] + fn non_verb_rejected() { + assert!(Verb::from_infinitive("").is_err()); + assert!(Verb::from_infinitive("amo").is_err()); // not an -i infinitive + assert!(Verb::from_infinitive("i").is_err()); // empty stem + assert!(Verb::from_infinitive("du vortoj").is_err()); + } +} diff --git a/src/gla.rs b/src/gla.rs new file mode 100644 index 00000000..641e50c4 --- /dev/null +++ b/src/gla.rs @@ -0,0 +1,406 @@ +//! Scottish Gaelic conjugation. Goidelic sibling of Irish, but with +//! no synthetic present (the present/progressive is periphrastic with +//! *bi* and out of scope). The synthetic core is derived off the +//! broad/slender shape of the imperative stem (= the lemma): future +//! `-aidh/-idh`, relative future `-as/-eas`, past (lenition only), +//! conditional/past-habitual `-adh/-eadh` with synthetic 1sg/1pl, the +//! imperative persons, and the impersonal/passive. The four lexical +//! principal parts — past, future, verbal noun, verbal adjective — +//! are mined in `data/gla/verbs.tsv`; a handful of suppletive verbs +//! are overridden from `data/gla/parts.tsv`. Forms are unmutated +//! citation forms (the shared oracle convention); `lenite` applies the +//! display mutation (glan → ghlan, òl → dh'òl). + +use std::collections::HashMap; +use std::sync::OnceLock; + +/// The synthetic tenses/moods Scottish Gaelic inflects. +#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash)] +pub enum Tense { + Past, + Future, + /// The relative future (`a ghlanas`), a distinct synthetic form. + RelativeFuture, + /// The conditional / past-habitual (`ghlanadh`). + Conditional, + Imperative, +} + +/// The person/voice slots. Scottish Gaelic inflects synthetically only +/// a few persons; the rest are analytic (verb + pronoun). +#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash)] +pub enum Slot { + /// The independent base form used with pronouns (glanaidh mi). + Independent, + /// The dependent form after a particle (cha ghlan, an glan). + Dependent, + FirstSingular, + SecondSingular, + FirstPlural, + SecondPlural, + /// The third-person / analytic base of the conditional and + /// imperative (glanadh). + Third, + /// The impersonal / passive form (glanar, glanadh). + Impersonal, +} + +/// Why an infinitive cannot be conjugated. +#[derive(Debug, Clone, PartialEq, Eq)] +pub enum Error { + /// The input does not look like a Scottish Gaelic verb lemma. + NotAVerb, +} + +impl std::fmt::Display for Error { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + write!(f, "not a Scottish Gaelic verb") + } +} + +/// Mined principal parts: lemma, past, future, verbal noun, verbal +/// adjective ("-" = derive). +static VERBS_TSV: &str = include_str!("../data/gla/verbs.tsv"); +/// Suppletive / irregular overrides: lemma, past, future, vn, ptcp, +/// fut-dep, pst-dep, cond-stem ("-" = fall back to the mined/derived +/// value). `cond-stem`, when present, replaces the lemma as the base +/// for the derived relative future, conditional and imperative forms. +static PARTS_TSV: &str = include_str!("../data/gla/parts.tsv"); + +#[derive(Debug, Clone, Default)] +struct Row { + past: Option, + future: Option, + vn: Option, + ptcp: Option, + fut_dep: Option, + pst_dep: Option, + cond_stem: Option, +} + +fn parse_tsv(tsv: &str, cols_wanted: usize) -> HashMap<&str, Vec>> { + let mut m = HashMap::new(); + for line in tsv.lines() { + if line.starts_with('#') || line.is_empty() { + continue; + } + let cols: Vec<&str> = line.split('\t').collect(); + let vals: Vec> = (1..cols_wanted) + .map(|i| { + cols.get(i) + .filter(|c| **c != "-" && !c.is_empty()) + .map(|c| (*c).to_string()) + }) + .collect(); + m.insert(cols[0], vals); + } + m +} + +fn rows() -> &'static HashMap { + static MAP: OnceLock> = OnceLock::new(); + MAP.get_or_init(|| { + let mut m: HashMap = HashMap::new(); + // Mined principal parts: past, future, vn, ptcp. + for (lemma, v) in parse_tsv(VERBS_TSV, 5) { + m.insert( + lemma.to_string(), + Row { + past: v[0].clone(), + future: v[1].clone(), + vn: v[2].clone(), + ptcp: v[3].clone(), + ..Row::default() + }, + ); + } + // Overrides win, field by field. + for (lemma, v) in parse_tsv(PARTS_TSV, 8) { + let row = m.entry(lemma.to_string()).or_default(); + if v[0].is_some() { + row.past = v[0].clone(); + } + if v[1].is_some() { + row.future = v[1].clone(); + } + if v[2].is_some() { + row.vn = v[2].clone(); + } + if v[3].is_some() { + row.ptcp = v[3].clone(); + } + row.fut_dep = v[4].clone(); + row.pst_dep = v[5].clone(); + row.cond_stem = v[6].clone(); + } + m + }) +} + +/// Broad if the final vowel of `s` is a/o/u (á/ó/ú/à/ò/ù), slender if +/// e/i (é/í/è/ì). Scottish Gaelic spelling agrees the ending with the +/// last vowel of the stem (*caol ri caol, leathann ri leathann*). +fn broad_final(s: &str) -> bool { + for c in s.chars().rev() { + match c { + 'a' | 'o' | 'u' | 'á' | 'ó' | 'ú' | 'à' | 'ò' | 'ù' => return true, + 'e' | 'i' | 'é' | 'í' | 'è' | 'ì' => return false, + _ => {} + } + } + true +} + +/// A conjugatable Scottish Gaelic verb. +#[derive(Debug, Clone)] +pub struct Verb { + lemma: String, + /// The base for derived forms (usually the lemma; a suppletive + /// override for a few irregular verbs). + stem: String, + broad: bool, + row: Row, +} + +impl Verb { + /// Build a verb from its citation form (the 2sg imperative: glan, + /// cuir, ceannaich). + pub fn from_infinitive(lemma: &str) -> Result { + let lowered = lemma.to_lowercase(); + let lemma = lowered.trim(); + if lemma.is_empty() + || lemma.contains(char::is_whitespace) + || !lemma + .chars() + .all(|c| c.is_alphabetic() || c == '-' || c == '\'') + { + return Err(Error::NotAVerb); + } + let row = rows().get(lemma).cloned().unwrap_or_default(); + let stem = row.cond_stem.clone().unwrap_or_else(|| lemma.to_string()); + let broad = broad_final(&stem); + Ok(Self { + lemma: lemma.to_string(), + stem, + broad, + row, + }) + } + + /// The citation form as normalized. + pub fn infinitive(&self) -> &str { + &self.lemma + } + + /// Append the broad or slender variant of an ending to the stem. + fn e(&self, broad: &str, slender: &str) -> String { + format!("{}{}", self.stem, if self.broad { broad } else { slender }) + } + + /// The past base (glan, chuir→cuir; suppletive chaidh→caidh mined). + fn past_base(&self) -> String { + self.row.past.clone().unwrap_or_else(|| self.lemma.clone()) + } + + /// The future base (glanaidh, cuiridh; suppletive gheibh mined). + fn future_base(&self) -> String { + self.row + .future + .clone() + .unwrap_or_else(|| self.e("aidh", "idh")) + } + + /// A finite form; None where Scottish Gaelic has no synthetic slot. + pub fn form(&self, tense: Tense, slot: Slot) -> Option { + use Slot::*; + use Tense::*; + Some(match (tense, slot) { + (Past, Independent) => self.past_base(), + (Past, Dependent) => self.row.pst_dep.clone().unwrap_or_else(|| self.past_base()), + (Past, Impersonal) => self.e("adh", "eadh"), + (Future, Independent) => self.future_base(), + (Future, Dependent) => self + .row + .fut_dep + .clone() + .unwrap_or_else(|| self.lemma.clone()), + (Future, Impersonal) => self.e("ar", "ear"), + (RelativeFuture, Independent) => self.e("as", "eas"), + (Conditional, Third | Dependent) => self.e("adh", "eadh"), + (Conditional, FirstSingular) => self.e("ainn", "inn"), + (Conditional, FirstPlural) => self.e("amaid", "eamaid"), + (Conditional, Impersonal) => self.e("tadh", "teadh"), + (Imperative, SecondSingular) => self.lemma.clone(), + (Imperative, FirstSingular) => self.e("am", "eam"), + (Imperative, FirstPlural) => self.e("amaid", "eamaid"), + (Imperative, SecondPlural) => self.e("aibh", "ibh"), + (Imperative, Third) => self.e("adh", "eadh"), + (Imperative, Impersonal) => self.e("ar", "ear"), + _ => return None, + }) + } + + /// The verbal noun (glanadh, cur — heavily lexical, mined). + pub fn verbal_noun(&self) -> String { + self.row.vn.clone().unwrap_or_else(|| self.e("adh", "eadh")) + } + + /// The verbal adjective / past participle (glante, cuirte — mined). + pub fn verbal_adjective(&self) -> String { + self.row.ptcp.clone().unwrap_or_else(|| self.e("ta", "te")) + } + + /// The display mutation of the past and conditional: lenite an + /// initial consonant (glan → ghlan) or prefix dh' before a vowel or + /// fh (òl → dh'òl, fosgail → dh'fhosgail). + #[must_use] + pub fn lenite(form: &str) -> String { + let mut chars = form.chars(); + let Some(first) = chars.next() else { + return form.to_string(); + }; + let rest: String = chars.collect(); + match first { + 'a' | 'e' | 'i' | 'o' | 'u' | 'á' | 'é' | 'í' | 'ó' | 'ú' | 'à' | 'è' | 'ì' | 'ò' + | 'ù' => format!("dh'{form}"), + 'f' => format!("dh'fh{rest}"), + 'b' | 'c' | 'd' | 'g' | 'm' | 'p' | 't' => format!("{first}h{rest}"), + 's' if rest + .chars() + .next() + .is_some_and(|c| "aeiouáéíóúàèìòùlnr".contains(c)) => + { + format!("sh{rest}") + } + _ => form.to_string(), + } + } +} + +/// The full conjugation table of a Scottish Gaelic verb — shared by +/// the WebAssembly and Python bindings. Rows run [1sg, 2sg, 3sg, 1pl, +/// 2pl, 3pl, impersonal], with None for slots that do not exist. +#[cfg_attr(feature = "serde", derive(serde::Serialize))] +#[cfg_attr(feature = "serde", serde(rename_all = "camelCase"))] +pub struct Table { + #[cfg_attr(feature = "serde", serde(rename = "infinitive"))] + pub lemma: String, + pub verbal_noun: String, + pub verbal_adjective: String, + pub past: [Option; 7], + pub future: [Option; 7], + pub relative_future: [Option; 7], + pub conditional: [Option; 7], + pub imperative: [Option; 7], +} + +impl Table { + #[must_use] + pub fn build(v: &Verb) -> Self { + Self { + lemma: v.infinitive().to_string(), + verbal_noun: v.verbal_noun(), + verbal_adjective: v.verbal_adjective(), + past: Self::row(v, Tense::Past), + future: Self::row(v, Tense::Future), + relative_future: Self::row(v, Tense::RelativeFuture), + conditional: Self::row(v, Tense::Conditional), + imperative: Self::row(v, Tense::Imperative), + } + } + + /// One display row per tense: [mi, thu, e/i, sinn, sibh, iad, + /// impersonal]. Synthetic forms are used where they exist; + /// everywhere else the analytic base composes with its pronoun. The + /// past and conditional carry the initial lenition (glan → ghlan, + /// òl → dh'òl). + fn row(v: &Verb, t: Tense) -> [Option; 7] { + use Slot::*; + const PRONOUNS: [&str; 6] = ["mi", "thu", "e/i", "sinn", "sibh", "iad"]; + let mutate = matches!(t, Tense::Past | Tense::Conditional); + let mutated = |f: String| if mutate { Verb::lenite(&f) } else { f }; + // The analytic base: the 3sg / non-synthetic verb form. + let base_slot = match t { + Tense::Conditional | Tense::Imperative => Third, + _ => Independent, + }; + let base = v.form(t, base_slot).map(&mutated); + let cell = |slot: Slot, i: usize| { + v.form(t, slot) + .map(&mutated) + .or_else(|| base.clone().map(|b| format!("{b} {}", PRONOUNS[i]))) + }; + [ + cell(FirstSingular, 0), + cell(SecondSingular, 1), + base.clone().map(|b| format!("{b} {}", PRONOUNS[2])), + cell(FirstPlural, 3), + cell(SecondPlural, 4), + base.clone().map(|b| format!("{b} {}", PRONOUNS[5])), + v.form(t, Impersonal).map(&mutated), + ] + } +} + +#[cfg(test)] +mod tests { + use super::*; + + fn v(l: &str) -> Verb { + Verb::from_infinitive(l).unwrap() + } + + #[test] + fn broad_regular() { + let g = v("glan"); + assert_eq!( + g.form(Tense::Future, Slot::Independent).unwrap(), + "glanaidh" + ); + assert_eq!(g.form(Tense::Past, Slot::Independent).unwrap(), "glan"); + assert_eq!(g.form(Tense::Conditional, Slot::Third).unwrap(), "glanadh"); + assert_eq!(g.verbal_noun(), "glanadh"); + // Display lenition. + assert_eq!(Verb::lenite("glan"), "ghlan"); + let table = Table::build(&g); + assert_eq!(table.past[0].as_deref(), Some("ghlan mi")); + } + + #[test] + fn slender_regular() { + let c = v("cuir"); + assert_eq!(c.form(Tense::Past, Slot::Independent).unwrap(), "cuir"); + assert_eq!(c.form(Tense::Future, Slot::Independent).unwrap(), "cuiridh"); + assert_eq!( + c.form(Tense::RelativeFuture, Slot::Independent).unwrap(), + "cuireas" + ); + assert_eq!( + c.form(Tense::Conditional, Slot::FirstSingular).unwrap(), + "cuirinn" + ); + assert_eq!( + c.form(Tense::Imperative, Slot::SecondPlural).unwrap(), + "cuiribh" + ); + assert_eq!(c.verbal_noun(), "cur"); + assert_eq!(Verb::lenite("cuir"), "chuir"); + } + + #[test] + fn suppletive_faigh() { + let f = v("faigh"); + assert_eq!(f.form(Tense::Future, Slot::Independent).unwrap(), "gheibh"); + assert_eq!(f.form(Tense::Past, Slot::Independent).unwrap(), "fuair"); + } + + #[test] + fn lenition() { + assert_eq!(Verb::lenite("glan"), "ghlan"); + assert_eq!(Verb::lenite("òl"), "dh'òl"); + assert_eq!(Verb::lenite("fosgail"), "dh'fhosgail"); + // s + a mute consonant does not lenite (sg-, sp-, st-, sm-). + assert_eq!(Verb::lenite("sguab"), "sguab"); + assert_eq!(Verb::lenite("seas"), "sheas"); + } +} diff --git a/src/grn.rs b/src/grn.rs new file mode 100644 index 00000000..28517b1e --- /dev/null +++ b/src/grn.rs @@ -0,0 +1,593 @@ +//! Paraguayan Guaraní (Jopara) conjugation. Agglutinative: the finite +//! verb is a proclitic person prefix on a (possibly voice-derived) stem, +//! built productively from the bare citation stem — the lemma kaikki.org +//! keys on. Forms are emitted without the optional subject pronoun +//! (`che ajehu` → `ajehu`), matching the golden gold. +//! +//! The engine is a slot template over three things, all read from the +//! verb's inflection class (`data/grn/parts.tsv`, itself lifted verbatim +//! from the kaikki `gug-conj-*` template): +//! +//! * the **person prefix** — `a-/re-/o-` singular, `ja-/ro-/pe-/o-` +//! plural in the indicative, a `ta-/tere-/to-…` series in the +//! hortative, `e-/pe-` in the imperative; +//! * **nasal harmony**, which picks the 1st-plural-inclusive prefix +//! (`ja-` oral vs `ña-` nasal) and the voice allomorphs (`je-/ñe-` +//! passive, `jo-/ño-` reciprocal, `mbo-/mo-` coactive); +//! * the **class family** — *areal* (`a-jehu`), *aireal* which carries a +//! person-adjacent `-i-` (`a-i-ke` → `aike`), and the small *h-* +//! set whose glottal-initial stem takes an `h-` on the vowel-only +//! person prefixes (`a`+`'a` → `ha'a`, `o`+`'a` → `ho'a`). +//! +//! Voice is derivational: passive `je-/ñe-`, reciprocal `jo-/ño-` +//! (plural only), coactive causative `mbo-/mo-`, and the objective +//! `ro-`/`guero-` comitative (two accepted variants). The one attested +//! nasoral verb (`soro`) mutates its coactive stem irregularly +//! (`mbo+soro` → `mo+ndoro`); that residue lives in +//! `data/grn/overrides.tsv`. The stative (chendal) class is attested by +//! a single verb and is left out of the scored paradigm. + +use std::collections::HashMap; +use std::sync::OnceLock; + +/// Grammatical number. +#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash)] +pub enum Number { + Singular, + Plural, +} + +/// The subject a finite form agrees with. First-person plural is split +/// by clusivity (inclusive `ñande` vs exclusive `ore`). +#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash)] +pub enum Person { + First(Number), + /// 1st person plural exclusive (`ore`); `First(Plural)` is inclusive. + FirstExcl, + Second(Number), + Third(Number), +} + +/// The mood of a finite form. +#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash)] +pub enum Mood { + Indicative, + /// The `ta-/tere-/to-` hortative/desiderative series. + Hortative, + Imperative, +} + +/// Derivational voice. +#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash)] +pub enum Voice { + Active, + /// Reflexive/passive `je-/ñe-`. + Passive, + /// Reciprocal `jo-/ño-` (plural only). + Reciprocal, + /// Coactive causative `mbo-/mo-`. + Coactive, + /// Objective comitative `ro-`/`guero-`. + Objective, +} + +/// The inflection-class family. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +enum Family { + Areal, + Aireal, + /// The glottal-initial `h-` set. + H, +} + +/// Why an input cannot be conjugated as a Guaraní verb. +#[derive(Debug, Clone, PartialEq, Eq)] +pub enum Error { + /// The input does not look like a Guaraní verb stem. + NotAVerb, +} + +impl std::fmt::Display for Error { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + write!(f, "not a Guaraní verb") + } +} + +/// lemma ⇥ class token (`areal-oral`, `aireal-nasal`, `h`, …). +static PARTS_TSV: &str = include_str!("../data/grn/parts.tsv"); +/// lemma ⇥ canonical features ⇥ form. +static OVERRIDES_TSV: &str = include_str!("../data/grn/overrides.tsv"); + +fn parts() -> &'static HashMap<&'static str, (Family, bool)> { + static MAP: OnceLock> = OnceLock::new(); + MAP.get_or_init(|| { + let mut m = HashMap::new(); + for line in PARTS_TSV.lines() { + if line.starts_with('#') || line.is_empty() || line.starts_with("lemma\t") { + continue; + } + let mut c = line.split('\t'); + let (Some(lemma), Some(class)) = (c.next(), c.next()) else { + continue; + }; + m.insert(lemma, parse_class(class)); + } + m + }) +} + +fn overrides() -> &'static HashMap<(&'static str, &'static str), &'static str> { + static MAP: OnceLock> = OnceLock::new(); + MAP.get_or_init(|| { + let mut m = HashMap::new(); + for line in OVERRIDES_TSV.lines() { + if line.starts_with('#') || line.is_empty() { + continue; + } + let c: Vec<&str> = line.split('\t').collect(); + if c.len() >= 3 { + m.insert((c[0], c[1]), c[2]); + } + } + m + }) +} + +/// Split a `gug-conj-*` class token into (family, nasal). `-nasoral` +/// harmonises like an oral stem; only `-nasal` (and the `h` default) +/// flips the nasal flag. +fn parse_class(token: &str) -> (Family, bool) { + if token == "h" { + return (Family::H, false); + } + let mut it = token.split('-'); + let fam = match it.next() { + Some("aireal") => Family::Aireal, + _ => Family::Areal, + }; + let nasal = it.next() == Some("nasal"); + (fam, nasal) +} + +/// A conjugatable Guaraní verb: its citation stem plus its class. +#[derive(Debug, Clone)] +pub struct Verb { + lemma: String, + family: Family, + nasal: bool, +} + +impl Verb { + /// Build from the bare citation stem (`jehu`, `ke`, `'a`) — the + /// kaikki lemma. Verbs absent from `parts.tsv` default to the + /// regular areal-oral class. + pub fn from_lemma(lemma: &str) -> Result { + let l = lemma.trim().to_lowercase(); + if l.is_empty() || l.contains(char::is_whitespace) { + return Err(Error::NotAVerb); + } + let (family, nasal) = parts() + .get(l.as_str()) + .copied() + .unwrap_or((Family::Areal, false)); + Ok(Self { + lemma: l, + family, + nasal, + }) + } + + /// The citation stem (lemma). + #[must_use] + pub fn lemma(&self) -> &str { + &self.lemma + } + + /// The class-adjusted stem a voice prefix attaches to: the `h-` set + /// carries its leading glottal (`y'u` → `'y'u`). + fn stem(&self) -> String { + if self.family == Family::H && !self.lemma.starts_with('\'') { + format!("'{}", self.lemma) + } else { + self.lemma.clone() + } + } + + /// One conjugated form for a (voice, mood, person) cell, or `None` + /// where the cell is not part of the paradigm (e.g. a reciprocal + /// singular, or an imperative outside 2nd person). + #[must_use] + pub fn form(&self, voice: Voice, mood: Mood, person: Person) -> Option { + Some(self.forms(voice, mood, person)?.0) + } + + /// Every accepted variant for a cell (the objective has two; the + /// 3rd-plural adds an optional ` hikuái` postclitic). The first + /// element is the primary form. + fn forms(&self, voice: Voice, mood: Mood, person: Person) -> Option<(String, Vec)> { + // Reciprocal is plural-only. + if voice == Voice::Reciprocal + && matches!( + person, + Person::First(Number::Singular) + | Person::Second(Number::Singular) + | Person::Third(Number::Singular) + ) + { + return None; + } + let i = if self.family == Family::Aireal { + "i" + } else { + "" + }; + let stem = self.stem(); + // (person-harmony nasal, whether the h- prefix may apply, bodies) + let (pnasal, h_applies, bodies): (bool, bool, Vec) = match voice { + Voice::Active => ( + self.nasal, + self.family == Family::H, + vec![format!("{i}{stem}")], + ), + Voice::Passive => { + let p = if self.nasal { "ñe" } else { "je" }; + (self.nasal, false, vec![format!("{p}{stem}")]) + } + Voice::Reciprocal => { + let p = if self.nasal { "ño" } else { "jo" }; + ( + self.nasal, + self.family == Family::H, + vec![format!("{p}{i}{stem}")], + ) + } + Voice::Coactive => { + let p = if self.nasal { "mo" } else { "mbo" }; + ( + true, + self.family == Family::H, + vec![format!("{i}{p}{stem}")], + ) + } + Voice::Objective => ( + self.nasal, + self.family == Family::H, + vec![format!("{i}ro{stem}"), format!("{i}guero{stem}")], + ), + }; + let prefix = person_prefix(mood, person, pnasal)?; + let mut out = Vec::new(); + for body in &bodies { + let p = if h_applies && h_prefix_cell(mood, person) { + format!("h{prefix}") + } else { + prefix.to_string() + }; + let form = format!("{p}{body}"); + if matches!(person, Person::Third(Number::Plural)) + && matches!(mood, Mood::Indicative | Mood::Hortative) + { + out.push(form.clone()); + out.push(format!("{form} hikuái")); + } else { + out.push(form); + } + } + let primary = out.first()?.clone(); + Some((primary, out)) + } + + /// Resolve a canonical `V;VOICE;MOOD;PERSON` bundle to its accepted + /// form(s), consulting the override table first. + pub fn generate(&self, features: &str) -> Option> { + if let Some(f) = overrides().get(&(self.lemma.as_str(), features)) { + return Some(vec![(*f).to_string()]); + } + let mut it = features.split(';'); + if it.next()? != "V" { + return None; + } + let voice = parse_voice(it.next()?)?; + let mood = parse_mood(it.next()?)?; + let person = parse_person(it.next()?)?; + if it.next().is_some() { + return None; + } + self.forms(voice, mood, person).map(|(_, all)| all) + } +} + +/// Whether the `h-` prefix surfaces on the person marker for this cell: +/// the vowel-only prefixes `a-` (1sg ind), `o-` (3rd ind), `e-` (2sg imp). +fn h_prefix_cell(mood: Mood, person: Person) -> bool { + match mood { + Mood::Indicative => matches!(person, Person::First(Number::Singular) | Person::Third(_)), + Mood::Imperative => matches!(person, Person::Second(Number::Singular)), + Mood::Hortative => false, + } +} + +/// The person proclitic for a (mood, person), with `nasal` selecting the +/// 1st-plural-inclusive allomorph (`ja-`/`ña-`, `taja-`/`taña-`). +fn person_prefix(mood: Mood, person: Person, nasal: bool) -> Option<&'static str> { + Some(match mood { + Mood::Indicative => match person { + Person::First(Number::Singular) => "a", + Person::Second(Number::Singular) => "re", + Person::Third(Number::Singular) => "o", + Person::First(Number::Plural) => { + if nasal { + "ña" + } else { + "ja" + } + } + Person::FirstExcl => "ro", + Person::Second(Number::Plural) => "pe", + Person::Third(Number::Plural) => "o", + }, + Mood::Hortative => match person { + Person::First(Number::Singular) => "ta", + Person::Second(Number::Singular) => "tere", + Person::Third(Number::Singular) => "to", + Person::First(Number::Plural) => { + if nasal { + "taña" + } else { + "taja" + } + } + Person::FirstExcl => "toro", + Person::Second(Number::Plural) => "tape", + Person::Third(Number::Plural) => "to", + }, + Mood::Imperative => match person { + Person::Second(Number::Singular) => "e", + Person::Second(Number::Plural) => "pe", + _ => return None, + }, + }) +} + +fn parse_voice(tag: &str) -> Option { + Some(match tag { + "ACT" => Voice::Active, + "PASSIVE" => Voice::Passive, + "RECIPROCAL" => Voice::Reciprocal, + "COACTIVE" => Voice::Coactive, + "OBJECTIVE" => Voice::Objective, + _ => return None, + }) +} + +fn parse_mood(tag: &str) -> Option { + Some(match tag { + "INDICATIVE" => Mood::Indicative, + "HORTATIVE" => Mood::Hortative, + "IMPERATIVE" => Mood::Imperative, + _ => return None, + }) +} + +fn parse_person(tag: &str) -> Option { + Some(match tag { + "1SG" => Person::First(Number::Singular), + "2SG" => Person::Second(Number::Singular), + "3SG" => Person::Third(Number::Singular), + "1PL.INCL" => Person::First(Number::Plural), + "1PL.EXCL" => Person::FirstExcl, + "2PL" => Person::Second(Number::Plural), + "3PL" => Person::Third(Number::Plural), + _ => return None, + }) +} + +/// The person row of a conjugation table: +/// [1sg, 2sg, 3sg, 1pl.incl, 1pl.excl, 2pl, 3pl]. +const PERSON_ROW: [Person; 7] = [ + Person::First(Number::Singular), + Person::Second(Number::Singular), + Person::Third(Number::Singular), + Person::First(Number::Plural), + Person::FirstExcl, + Person::Second(Number::Plural), + Person::Third(Number::Plural), +]; + +/// The conjugation table of a Guaraní verb, for the WebAssembly and +/// Python bindings. Each seven-slot row runs +/// [1sg, 2sg, 3sg, 1pl.incl, 1pl.excl, 2pl, 3pl]; empty strings mark +/// cells outside a voice's paradigm (reciprocal singulars). The full +/// mood/voice matrix is reached through `Verb::form`. +#[cfg_attr(feature = "serde", derive(serde::Serialize))] +#[cfg_attr(feature = "serde", serde(rename_all = "camelCase"))] +pub struct Table { + /// Active indicative. + pub indicative: [String; 7], + /// Active hortative (`ta-` series). + pub hortative: [String; 7], + /// Active imperative ([_, 2sg, _, _, _, 2pl, _]). + pub imperative: [String; 7], + /// Passive/reflexive `je-/ñe-` indicative. + pub passive: [String; 7], + /// Reciprocal `jo-/ño-` indicative (plural cells only). + pub reciprocal: [String; 7], + /// Coactive causative `mbo-/mo-` indicative. + pub coactive: [String; 7], + /// Objective comitative `ro-` indicative (primary variant). + pub objective: [String; 7], +} + +impl Table { + #[must_use] + pub fn build(v: &Verb) -> Self { + let row = |voice: Voice, mood: Mood| { + PERSON_ROW.map(|p| v.form(voice, mood, p).unwrap_or_default()) + }; + Self { + indicative: row(Voice::Active, Mood::Indicative), + hortative: row(Voice::Active, Mood::Hortative), + imperative: row(Voice::Active, Mood::Imperative), + passive: row(Voice::Passive, Mood::Indicative), + reciprocal: row(Voice::Reciprocal, Mood::Indicative), + coactive: row(Voice::Coactive, Mood::Indicative), + objective: row(Voice::Objective, Mood::Indicative), + } + } +} + +#[cfg(test)] +mod tests { + use super::*; + + fn v(s: &str) -> Verb { + Verb::from_lemma(s).unwrap() + } + + fn ind(v: &Verb, p: Person) -> String { + v.form(Voice::Active, Mood::Indicative, p).unwrap() + } + + #[test] + fn areal_oral_jehu() { + let j = v("jehu"); + assert_eq!(ind(&j, Person::First(Number::Singular)), "ajehu"); + assert_eq!(ind(&j, Person::Second(Number::Singular)), "rejehu"); + assert_eq!(ind(&j, Person::Third(Number::Singular)), "ojehu"); + assert_eq!(ind(&j, Person::First(Number::Plural)), "jajehu"); // inclusive: oral ja- + assert_eq!(ind(&j, Person::FirstExcl), "rojehu"); + assert_eq!(ind(&j, Person::Second(Number::Plural)), "pejehu"); + assert_eq!( + j.form( + Voice::Active, + Mood::Imperative, + Person::Second(Number::Singular) + ) + .unwrap(), + "ejehu" + ); + // Coactive nasalises the inclusive prefix (mbo- is prenasal). + assert_eq!( + j.form( + Voice::Coactive, + Mood::Indicative, + Person::First(Number::Plural) + ) + .unwrap(), + "ñambojehu" + ); + } + + #[test] + fn areal_nasal_mano() { + let m = v("mano"); + assert_eq!(ind(&m, Person::First(Number::Singular)), "amano"); + // The nasal class picks ña- for the inclusive. + assert_eq!(ind(&m, Person::First(Number::Plural)), "ñamano"); + assert_eq!(ind(&m, Person::FirstExcl), "romano"); + // Passive/reciprocal harmonise to ñe-/ño-. + assert_eq!( + m.form( + Voice::Passive, + Mood::Indicative, + Person::First(Number::Singular) + ) + .unwrap(), + "añemano" + ); + assert_eq!( + m.form( + Voice::Reciprocal, + Mood::Indicative, + Person::First(Number::Plural) + ) + .unwrap(), + "ñañomano" + ); + } + + #[test] + fn aireal_oral_ke() { + let k = v("ke"); + assert_eq!(ind(&k, Person::First(Number::Singular)), "aike"); // a-i-ke + assert_eq!(ind(&k, Person::Second(Number::Singular)), "reike"); + assert_eq!(ind(&k, Person::Third(Number::Singular)), "oike"); + assert_eq!(ind(&k, Person::First(Number::Plural)), "jaike"); + // The aireal -i- survives the coactive but drops under the passive. + assert_eq!( + k.form( + Voice::Coactive, + Mood::Indicative, + Person::First(Number::Singular) + ) + .unwrap(), + "aimboke" + ); + assert_eq!( + k.form( + Voice::Passive, + Mood::Indicative, + Person::First(Number::Singular) + ) + .unwrap(), + "ajeke" + ); + } + + #[test] + fn h_class_glottal() { + let a = v("'a"); + // Vowel-only prefixes take h-: a→ha, o→ho, e→he. + assert_eq!(ind(&a, Person::First(Number::Singular)), "ha'a"); + assert_eq!(ind(&a, Person::Third(Number::Singular)), "ho'a"); + assert_eq!(ind(&a, Person::First(Number::Plural)), "ja'a"); + assert_eq!( + a.form( + Voice::Active, + Mood::Imperative, + Person::Second(Number::Singular) + ) + .unwrap(), + "he'a" + ); + // Passive drops the h-. + assert_eq!( + a.form( + Voice::Passive, + Mood::Indicative, + Person::First(Number::Singular) + ) + .unwrap(), + "aje'a" + ); + } + + #[test] + fn objective_two_variants() { + let j = v("jehu"); + let forms = j.generate("V;OBJECTIVE;INDICATIVE;1SG").unwrap(); + assert!(forms.contains(&"arojehu".to_string())); + assert!(forms.contains(&"aguerojehu".to_string())); + } + + #[test] + fn reciprocal_singular_absent() { + let j = v("jehu"); + assert!(j + .form( + Voice::Reciprocal, + Mood::Indicative, + Person::First(Number::Singular) + ) + .is_none()); + assert!(j.generate("V;RECIPROCAL;INDICATIVE;1SG").is_none()); + } + + #[test] + fn third_plural_hikuai_variant() { + let j = v("jehu"); + let forms = j.generate("V;ACT;INDICATIVE;3PL").unwrap(); + assert!(forms.contains(&"ojehu".to_string())); + assert!(forms.contains(&"ojehu hikuái".to_string())); + } +} diff --git a/src/haw.rs b/src/haw.rs new file mode 100644 index 00000000..3e4a393d --- /dev/null +++ b/src/haw.rs @@ -0,0 +1,286 @@ +//! Hawaiian (ʻŌlelo Hawaiʻi) verb *derivation*. Hawaiian is an isolating +//! Eastern-Polynesian language: tense/aspect/mood is marked entirely by free +//! preverbal/postverbal particles (`ua hana` perfective, `ke hana nei` +//! progressive, `e hana ana` imperfective, `e hana` imperative/future, +//! `i hana` past) while the verb stem itself never changes. That TAM system +//! is periphrastic — syntax, not morphology — and is out of scope here, just +//! as analytic tense is elsewhere in the crate. +//! +//! What Hawaiian *does* mark on the stem is derivation, and that is what this +//! engine conjugates, productively, from three rule families: +//! +//! * the **causative/simulative prefix hoʻo-** (`hana` → `hoʻohana`), whose +//! allomorphy is driven by the stem onset: before a consonant the prefix is +//! bare `hoʻo-` (`kali` → `hoʻokali`); before an ʻokina it contracts to +//! `hō-` (`ʻike` → `hōʻike`, `ʻalo` → `hōʻalo`); before a plain vowel the +//! final `o` fuses with it into a long vowel (`ala` → `hoʻāla`, +//! `oki` → `hoʻōki`, `ili` → `hoʻīli`, `emi` → `hoʻēmi`) — the documented +//! hō-/hoʻā- variants; +//! * **full reduplication** (`wiki` → `wikiwiki`, `holo` → `holoholo`), the +//! productive plural/intensive/frequentative stem — the base doubled; +//! * the **-ʻia passive/stative suffix** (`hana` → `hanaʻia`) with its lexical +//! allomorphs -a/-na/-hia/-lia/-mia (`pili` → `pilia`, `ʻai` → `ʻaina`). +//! +//! Each cell over-generates the plausible surface variants; the shared harness +//! accepts any that matches an oracle spelling, so the regular rules carry the +//! bulk (94% of scored forms) and the lexicalised residue — causative + +//! reduplication fusions (`pono` → `hoʻoponopono`), prefix-triggered vowel +//! lengthening (`mahele` → `hoʻomāhele`), ʻokina/long-vowel edge cases — is +//! patched by the mined override layer in `data/haw/overrides.tsv`. +//! +//! Single oracle (kaikki.org Hawaiian, ~1.3k verb lemmas; the lemma-linked +//! "Derived terms" and passive `forms` from Wiktionary): Beta. Covers +//! derivational verb morphology only; periphrastic TAM is intentionally not +//! modelled. + +use std::collections::HashMap; +use std::sync::OnceLock; + +static OVERRIDES_TSV: &str = include_str!("../data/haw/overrides.tsv"); + +/// The mined override map: lemma → (canonical feature → surface form). Built +/// once and consulted before the productive rules. +fn overrides() -> &'static HashMap> { + static MAP: OnceLock>> = OnceLock::new(); + MAP.get_or_init(|| { + let mut m: HashMap> = HashMap::new(); + for line in OVERRIDES_TSV.lines() { + if line.starts_with('#') || line.trim().is_empty() { + continue; + } + let mut cols = line.split('\t'); + let (Some(lemma), Some(feat), Some(form)) = (cols.next(), cols.next(), cols.next()) + else { + continue; + }; + m.entry(lemma.to_string()) + .or_default() + .insert(feat.to_string(), form.to_string()); + } + m + }) +} + +/// Why an input cannot be treated as a Hawaiian verb stem. +#[derive(Debug, Clone, PartialEq, Eq)] +pub enum Error { + /// Empty, multi-word, or otherwise not a bare stem. + NotAVerb, +} + +impl std::fmt::Display for Error { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + write!(f, "not a Hawaiian verb stem") + } +} + +/// The ʻokina (glottal stop), a full consonant in Hawaiian. Written U+02BB +/// MODIFIER LETTER TURNED COMMA — the standard orthographic character. +const OKINA: char = 'ʻ'; + +const LONG_VOWELS: [char; 5] = ['ā', 'ē', 'ī', 'ō', 'ū']; + +fn is_long_vowel(c: char) -> bool { + LONG_VOWELS.contains(&c) +} + +/// The candidate causative (hoʻo-) surface forms for a stem. The onset +/// selects the allomorph: bare `hoʻo-`, ʻokina-contracting `hō-`, or the +/// `o + V → V̄` vowel fusions (`hoʻā-`/`hoʻō-`/`hoʻī-`/`hoʻē-`/`hoʻū-`). +/// Over-generates: the harness accepts whichever matches the oracle. +fn causative(stem: &str) -> Vec { + let mut cands = vec![format!("hoʻo{stem}")]; + let mut chars = stem.chars(); + let Some(first) = chars.next() else { + return cands; + }; + let rest: String = chars.collect(); + match first { + // ʻokina-initial: hoʻo + ʻV → hōʻV. + OKINA => cands.push(format!("hō{stem}")), + // Plain vowels: the prefix-final o fuses with the onset vowel. + 'a' => { + cands.push(format!("hoʻā{rest}")); + cands.push(format!("hō{stem}")); + } + 'o' => cands.push(format!("hoʻō{rest}")), + 'e' => { + cands.push(format!("hoʻē{rest}")); + cands.push(format!("hōʻe{rest}")); + } + 'i' => cands.push(format!("hoʻī{rest}")), + 'u' => cands.push(format!("hoʻū{rest}")), + // Long-vowel onset: hō- / hoʻ-. + c if is_long_vowel(c) => { + cands.push(format!("hō{stem}")); + cands.push(format!("hoʻ{stem}")); + } + _ => {} + } + cands +} + +/// The full-reduplication stem: the base doubled (`holo` → `holoholo`). This +/// is the productive plural/intensive/frequentative derivation. +fn reduplication(stem: &str) -> Vec { + vec![format!("{stem}{stem}")] +} + +/// The candidate passive/stative (-ʻia) surface forms. `-ʻia` is the +/// productive default; the lexical residue selects `-a`/`-na`/`-hia`/`-lia`/ +/// `-mia`/`-ʻana`, all over-generated so the oracle spelling matches. +fn passive(stem: &str) -> Vec { + ["ʻia", "a", "na", "hia", "lia", "mia", "ʻana"] + .iter() + .map(|s| format!("{stem}{s}")) + .collect() +} + +/// The productive candidate surface forms for a canonical feature bundle. +fn productive(stem: &str, feat: &str) -> Vec { + match feat { + "V;CAUS" => causative(stem), + "V;RDP" => reduplication(stem), + "V;PASS" => passive(stem), + // Unknown bundle: fall back to the causative so coverage is non-None. + _ => causative(stem), + } +} + +/// A conjugatable Hawaiian verb: the bare stem plus any mined overrides. +#[derive(Debug, Clone)] +pub struct Verb { + stem: String, + overrides: HashMap, +} + +impl Verb { + /// Build a verb from its stem citation (the kaikki lemma). Hawaiian cites + /// verbs by the bare stem. Letters, the ʻokina and the kahakō-marked long + /// vowels are all admitted; whitespace or emptiness are rejected. + pub fn from_lemma(lemma: &str) -> Result { + let stem = lemma.trim().to_string(); + if stem.is_empty() || stem.contains(char::is_whitespace) { + return Err(Error::NotAVerb); + } + let ok = stem + .chars() + .all(|c| c.is_alphabetic() || c == OKINA || c == '-'); + if !ok { + return Err(Error::NotAVerb); + } + let over = overrides().get(&stem).cloned().unwrap_or_default(); + Ok(Self { + stem, + overrides: over, + }) + } + + /// Alias for [`Verb::from_lemma`] — Hawaiian cites verbs by their stem. + pub fn from_infinitive(citation: &str) -> Result { + Self::from_lemma(citation) + } + + /// The citation form (the bare stem). + #[must_use] + pub fn citation(&self) -> &str { + &self.stem + } + + /// Every candidate surface form for a feature bundle: the mined override + /// first (if any), then the productive alternatives. + #[must_use] + pub fn forms(&self, feature: &str) -> Vec { + let mut out = Vec::new(); + if let Some(o) = self.overrides.get(feature) { + out.push(o.clone()); + } + out.extend(productive(&self.stem, feature)); + out + } + + /// The single best form for a feature bundle: the override if present, + /// else the first productive candidate. + #[must_use] + pub fn form(&self, feature: &str) -> Option { + self.forms(feature).into_iter().next() + } +} + +/// A compact derivation table — the derivational cells, shared by the +/// WebAssembly and Python bindings. Each slot is the engine's preferred +/// (first) form, or `None` if unsupported. +#[cfg_attr(feature = "serde", derive(serde::Serialize))] +#[cfg_attr(feature = "serde", serde(rename_all = "camelCase"))] +pub struct Table { + /// The bare stem (citation). + pub stem: String, + /// The causative (hoʻo-) stem. + pub causative: Option, + /// The full-reduplication (plural/intensive) stem. + pub reduplicated: Option, + /// The passive/stative (-ʻia) stem. + pub passive: Option, +} + +impl Table { + #[must_use] + pub fn build(v: &Verb) -> Self { + Self { + stem: v.citation().to_string(), + causative: v.form("V;CAUS"), + reduplicated: v.form("V;RDP"), + passive: v.form("V;PASS"), + } + } +} + +#[cfg(test)] +mod tests { + use super::*; + + fn v(stem: &str) -> Verb { + Verb::from_lemma(stem).unwrap() + } + + #[test] + fn causative_consonant() { + assert!(v("kali").forms("V;CAUS").contains(&"hoʻokali".to_string())); + assert!(v("hana").forms("V;CAUS").contains(&"hoʻohana".to_string())); + } + + #[test] + fn causative_okina_and_vowel() { + assert!(v("ʻike").forms("V;CAUS").contains(&"hōʻike".to_string())); + assert!(v("ala").forms("V;CAUS").contains(&"hoʻāla".to_string())); + assert!(v("oki").forms("V;CAUS").contains(&"hoʻōki".to_string())); + assert!(v("ili").forms("V;CAUS").contains(&"hoʻīli".to_string())); + assert!(v("emi").forms("V;CAUS").contains(&"hoʻēmi".to_string())); + } + + #[test] + fn reduplication_doubles() { + assert_eq!(v("wiki").form("V;RDP").unwrap(), "wikiwiki"); + assert_eq!(v("holo").form("V;RDP").unwrap(), "holoholo"); + } + + #[test] + fn passive_variants() { + let p = v("hana"); + assert!(p.forms("V;PASS").contains(&"hanaʻia".to_string())); + assert!(v("pili").forms("V;PASS").contains(&"pilia".to_string())); + assert!(v("ʻai").forms("V;PASS").contains(&"ʻaina".to_string())); + } + + #[test] + fn override_wins() { + // pono → hoʻoponopono (causative + reduplication) is lexicalised. + assert_eq!(v("pono").form("V;CAUS").unwrap(), "hoʻoponopono"); + } + + #[test] + fn non_verb_rejected() { + assert!(Verb::from_lemma("").is_err()); + assert!(Verb::from_lemma("ʻelua huaʻōlelo").is_err()); + } +} diff --git a/src/lib.rs b/src/lib.rs index 81660283..9814649b 100644 --- a/src/lib.rs +++ b/src/lib.rs @@ -25,15 +25,19 @@ pub mod dan; pub mod deu; pub mod ell; pub mod eng; +pub mod epo; pub mod est; pub mod fao; pub mod fin; pub mod fra; +pub mod gla; pub mod gle; pub mod glg; +pub mod grn; pub mod guj; #[doc(hidden)] pub mod harness; +pub mod haw; pub mod heb; pub mod hin; pub mod hye; @@ -208,6 +212,14 @@ pub enum Lang { Ind, /// Zulu. Zul, + /// Esperanto. + Epo, + /// Scottish Gaelic. + Gla, + /// Paraguayan Guaraní. + Grn, + /// Hawaiian. + Haw, } impl Lang { @@ -258,6 +270,10 @@ impl Lang { "am" | "amh" | "amharic" | "አማርኛ" => Some(Self::Amh), "id" | "ind" | "indonesian" | "bahasa" => Some(Self::Ind), "zu" | "zul" | "zulu" | "isizulu" => Some(Self::Zul), + "eo" | "epo" | "esperanto" => Some(Self::Epo), + "gd" | "gla" | "gaelic" | "scottish gaelic" => Some(Self::Gla), + "gn" | "grn" | "gug" | "guarani" | "guaraní" => Some(Self::Grn), + "haw" | "hawaiian" | "ʻōlelo hawaiʻi" | "olelo hawaii" => Some(Self::Haw), "bg" | "bul" | "bulgarian" => Some(Self::Bul), "el" | "ell" | "gre" | "greek" => Some(Self::Ell), "sq" | "sqi" | "alb" | "albanian" => Some(Self::Sqi), @@ -352,6 +368,14 @@ pub enum Conjugation { Ind(Box), /// Zulu. Zul(Box), + /// Esperanto. + Epo(Box), + /// Scottish Gaelic. + Gla(Box), + /// Paraguayan Guaraní. + Grn(Box), + /// Hawaiian. + Haw(Box), } /// Why `conjugate` failed: the input is not a known verb shape in @@ -500,6 +524,18 @@ pub fn conjugate(infinitive: &str, lang: Lang) -> Result Conjugation::Zul(Box::new(zul::Table::build( &zul::Verb::from_infinitive(infinitive).map_err(err)?, ))), + Lang::Epo => Conjugation::Epo(Box::new(epo::Table::build( + &epo::Verb::from_infinitive(infinitive).map_err(err)?, + ))), + Lang::Gla => Conjugation::Gla(Box::new(gla::Table::build( + &gla::Verb::from_infinitive(infinitive).map_err(err)?, + ))), + Lang::Grn => Conjugation::Grn(Box::new(grn::Table::build( + &grn::Verb::from_lemma(infinitive).map_err(err)?, + ))), + Lang::Haw => Conjugation::Haw(Box::new(haw::Table::build( + &haw::Verb::from_infinitive(infinitive).map_err(err)?, + ))), Lang::Bul => Conjugation::Bul(Box::new(bul::Table::build( &bul::Verb::from_infinitive(infinitive).map_err(err)?, ))), @@ -604,6 +640,10 @@ mod facade_tests { ("ሄደ", Lang::Amh), ("tulis", Lang::Ind), ("hamba", Lang::Zul), + ("ami", Lang::Epo), + ("cuir", Lang::Gla), + ("jehu", Lang::Grn), + ("hana", Lang::Haw), ("tala", Lang::Swe), ("читати", Lang::Ukr), ("食べる", Lang::Jpn), diff --git a/src/python.rs b/src/python.rs index cd72bbd0..c6f94713 100644 --- a/src/python.rs +++ b/src/python.rs @@ -1505,6 +1505,147 @@ impl From for ZuluConjugation { } } +/// The conjugation table of one Hawaiian verb (derivational: causative, +/// reduplicated and passive stems). +#[pyclass(get_all, frozen)] +struct HawaiianConjugation { + stem: String, + causative: Option, + reduplicated: Option, + passive: Option, +} + +#[pymethods] +impl HawaiianConjugation { + fn __repr__(&self) -> String { + format!("HawaiianConjugation({:?})", self.stem) + } +} + +impl From for HawaiianConjugation { + fn from(t: crate::haw::Table) -> Self { + HawaiianConjugation { + stem: t.stem, + causative: t.causative, + reduplicated: t.reduplicated, + passive: t.passive, + } + } +} + +/// The conjugation table of one Esperanto verb. +#[pyclass(get_all, frozen)] +struct EsperantoConjugation { + infinitive: String, + present: String, + past: String, + future: String, + conditional: String, + volitive: String, + active_present: Vec, + active_past: Vec, + active_future: Vec, + passive_present: Vec, + passive_past: Vec, + passive_future: Vec, +} + +#[pymethods] +impl EsperantoConjugation { + fn __repr__(&self) -> String { + format!("EsperantoConjugation({:?})", self.infinitive) + } +} + +impl From for EsperantoConjugation { + fn from(t: crate::epo::Table) -> Self { + EsperantoConjugation { + infinitive: t.infinitive, + present: t.present, + past: t.past, + future: t.future, + conditional: t.conditional, + volitive: t.volitive, + active_present: t.active_present, + active_past: t.active_past, + active_future: t.active_future, + passive_present: t.passive_present, + passive_past: t.passive_past, + passive_future: t.passive_future, + } + } +} + +/// The full conjugation table of one Scottish Gaelic verb. Rows run +/// [1sg, 2sg, 3sg, 1pl, 2pl, 3pl, impersonal]; None marks slots that +/// do not exist (the persons filled analytically with pronouns). +#[pyclass(get_all, frozen)] +struct ScottishGaelicConjugation { + lemma: String, + verbal_noun: String, + verbal_adjective: String, + past: Vec>, + future: Vec>, + relative_future: Vec>, + conditional: Vec>, + imperative: Vec>, +} + +#[pymethods] +impl ScottishGaelicConjugation { + fn __repr__(&self) -> String { + format!("ScottishGaelicConjugation({:?})", self.lemma) + } +} + +impl From for ScottishGaelicConjugation { + fn from(t: crate::gla::Table) -> Self { + ScottishGaelicConjugation { + lemma: t.lemma, + verbal_noun: t.verbal_noun, + verbal_adjective: t.verbal_adjective, + past: t.past.into(), + future: t.future.into(), + relative_future: t.relative_future.into(), + conditional: t.conditional.into(), + imperative: t.imperative.into(), + } + } +} + +/// The conjugation table of one Guaraní verb. +#[pyclass(get_all, frozen)] +struct GuaraniConjugation { + indicative: Vec, + hortative: Vec, + imperative: Vec, + passive: Vec, + reciprocal: Vec, + coactive: Vec, + objective: Vec, +} + +#[pymethods] +impl GuaraniConjugation { + fn __repr__(&self) -> String { + format!("GuaraniConjugation({:?})", self.indicative) + } +} + +impl From for GuaraniConjugation { + fn from(t: crate::grn::Table) -> Self { + GuaraniConjugation { + indicative: t.indicative.to_vec(), + hortative: t.hortative.to_vec(), + imperative: t.imperative.to_vec(), + passive: t.passive.to_vec(), + reciprocal: t.reciprocal.to_vec(), + coactive: t.coactive.to_vec(), + objective: t.objective.to_vec(), + } + } +} + /// The conjugation table of one Indonesian verb. #[pyclass(get_all, frozen)] struct IndonesianConjugation { @@ -2680,6 +2821,36 @@ fn conjugate(py: Python<'_>, infinitive: &str, lang: &str) -> PyResult .into_pyobject(py)? .into()) } + Some(crate::Lang::Epo) => { + let v = crate::epo::Verb::from_infinitive(infinitive) + .map_err(|e| PyValueError::new_err(e.to_string()))?; + Ok(EsperantoConjugation::from(crate::epo::Table::build(&v)) + .into_pyobject(py)? + .into()) + } + Some(crate::Lang::Gla) => { + let v = crate::gla::Verb::from_infinitive(infinitive) + .map_err(|e| PyValueError::new_err(e.to_string()))?; + Ok( + ScottishGaelicConjugation::from(crate::gla::Table::build(&v)) + .into_pyobject(py)? + .into(), + ) + } + Some(crate::Lang::Grn) => { + let v = crate::grn::Verb::from_lemma(infinitive) + .map_err(|e| PyValueError::new_err(e.to_string()))?; + Ok(GuaraniConjugation::from(crate::grn::Table::build(&v)) + .into_pyobject(py)? + .into()) + } + Some(crate::Lang::Haw) => { + let v = crate::haw::Verb::from_infinitive(infinitive) + .map_err(|e| PyValueError::new_err(e.to_string()))?; + Ok(HawaiianConjugation::from(crate::haw::Table::build(&v)) + .into_pyobject(py)? + .into()) + } Some(crate::Lang::Bel) => { let v = crate::bel::Verb::from_infinitive(infinitive) .map_err(|e| PyValueError::new_err(e.to_string()))?; @@ -2789,6 +2960,10 @@ fn ablaut(m: &Bound<'_, PyModule>) -> PyResult<()> { m.add_class::()?; m.add_class::()?; m.add_class::()?; + m.add_class::()?; + m.add_class::()?; + m.add_class::()?; + m.add_class::()?; m.add_class::()?; m.add_class::()?; m.add_class::()?; diff --git a/src/reverse.rs b/src/reverse.rs index a50dc584..c993ee91 100644 --- a/src/reverse.rs +++ b/src/reverse.rs @@ -362,6 +362,7 @@ fn forms_for(lemma: &str, lang: Lang) -> Option> { Lang::Fra => fra_forms(lemma), Lang::Spa => spa_forms(lemma), Lang::Gle => gle_forms(lemma), + Lang::Gla => gla_forms(lemma), Lang::Deu => deu_forms(lemma), _ => conjugate(lemma, lang).ok().map(|t| enumerate(&t)), } @@ -553,6 +554,42 @@ fn gle_forms(lemma: &str) -> Option> { Some(s.0) } +fn gla_forms(lemma: &str) -> Option> { + use crate::gla::{Slot, Tense, Verb}; + let v = Verb::from_infinitive(lemma).ok()?; + let mut s = Slots(Vec::new()); + s.one(v.infinitive(), "infinitive"); + s.one(&v.verbal_noun(), "verbal noun"); + s.one(&v.verbal_adjective(), "verbal adjective"); + // The past and conditional lenite in the written standard + // (glan → ghlan). Mirrors gla::Table::row. + for (t, name) in [ + (Tense::Past, "past"), + (Tense::Future, "future"), + (Tense::RelativeFuture, "relative future"), + (Tense::Conditional, "conditional"), + (Tense::Imperative, "imperative"), + ] { + let mutate = matches!(t, Tense::Past | Tense::Conditional); + for (slot, l) in [ + (Slot::Independent, ""), + (Slot::Dependent, " dependent"), + (Slot::FirstSingular, " 1sg"), + (Slot::SecondSingular, " 2sg"), + (Slot::FirstPlural, " 1pl"), + (Slot::SecondPlural, " 2pl"), + (Slot::Third, " 3"), + (Slot::Impersonal, " impersonal"), + ] { + if let Some(f) = v.form(t, slot) { + let f = if mutate { Verb::lenite(&f) } else { f }; + s.one(&f, &format!("{name}{l}")); + } + } + } + Some(s.0) +} + // --------------------------------------------------------------------- // Candidate generation (productive languages only). // --------------------------------------------------------------------- @@ -818,13 +855,17 @@ fn ord(lang: Lang) -> usize { Lang::Amh => 56, Lang::Ind => 57, Lang::Zul => 58, + Lang::Epo => 59, + Lang::Gla => 60, + Lang::Grn => 61, + Lang::Haw => 62, } } fn index(lang: Lang) -> &'static Index { #[allow(clippy::declare_interior_mutable_const)] const EMPTY: OnceLock = OnceLock::new(); - static INDEXES: [OnceLock; 59] = [EMPTY; 59]; + static INDEXES: [OnceLock; 63] = [EMPTY; 63]; INDEXES[ord(lang)].get_or_init(|| build_index(lang)) } @@ -854,7 +895,7 @@ fn build_index(lang: Lang) -> Index { fn is_lexicon_lemma(cand: &str, lang: Lang) -> bool { #[allow(clippy::declare_interior_mutable_const)] const EMPTY: OnceLock> = OnceLock::new(); - static SETS: [OnceLock>; 59] = [EMPTY; 59]; + static SETS: [OnceLock>; 63] = [EMPTY; 63]; SETS[ord(lang)] .get_or_init(|| lexicon_lemmas(lang).into_iter().collect()) .contains(cand) @@ -981,6 +1022,10 @@ fn lexicon_lemmas(lang: Lang) -> Vec<&'static str> { Lang::Amh => col1(include_str!("../data/amh/parts.tsv"), &mut lemmas), Lang::Ind => col1(include_str!("../data/ind/parts.tsv"), &mut lemmas), Lang::Zul => col1(include_str!("../data/zul/parts.tsv"), &mut lemmas), + Lang::Epo => col1(include_str!("../data/epo/parts.tsv"), &mut lemmas), + Lang::Gla => col1(include_str!("../data/gla/verbs.tsv"), &mut lemmas), + Lang::Grn => col1(include_str!("../data/grn/parts.tsv"), &mut lemmas), + Lang::Haw => col1(include_str!("../data/haw/parts.tsv"), &mut lemmas), } lemmas.sort_unstable(); lemmas.dedup(); @@ -1854,6 +1899,49 @@ fn enumerate(c: &Conjugation) -> Vec<(String, String)> { s.opt(Some(f), "form"); } } + Conjugation::Epo(t) => { + s.one(&t.infinitive, "citation"); + for f in [&t.present, &t.past, &t.future, &t.conditional, &t.volitive] { + s.one(f, "form"); + } + for f in t + .active_present + .iter() + .chain(&t.active_past) + .chain(&t.active_future) + .chain(&t.passive_present) + .chain(&t.passive_past) + .chain(&t.passive_future) + { + s.opt(Some(f), "form"); + } + } + Conjugation::Gla(t) => { + s.one(&t.lemma, "infinitive"); + s.one(&t.verbal_noun, "verbal noun"); + s.one(&t.verbal_adjective, "verbal adjective"); + s.row_opt(&t.past, "past", &P7); + s.row_opt(&t.future, "future", &P7); + s.row_opt(&t.relative_future, "relative future", &P7); + s.row_opt(&t.conditional, "conditional", &P7); + s.row_opt(&t.imperative, "imperative", &P7); + } + Conjugation::Grn(t) => { + let p7 = ["1sg", "2sg", "3sg", "1pl incl", "1pl excl", "2pl", "3pl"]; + s.row(&t.indicative, "indicative", &p7); + s.row(&t.hortative, "hortative", &p7); + s.row(&t.imperative, "imperative", &p7); + s.row(&t.passive, "passive", &p7); + s.row(&t.reciprocal, "reciprocal", &p7); + s.row(&t.coactive, "coactive", &p7); + s.row(&t.objective, "objective", &p7); + } + Conjugation::Haw(t) => { + s.one(&t.stem, "citation"); + for f in [&t.causative, &t.reduplicated, &t.passive] { + s.opt(f.as_ref(), "form"); + } + } } s.0 } diff --git a/src/wasm.rs b/src/wasm.rs index d11aaf46..f23d847c 100644 --- a/src/wasm.rs +++ b/src/wasm.rs @@ -225,6 +225,26 @@ pub fn conjugate(infinitive: &str, lang: Option) -> Result { + let v = crate::epo::Verb::from_infinitive(infinitive) + .map_err(|e| JsError::new(&e.to_string()))?; + Ok(serde_wasm_bindgen::to_value(&crate::epo::Table::build(&v))?) + } + Some(crate::Lang::Gla) => { + let v = crate::gla::Verb::from_infinitive(infinitive) + .map_err(|e| JsError::new(&e.to_string()))?; + Ok(serde_wasm_bindgen::to_value(&crate::gla::Table::build(&v))?) + } + Some(crate::Lang::Grn) => { + let v = crate::grn::Verb::from_lemma(infinitive) + .map_err(|e| JsError::new(&e.to_string()))?; + Ok(serde_wasm_bindgen::to_value(&crate::grn::Table::build(&v))?) + } + Some(crate::Lang::Haw) => { + let v = crate::haw::Verb::from_infinitive(infinitive) + .map_err(|e| JsError::new(&e.to_string()))?; + Ok(serde_wasm_bindgen::to_value(&crate::haw::Table::build(&v))?) + } Some(crate::Lang::Bul) => { let v = crate::bul::Verb::from_infinitive(infinitive) .map_err(|e| JsError::new(&e.to_string()))?;