diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index 57414d7e..68cfe2cf 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -950,5 +950,36 @@ jobs: run: ./scripts/amh/fetch_unimorph.sh - name: Amharic golden harness with single-oracle regression gate (Beta) run: cargo run --release --bin golden_amh -- data/amh/unimorph.tsv --check + - name: Cache UniMorph ind + id: unimorph-ind + uses: actions/cache@v5 + with: + path: data/ind/unimorph.tsv + key: unimorph-ind-v1 + - name: Fetch UniMorph ind + if: steps.unimorph-ind.outputs.cache-hit != 'true' + run: ./scripts/ind/fetch_unimorph.sh + - name: Indonesian golden harness with single-oracle regression gate (Beta) + run: cargo run --release --bin golden_ind -- data/ind/unimorph.tsv --check + - name: Cache UniMorph zul + id: unimorph-zul + uses: actions/cache@v5 + with: + path: data/zul/unimorph.tsv + key: unimorph-zul-v1 + - name: Fetch UniMorph zul + if: steps.unimorph-zul.outputs.cache-hit != 'true' + run: ./scripts/zul/fetch_unimorph.sh + - name: Cache kaikki zul + id: kaikki-zul + uses: actions/cache@v5 + with: + path: data/zul/kaikki.tsv + key: kaikki-zul-v1 + - name: Fetch kaikki zul + if: steps.kaikki-zul.outputs.cache-hit != 'true' + run: ./scripts/zul/fetch_kaikki.sh + - name: Zulu golden harness with two-oracle regression gate + run: cargo run --release --bin golden_zul -- data/zul/unimorph.tsv data/zul/kaikki.tsv --check - name: Reverse-lookup gate (fra, spa, eng; deu needs the kaikki dump) run: cargo run --release --bin reverse_gate -- --check --langs fra,spa,eng diff --git a/Cargo.toml b/Cargo.toml index cb71110a..90299f45 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -68,6 +68,10 @@ include = [ "data/heb/overrides.tsv", "data/amh/parts.tsv", "data/amh/overrides.tsv", + "data/ind/parts.tsv", + "data/ind/overrides.tsv", + "data/zul/parts.tsv", + "data/zul/overrides.tsv", "data/por/classes.tsv", "data/por/verbs.tsv", "data/ron/classes.tsv", diff --git a/data/ind/overrides.tsv b/data/ind/overrides.tsv new file mode 100644 index 00000000..9dee78ec --- /dev/null +++ b/data/ind/overrides.tsv @@ -0,0 +1,4 @@ +# lemma features form +galak V;ACT berpenggalak +sakit V;ACT berpenyakit +tampil V;ACT berpenampilan diff --git a/data/ind/parts.tsv b/data/ind/parts.tsv new file mode 100644 index 00000000..8586b67e --- /dev/null +++ b/data/ind/parts.tsv @@ -0,0 +1,2663 @@ +abad +abadi +abai +abdi +absah +absen +acak +acara +acu +acuh +ada +adab +adaptasi +adat +adegan +adik +adil +administrasi +adopsi +adu +aduk +afiliasi +agama +agenda +agun +agung +ahli +air +ajak +ajar +aju +akal +akar +akhir +akhlak +akibat +akomodasi +akomodir +akrab +aksara +akses +aksi +aktif +aktivitas +aku +akuisisi +akumulasi +ala +alamat +alami +alas +alasan +alat +aliansi +alih +alir +alis +alkohol +alokasi +alun +alur +amal +aman +amanat +amandemen +ambang +ambil +ambisi +amendemen +amin +ampas +amplop +ampun +amuk +anak +analis +analisa +analisis +analog +ancam +andai +andal +andil +aneka +angan +anggap +anggar +anggota +angguk +anggur +angin +angkasa +angkat +angkut +angsur +aniaya +anjur +antar +anti +anting +antisipasi +antusias +anugerah +anulir +anut +anyam +api +apit +aplikasi +apresiasi +apung +arah +arak +argumen +argumentasi +aroma +arsip +arsitektur +arti +arus +asa +asah +asal +asam +asap +asas +asin +asing +asosiasi +aspek +aspirasi +asrama +asuh +asumsi +asuransi +asyik +atap +atas +atom +atur +audiensi +audit +aura +awak +awal +awas +awet +ayah +ayak +ayun +babat +babi +baca +badan +badar +bagi +bagus +bahagia +bahan +bahana +baharu +bahas +bahasa +bahaya +bahu +baik +bajak +baju +bakar +bakat +bakti +baku +bala +balap +balas +bali +balik +baling +bandel +banding +bangga +bangkang +bangkit +bangkrut +bangsa +bangun +banjir +bantah +bantai +banting +bantu +banyak +bara +barat +baret +barikade +baring +baris +baru +basah +basis +basmi +basuh +bata +batal +batang +batas +batu +bau +baur +baut +bawa +bawah +bayang +bayar +beban +bebas +beber +beda +bedah +bekal +bekas +bekerja +beku +bekuk +bela +belah +belajar +belakang +belanja +belenggu +beli +belit +belok +belot +benah +benam +benar +benci +bendera +bendung +bengkak +bengkok +bentang +bentrok +bentuk +bentur +berangkat +berangus +berani +berantas +berat +bercak +beres +berhala +berhasil +beri +berita +berkat +berondong +berontak +bersih +besar +betul +biak +biar +biasa +biaya +bibir +bicara +bidan +bidang +bidik +biji +bikin +bilah +bilang +bilas +bilik +bimbing +bina +binasa +bincang +bingkai +bingung +bintang +bintik +bintil +biru +bisa +bisik +bisnis +bisu +bius +blok +blokade +blokir +bobol +bobot +bocor +bodoh +bohong +boikot +boleh +bolos +bom +bombardir +bondong +bongkar +borong +boros +bosan +buah +buai +buang +buat +bubar +budak +budaya +budi +bujuk +bujur +buka +bukit +bukti +buku +bulan +bulat +bulir +bulu +bumbu +bumbung +bumi +bundar +bunga +bungkam +bungkuk +bungkus +buntut +bunuh +bunyi +buru +buruk +busa +busana +busuk +buta +butir +butuh +buyar +cabang +cabik +cabut +cacah +cacat +caci +cadang +cahaya +cair +cakap +cakar +cakup +calon +camat +cambuk +campak +campur +canang +canda +canggih +cangkang +cangkok +cangkul +cantik +cantum +cap +capai +caplok +cara +cari +catat +cebur +cedera +cegah +cegat +cek +cekam +cekik +cela +celah +celaka +celana +celup +cemar +cemas +cengang +cengkeram +cepat +cerah +cerai +ceramah +cerca +cerdas +cerita +cermat +cermin +cerna +cetak +cetus +cicil +cicip +cicit +cincang +cincin +cinta +ciprat +cipta +ciri +cita +citra +cium +ciut +coba +cocok +colok +condong +contek +contoh +corak +coreng +coret +cuaca +cuat +cuci +cucur +cuka +cukup +cukur +culik +curah +curang +curi +curiga +dada +daftar +dagang +daging +daki +dakwa +dalam +dalang +dalih +damai +damba +dampak +damping +dana +danau +dansa +dapat +darah +darat +dasar +dasawarsa +dasi +data +datang +datar +daulat +daun +daur +daya +dayung +debar +debat +debu +debut +dedikasi +definisi +degradasi +dekade +dekam +dekat +deklamasi +deklarasi +dekrit +delegasi +demo +demonstrasi +denda +dendam +dengar +dengkur +dengung +denyut +depak +depan +deportasi +dera +derajat +deret +derit +derita +derma +desa +desah +desain +desak +desentralisasi +deskripsi +detail +detak +deteksi +detil +devaluasi +dewa +dewasa +diagnosa +diagnosis +dialog +diam +diameter +didih +didik +difusi +digital +dikte +dinas +dinding +dingin +diplomasi +diri +disiplin +diskredit +diskriminasi +diskusi +distribusi +diversifikasi +divestasi +doa +dobrak +doktrin +dokumen +dokumentasi +dominasi +domisili +donasi +dongak +dongeng +dongkrak +dorong +dosa +duda +duduk +duga +duka +dukung +dulang +dunia +duplikasi +durasi +duri +dusta +edar +edit +efek +efektif +efisien +efisiensi +ejek +ekonomi +ekor +eksekusi +ekspansi +eksperimen +eksploitasi +eksplorasi +ekspor +ekspos +ekspresi +ekstradisi +ekstrak +ekstraksi +elaborasi +elak +elektrik +elektron +elemen +eliminasi +eliminir +elu +emas +emban +embun +emigrasi +emisi +enak +encer +endus +enkripsi +entas +enyah +eram +erang +erat +esa +esens +estimasi +etika +etnik +etnis +evakuasi +evaluasi +evolusi +fasilitas +favorit +fermentasi +film +filter +firdaus +firman +fisik +fitnah +fitur +fluktuasi +fokus +format +formulasi +fosil +foto +fotosintesis +fraksi +fungsi +gabung +gadai +gagal +gagang +gagas +gairah +gaji +galak +galang +gali +gambar +gampang +ganda +gandeng +ganggu +ganjal +ganjar +ganti +gantung +gapai +garam +garap +garis +gas +gaul +gaya +gebrak +gegas +gejala +gejolak +gelandang +gelang +gelantung +gelap +gelar +geledah +gelegak +gelegar +gelembung +geli +geliat +gelimpang +gelinding +gelisah +gelitik +gelombang +gelora +gelut +gema +gemar +gembala +gembira +gembleng +gembung +geming +gempar +gempur +gemuk +gemuruh +gen +genang +genap +gencar +gencet +gendong +generasi +genggam +gengsi +genjot +gerai +gerak +geram +gerebek +gereja +gerigi +gerilya +gerus +gerutu +geser +getah +getar +giat +gigi +gigit +gila +giling +gilir +giring +gitar +giur +gizi +goda +godok +gol +golak +golong +goncang +goreng +gosip +gosok +goyah +goyang +gratis +gubris +gugah +gugat +gugur +gugus +gula +gulat +guling +gulir +gulung +gumam +gumpal +gumul +guna +guncang +gundul +gunting +gunung +gurau +guru +gurun +gusar +gusur +habib +habis +habitat +hadang +hadap +hadiah +hadir +hafal +hajar +hak +hakim +halal +halaman +halang +halau +halus +hamba +hambat +hamil +hampar +hancur +hangat +hangus +hantam +hantar +hantu +hanyut +hapus +haram +harap +hardik +harga +hari +harmonis +harta +harum +harus +hasil +hasrat +hasut +hati +hawa +hayat +hebat +heboh +hektar +hela +hemat +hembus +hening +henti +heran +hewan +hias +hibah +hibur +hidang +hidung +hidup +hijab +hijau +hijrah +hikmat +hilang +himbau +himpun +hina +hindar +hipotesis +hirau +hirup +hisap +hitam +hitung +hormat +hubung +hujan +hujat +hukum +hulu +huni +huruf +hutan +hutang +ibadah +ibarat +ibu +idap +identifikasi +identik +identitas +igau +ikan +ikat +ikhlas +iklan +iklim +ikrar +ikut +ilham +ilmu +ilustrasi +imajinasi +imam +iman +imbang +imbas +imigrasi +implementasi +implikasi +impor +improvisasi +incar +indah +indeks +indera +indikasi +indra +induk +induksi +infeksi +informasi +ingat +ingin +ingkar +inisial +inisiatif +injak +injeksi +inspirasi +instalasi +institusi +instruksi +intai +integrasi +integritas +intens +intensif +intensitas +interaksi +interogasi +interpretasi +interupsi +intervensi +inti +intimidasi +invasi +investasi +investigasi +ionisasi +irama +iri +iring +iris +isak +isap +isi +isolasi +istana +istilah +istimewa +istirahat +istri +isyarat +izin +jabar +jabat +jadi +jadwal +jaga +jago +jahit +jaja +jajah +jajak +jajal +jajar +jaket +jala +jalan +jalar +jalin +jalur +jam +jambak +jamin +jamu +jamur +janda +jangka +jangkau +jangkit +janji +jarah +jarak +jari +jaring +jas +jasa +jatuh +jauh +jawab +jaya +jebak +jeblos +jebol +jegal +jejak +jejal +jelajah +jelas +jelek +jelma +jemaah +jemaat +jembatan +jemput +jemu +jendela +jengkel +jenguk +jenis +jenjang +jentik +jenuh +jepit +jerat +jerit +jerumus +jibaku +jihad +jijik +jilat +jinak +jinjing +jiwa +jual +juang +juara +jubah +judi +judul +julang +juluk +julur +jumlah +jumpa +jungkir +junjung +juntai +jurus +justifikasi +kabar +kabel +kabil +kabul +kabur +kabut +kaca +kacau +kadar +kaget +kagum +kain +kait +kaji +kaki +kaku +kala +kalah +kali +kalung +kamar +kamera +kampanye +kampung +kamuflase +kanan +kancing +kandang +kandas +kandung +kantong +kantor +kantung +kapal +kapasitas +kapsul +karakter +karakterisasi +karakteristik +karang +karantina +karat +karbohidrat +karbon +karier +karung +karunia +karya +kasih +kata +katalis +kategori +kategorisasi +kaum +kawah +kawal +kawan +kawin +kaya +kayak +kayu +kayuh +kebiri +kebun +kecam +kecambah +kecap +kecenderungan +kecewa +kecil +kedip +kehendak +kejar +kejut +kekal +kekang +kelahi +kelakar +kelamin +kelas +kelenjar +keliling +keliru +kelola +kelompok +kelopak +keluar +keluarga +keluh +kelupas +kemas +kembali +kembang +kembara +kemih +kemudi +kemungkinan +kena +kenal +kenan +kenang +kencan +kencang +kendala +kendali +kendara +kendor +kendur +kening +kental +kepak +kepal +kepala +keping +kepung +kera +kerabat +kerah +keran +kerangka +keras +kereta +kerikil +kering +keringat +keriput +keritik +kerja +keroyok +kerucut +keruh +keruk +kerumun +kerut +kesal +kesan +kesempatan +ketat +keterampilan +ketik +ketua +ketuk +khas +khasiat +khawatir +khayal +khianat +khitan +khotbah +khusus +kibar +kibas +kidung +kikis +kilah +kilap +kilat +kilau +kilo +kinerja +kipas +kiprah +kira +kirim +kisah +kisar +kitar +klaim +klarifikasi +klasifikasi +klik +kloning +klorofil +koalisi +kobar +kocok +kode +kokoh +kolaborasi +koleksi +kolom +komando +kombinasi +komentar +komersial +komitmen +kompeten +kompetensi +komplain +komposisi +kompromi +komunikasi +komunitas +kondisi +koneksi +konfigurasi +konfirmasi +konflik +kongsi +konsentrasi +konsep +konsolidasi +konspirasi +konsultasi +konsumsi +kontak +kontrak +kontraksi +kontras +kontribusi +kontrol +koordinasi +koordinir +koper +koperasi +korban +korek +koreksi +korelasi +kosong +kostum +kota +kotak +kotor +koyak +kreasi +kreativitas +kredit +kritik +kritis +kuak +kualifikasi +kualitas +kuas +kuasa +kuat +kubu +kubur +kucil +kucur +kuda +kudeta +kudus +kuku +kukuh +kukus +kuliah +kulit +kumandang +kumis +kumpul +kunci +kungkung +kuning +kunjung +kunyah +kupas +kuping +kurang +kuras +kurikulum +kurung +kusut +kutil +kutip +kutub +kutuk +laba +label +labrak +labuh +lacak +lacur +ladang +laga +lagak +lagu +lahan +lahir +lain +laju +laknat +laksana +laku +lalai +lalap +lalu +lama +lamar +lambai +lambang +lambat +lambung +lampau +lampir +lampu +lancar +landa +landas +langgan +langgar +langgeng +langit +langka +langkah +langsung +lanjut +lansir +lantai +lantik +lantun +lapang +lapis +lapor +larang +lari +larut +latar +latih +laut +lawan +layan +layang +layar +lebar +lebih +lebur +leceh +ledak +lega +legal +legenda +legitimasi +leher +lejit +lekat +lelah +lelang +leleh +lemah +lemak +lembab +lembaga +lembah +lembar +lembut +lempar +lempem +lenceng +lendir +lengan +lengas +lengkap +lengket +lengkung +lengser +lensa +lenting +lentur +lenyap +lepas +lereng +lesak +lestari +lesu +letak +letih +letus +lewat +liar +liat +libas +libat +liberalisasi +libur +licin +lidah +lihat +liku +likuidasi +lilin +lilit +limpah +lindung +lingkar +lingkung +lingkup +lintas +lipat +lipit +liput +lirik +lisan +lisensi +listrik +liuk +lobang +lobi +logo +lokalisasi +lokalisir +lokasi +lolos +lomba +lompat +loncat +longgar +lonjak +lontar +lorong +luap +luar +luas +lubang +luber +lucu +lucut +ludah +luka +lukis +luluh +lulus +lumat +lumpuh +lumpur +lumur +lunak +lunas +luncur +luntur +lupa +luput +luruh +lurus +lutut +maaf +mabuk +macam +macet +madam +madu +magnitudo +magrib +main +maju +makam +makan +makmur +makna +maksiat +maksimal +maksimum +maksud +maktub +malam +malas +malu +mama +mampu +mana +manajemen +mandat +mandi +mandul +manfaat +manipulasi +manis +manja +mantap +manusia +manuver +marah +marga +markas +martabat +masa +masak +masalah +masuk +masyarakat +mata +matang +materi +mati +mayoritas +medan +media +mediasi +meditasi +megah +mekar +membran +menang +menanti +menara +meniran +menit +mentah +mental +mentega +menyerah +merah +merdeka +merek +meriah +meriam +mesin +mesra +metabolisme +metamorfosis +mewah +migrasi +mil +milik +mimpi +minat +minimal +minimum +minta +minum +minyak +miring +mirip +misal +miskin +mitra +mobilisasi +modal +model +modern +modernisasi +modifikasi +mohon +moncong +monitor +monopoli +moral +motif +motivasi +motor +muara +muat +muda +mudah +muka +mukim +mula +mulai +mulia +mulus +mulut +mumi +muncul +mundur +mungkir +muntah +mur +murah +murni +murtad +musik +musim +musnah +mustahil +musuh +musyawarah +mutakhir +mutasi +mutu +nada +nafas +nafsu +naik +nalar +nama +nanti +napas +nasehat +nasib +nasihat +nasional +nasionalisasi +naskah +naung +negara +negeri +negosiasi +netral +netralisir +ngeri +niat +nihil +nikah +nikmat +nilai +noda +nominasi +nomor +nonaktif +norma +normal +normalisasi +nuansa +nubuat +nyala +nyaman +nyanyi +nyata +nyawa +obat +observasi +obsesi +oksida +oksidasi +oksigen +oktan +olah +olahraga +oles +olok +omel +operasi +operasional +oplah +optimal +orbit +orde +organisasi +orientasi +otak +otomatis +otonomi +otoritas +otot +pacar +pacu +padat +padu +pagar +paham +pahat +pajang +pakai +paket +paksa +paku +pala +palang +paling +palsu +pamer +panah +panas +pancang +pancar +pancing +pandang +pandu +panen +panggang +panggil +panggung +pangkal +pangkas +pangkat +pangku +pangsa +panik +panjang +panjat +pantau +pantul +papan +papar +paradigma +parah +parit +parkir +parlemen +partai +partisipasi +paruh +parut +pasak +pasang +pasar +pasir +pasok +pasti +pasukan +patah +paten +patok +patroli +patuh +patung +paut +pawai +payung +pecah +pecat +pedang +pedoman +peduli +pegang +pegunungan +pekak +pekan +pekerja +pekik +pelajar +pelat +pelihara +pelintir +peluang +peluk +peluru +pembuluh +pendam +pendar +pendek +penduduk +pengalaman +pengaruh +pengetahuan +penjara +penting +penuh +perahu +peran +perang +perangkat +peras +percaya +percik +performa +pergi +pergok +periksa +perilaku +peringkat +perintah +perkara +perkosa +perlu +permanen +peroleh +pers +perut +pesan +pesat +pesiar +pesona +pesta +peta +petak +peti +petik +piagam +picu +pidato +pigmen +pihak +pikat +pikir +pikul +pil +pilah +pilar +pilih +pilot +pimpin +pinang +pindah +pindai +pinggir +pinjam +pinta +pintas +pintu +pipih +pisah +pita +piutang +pohon +pojok +pola +politik +pompa +populer +pori +pos +posisi +potensi +potensial +potong +potret +prakarsa +prakira +praktek +praktik +prasangka +predikat +prediksi +presentasi +prestasi +pribadi +prihatin +prinsip +prioritas +produksi +profesi +profil +program +promosi +propaganda +proses +prospek +protein +protes +provokasi +proyeksi +puas +puasa +publikasi +pucat +pucuk +pudar +puja +puji +pukau +pukul +pulang +pulih +punah +puncak +punggung +pungut +punuk +punya +pupus +puruk +pusat +pusing +putar +putih +putra +putus +racun +radiasi +raga +ragam +ragi +ragu +rahang +rahasia +rahmat +raih +raja +rajalela +rajam +rajang +rajut +rakit +rakyat +ramah +ramai +ramal +rambah +rambat +rambut +rampas +ramping +rampok +rampung +ramu +rancang +rancu +rangka +rangkai +rangkak +rangkap +rangkul +rangkum +rangsang +rantai +rapat +rapi +rasa +rasional +rasuk +rasul +rata +ratap +ratifikasi +raup +rawa +rawan +rawat +raya +rayap +rayu +reaksi +realisasi +rebak +rebut +reda +redam +reduksi +redup +referensi +refleksi +regang +rehabilitasi +reka +rekam +rekan +rekat +rekayasa +rekomendasi +rekonsiliasi +rekonstruksi +rekreasi +rekrut +rela +relief +relokasi +rem +remaja +rembes +rembet +remeh +renang +rencana +rendah +rendam +renggang +renggut +renovasi +rentang +renung +repot +representasi +reproduksi +resah +resep +resmi +respons +restorasi +restrukturisasi +restu +retak +revisi +revolusi +revolusioner +ria +riang +riba +ribut +rilis +rinci +rindu +ringan +ringkas +ringkuk +rintang +rintis +risau +riset +risiko +ritme +robek +roboh +roda +rogoh +roket +rokok +rombak +rombong +rona +ronda +rongga +rontok +rotasi +ruas +rubah +rubuh +rugi +rujuk +rukun +rumah +rumit +rumpun +rumput +rumus +runding +runtuh +rupa +rusak +rusuh +rusuk +rute +sabar +sabda +sabet +sabot +sabotase +sadar +sah +sahabat +sahut +saing +saji +sakit +saksi +salah +salam +salat +salib +salin +salju +salur +sama +samar +sambang +sambar +sambung +sambut +sampai +sampan +samping +sampul +sana +sandang +sandar +sandera +sandi +sanding +sandung +sangga +sanggah +sanggup +sangka +sangkal +sangkut +sangsi +sanjung +santai +santap +santun +sapa +sapu +saran +sarang +sari +saring +sasar +saudara +sawah +sayang +sayap +sayat +sebab +sebagian +sebal +sebar +seberang +sebut +sedan +sedap +sedekah +sederhana +sedia +sedih +sedot +segan +segar +segel +segi +segmen +sehat +seimbang +sejahtera +sejajar +sejarah +sejuk +sekap +sekat +sekolah +seks +sekutu +sel +sela +selam +selamat +selancar +selang +selaras +seleksi +selenggara +selera +selesai +selidik +selimut +selinap +selip +selisih +selubung +seludup +selundup +selusup +semai +semak +semangat +semarak +semayam +sembah +sembahyang +sembelih +sembuh +sembunyi +sembur +sempit +semprot +sempurna +semu +semut +senang +senapan +senda +sendi +sendiri +sendok +sengat +sengketa +sengsara +seni +senjata +sensor +sentak +sentuh +senyawa +senyum +sepak +sepakat +sepatu +sepeda +sepele +seragam +serah +serak +seram +serang +serap +serat +serbu +serbuk +seret +sergap +seri +serikat +seringai +serobot +sertifikat +seru +serupa +seruput +servis +sesal +sesat +sesuai +setara +setel +setia +setir +setor +setuju +sewa +sia +siaga +siang +siap +siar +siasat +sibak +sibuk +sidang +sidik +sifat +sihir +sikap +sikat +siklus +siksa +sikut +silahkan +silakan +silang +simak +simbah +simbiosis +simbol +simpan +simpang +simpati +simpul +sinar +sindir +sinergi +singgung +singkap +singkat +singkir +singsing +sintesis +sinyal +sinyalir +siram +sirat +sirip +sisa +sisi +sisih +sisip +sisir +sistem +sita +sitir +siul +skala +skenario +slogan +soal +sobek +soda +sodor +sokong +solo +sombong +sorak +sorot +sortir +sosial +spektrum +spekulasi +sponsor +stabil +standar +status +steril +stimulasi +strata +struktur +suami +suap +suara +suasana +subsidi +subur +suci +sudah +sudut +suguh +suhu +sujud +suka +sukacita +sukaria +sukses +suku +sulam +sulap +suling +sulit +sulut +sumbang +sumbat +sumber +sumpah +sunah +sundul +sungkur +suntik +suplai +surat +suruh +surut +survei +susah +suspensi +susu +susul +susun +susup +susur +susut +sutradara +swasembada +swasta +syafaat +syair +syarat +taat +tabrak +tabung +tabur +tafakur +tafsir +tagih +tahan +tahap +tahta +tahu +tahun +tajam +tajuk +takar +takhta +takjub +takluk +taksir +takut +takwa +talenta +tali +taman +tambah +tambak +tambal +tambang +tampak +tampar +tampik +tampil +tampung +tamu +tanah +tanam +tancap +tanda +tandang +tandas +tanding +tangan +tangga +tanggal +tanggap +tangguh +tangguk +tanggulang +tanggung +tangis +tangkai +tangkal +tangkap +tangkis +tani +tanjak +tantang +tanya +tapak +target +tari +tarif +tarik +taring +taruh +tarung +tasbih +tata +tatap +tauhid +taut +tawa +tawan +tawar +tayang +tebak +tebal +tebang +tebar +tebus +teduh +tegak +tegang +tegas +teguh +tegur +tekad +tekan +tekel +teknik +tekuk +tekun +telaah +teladan +telan +telanjang +telefon +telepon +telinga +teliti +telpon +telur +tema +teman +tembak +tembok +tembus +tempa +tempat +tempel +tempo +tempuh +tempur +temu +tenaga +tenang +tenar +tendang +tendensi +tengah +tenggang +tenggelam +tengkar +tengok +tentang +tenteram +tentu +tenun +tenung +tepat +tepi +tepis +tepuk +tepung +terang +terap +terbang +terbit +teriak +terima +terjadi +terjang +terjemah +terjun +terka +terkam +terlantar +ternak +terobos +teror +terpa +terpenuhi +tersenyum +tertib +terus +tes +tetangga +tetap +tetas +tetes +tewas +tiada +tiang +tidur +tiket +tilik +timba +timbal +timbang +timbul +timbun +timpa +tindak +tindas +tindih +tinggal +tinggi +tingkah +tingkat +tinjau +tinju +tipe +tipis +tipu +tiris +tiru +titah +titel +titik +titip +tiup +tobat +todong +tol +tolak +toleh +toleransi +tolerir +tolong +tongkat +tonjol +tonton +top +topang +topeng +topik +toreh +transaksi +transfer +transformasi +tua +tuai +tuan +tuang +tubruk +tubuh +tuding +tuduh +tugas +tuju +tukang +tukar +tukik +tulang +tular +tulis +tumbang +tumbuh +tumbuk +tumis +tumpah +tumpang +tumpas +tumpu +tumpuk +tumpul +tunai +tunang +tunda +tunduk +tunggak +tunggal +tunggang +tunggu +tunjang +tunjuk +tuntas +tuntun +tuntut +turun +turut +tusuk +tutup +tutur +uang +uap +ubah +uban +ucap +udang +udara +ujar +uji +ujung +ukir +ukur +ulah +ulang +ular +ulas +ulir +ulur +umbi +umpama +umpan +umpat +umum +umur +undang +undi +unduh +undur +unggah +unggul +ungkap +ungkit +ungsi +unjuk +unsur +unta +untung +upah +upaya +urai +urat +urung +urus +urut +usah +usaha +usai +usia +usik +usir +usul +usung +utama +utang +utara +utus +variabel +variasi +vegetarian +verifikasi +veto +vokal +volume +vonis +wabah +wacana +wadah +wahyu +wajah +wajib +wakil +waktu +wali +wangi +waralaba +warga +waris +warna +wasiat +wasit +waspada +watak +wawancara +wawasan +wenang +wewenang +wibawa +wilayah +wiraswasta +wirausaha +wisata +wujud +yakin +zakat +zikir +zina diff --git a/data/zul/overrides.tsv b/data/zul/overrides.tsv new file mode 100644 index 00000000..7dc1fa91 --- /dev/null +++ b/data/zul/overrides.tsv @@ -0,0 +1,3 @@ +# Zulu per-cell overrides: lemma ⇥ canonical-features ⇥ form +# Consulted before the productive rules. Empty for now: the parts.tsv +# root remaps plus the productive template cover the agreement gold. diff --git a/data/zul/parts.tsv b/data/zul/parts.tsv new file mode 100644 index 00000000..c1fd8ec1 --- /dev/null +++ b/data/zul/parts.tsv @@ -0,0 +1,17 @@ +# Zulu principal parts: lemma ⇥ conjugation-root ⇥ mono ⇥ vowel_initial +# +# The productive template in src/zul.rs derives the whole paradigm from +# the bare stem (the UniMorph/kaikki lemma) by rule: the conjugation root +# is the lemma, monosyllabicity is one-vowel, vowel-initial is a +# vowel-first stem. A row overrides those defaults ("-" = derive). Only +# the residue that the rules cannot reach is listed here. +# +# The i-augment verbs (iza "come", ima "stand", imba "dig", izwa "hear/ +# feel") carry a citation i- that drops in the conjugation (ukuza, ukuma, +# ukumba, ukuzwa; ngaza, ngama), so their real roots are the augmentless +# stems. iza/ima are then monosyllabic. +lemma root mono vowel_initial +iza za 1 0 +ima ma 1 0 +imba mba 0 0 +izwa zwa 1 0 diff --git a/docs/ind/adjudications.tsv b/docs/ind/adjudications.tsv new file mode 100644 index 00000000..9af0364d --- /dev/null +++ b/docs/ind/adjudications.tsv @@ -0,0 +1 @@ +# lemma features chosen note diff --git a/docs/zul/adjudications.tsv b/docs/zul/adjudications.tsv new file mode 100644 index 00000000..9af0364d --- /dev/null +++ b/docs/zul/adjudications.tsv @@ -0,0 +1 @@ +# lemma features chosen note diff --git a/docs/zul/disagreements.tsv b/docs/zul/disagreements.tsv new file mode 100644 index 00000000..585a0249 --- /dev/null +++ b/docs/zul/disagreements.tsv @@ -0,0 +1,4 @@ +# lemma features resolution note +# ona V;NFIN: UniMorph ukwona vs kaikki ukona — glide vs no-glide before +# the back vowel o. Both are attested spellings; left unresolved (the +# single oracle-disagreement slot, excluded from the scored gold). diff --git a/scripts/correctness.py b/scripts/correctness.py index a6e766df..0470fcd6 100644 --- a/scripts/correctness.py +++ b/scripts/correctness.py @@ -37,7 +37,7 @@ "hin": "Hindi", "swa": "Swahili", "tam": "Tamil", "tel": "Telugu", "tgl": "Tagalog", "pes": "Persian", "kan": "Kannada", "guj": "Gujarati", "urd": "Urdu", "ben": "Bengali", "mar": "Marathi", - "mkd": "Macedonian", "afr": "Afrikaans", "bul": "Bulgarian", "ell": "Greek", "sqi": "Albanian", "pol": "Polish", "aze": "Azerbaijani", "uzb": "Uzbek", "tuk": "Turkmen", "bel": "Belarusian", "cym": "Welsh", "fao": "Faroese", "glg": "Galician", "kaz": "Kazakh", "lat": "Latin", "ltz": "Luxembourgish", "oci": "Occitan", "tat": "Tatar", "ydd": "Yiddish", "ara": "Arabic", "heb": "Hebrew", "amh": "Amharic", + "mkd": "Macedonian", "afr": "Afrikaans", "bul": "Bulgarian", "ell": "Greek", "sqi": "Albanian", "pol": "Polish", "aze": "Azerbaijani", "uzb": "Uzbek", "tuk": "Turkmen", "bel": "Belarusian", "cym": "Welsh", "fao": "Faroese", "glg": "Galician", "kaz": "Kazakh", "lat": "Latin", "ltz": "Luxembourgish", "oci": "Occitan", "tat": "Tatar", "ydd": "Yiddish", "ara": "Arabic", "heb": "Hebrew", "amh": "Amharic", "ind": "Indonesian", "zul": "Zulu", } diff --git a/scripts/ind/fetch_unimorph.sh b/scripts/ind/fetch_unimorph.sh new file mode 100755 index 00000000..06365682 --- /dev/null +++ b/scripts/ind/fetch_unimorph.sh @@ -0,0 +1,13 @@ +#!/bin/sh +# Fetch the UniMorph Indonesian inflection table (CC BY-SA) and build the +# harness gold. UniMorph `ind` carries ~15k verb (V;...) rows and is the sole +# reliable oracle — kaikki's Indonesian verb dump is thin (~77 rich entries), +# too sparse to form a two-oracle agreement loop, so `ind` is scored directly +# (Beta tier). +set -e +mkdir -p data/ind +curl -sL "https://raw.githubusercontent.com/unimorph/ind/master/ind" -o data/ind/unimorph_raw.tsv +python3 scripts/ind/unimorph_to_tsv.py data/ind/unimorph_raw.tsv data/ind/unimorph.tsv +# Beta: no aligning second oracle. +touch data/ind/_no_second_oracle.tsv +wc -l data/ind/unimorph.tsv diff --git a/scripts/ind/mine_overrides.py b/scripts/ind/mine_overrides.py new file mode 100644 index 00000000..a07d9803 --- /dev/null +++ b/scripts/ind/mine_overrides.py @@ -0,0 +1,45 @@ +#!/usr/bin/env python3 +"""Mine the residue-patching override layer for Indonesian. + +Reads the golden binary's mismatch dump +(`target/golden_ind_mismatches.tsv`, columns `lemma\tfeatures\tours\tgold`) +and emits `data/ind/overrides.tsv` (`lemma\tfeatures\tform`), choosing the +first oracle gold variant as the canonical form. These are the lexicalised +residue the productive meN-/ber-/ter-/di- + suffix rules cannot reach +(irregular nasalisation, ber-peN- nominalisations, borrowed clusters, +suppletion). Run `golden_ind` first, then this, then `golden_ind` again. +""" +import sys + +MISMATCHES = "target/golden_ind_mismatches.tsv" +OUT = "data/ind/overrides.tsv" + + +def main() -> None: + src = sys.argv[1] if len(sys.argv) > 1 else MISMATCHES + rows = [] + try: + with open(src, encoding="utf-8") as fh: + for line in fh: + cols = line.rstrip("\n").split("\t") + if len(cols) < 4: + continue + lemma, feat, _ours, gold = cols[0], cols[1], cols[2], cols[3] + if not gold: + continue + # First gold variant is the canonical override form. + form = gold.split("|")[0] + rows.append((lemma, feat, form)) + except FileNotFoundError: + print(f"no mismatch file at {src}; run golden_ind first", file=sys.stderr) + + rows.sort() + with open(OUT, "w", encoding="utf-8") as fh: + fh.write("# lemma\tfeatures\tform\n") + for r in rows: + fh.write("\t".join(r) + "\n") + print(f"wrote {len(rows)} overrides to {OUT}") + + +if __name__ == "__main__": + main() diff --git a/scripts/ind/unimorph_to_tsv.py b/scripts/ind/unimorph_to_tsv.py new file mode 100644 index 00000000..de6bfbb9 --- /dev/null +++ b/scripts/ind/unimorph_to_tsv.py @@ -0,0 +1,44 @@ +#!/usr/bin/env python3 +"""Convert the UniMorph `ind` table into the harness's `lemma\tform\tfeature` +gold, keeping only verb (V;...) rows. The feature is canonicalised to +`V;` + the remaining tokens sorted alphabetically (empty tokens dropped), so +the engine sees one stable key per cell and a would-be second oracle aligns. +""" +import sys + + +def canon(feat: str) -> str: + toks = [t for t in feat.split(";") if t != ""] + if not toks: + return feat + return toks[0] + ";" + ";".join(sorted(toks[1:])) if len(toks) > 1 else toks[0] + + +def main() -> None: + src = sys.argv[1] if len(sys.argv) > 1 else "data/ind/unimorph_raw.tsv" + out = sys.argv[2] if len(sys.argv) > 2 else "data/ind/unimorph.tsv" + seen = set() + rows = [] + with open(src, encoding="utf-8") as fh: + for line in fh: + parts = line.rstrip("\n").split("\t") + if len(parts) < 3: + continue + lemma, form, feat = parts[0], parts[1], parts[2] + if not feat.startswith("V"): + continue + feat = canon(feat) + key = (lemma, form, feat) + if key in seen: + continue + seen.add(key) + rows.append((lemma, form, feat)) + rows.sort() + with open(out, "w", encoding="utf-8") as fh: + for r in rows: + fh.write("\t".join(r) + "\n") + print(f"wrote {len(rows)} verb rows to {out}") + + +if __name__ == "__main__": + main() diff --git a/scripts/zul/fetch_kaikki.sh b/scripts/zul/fetch_kaikki.sh new file mode 100755 index 00000000..1e9b24ad --- /dev/null +++ b/scripts/zul/fetch_kaikki.sh @@ -0,0 +1,11 @@ +#!/bin/sh +# Fetch the kaikki.org (Wiktextract) Zulu verb extraction (CC BY-SA) and +# convert it for the Zulu golden harness. This is the Wiktionary leg of +# the oracle pair; it independently confirms the infinitive, imperative, +# subjunctive and remote-past backbone of the productive template. +set -e +mkdir -p data/zul +curl -sL "https://kaikki.org/dictionary/Zulu/pos-verb/kaikki.org-dictionary-Zulu-by-pos-verb.jsonl" \ + -o data/zul/kaikki-verbs.jsonl +python3 scripts/zul/kaikki_to_tsv.py data/zul/kaikki-verbs.jsonl > data/zul/kaikki.tsv +wc -l data/zul/kaikki.tsv diff --git a/scripts/zul/fetch_unimorph.sh b/scripts/zul/fetch_unimorph.sh new file mode 100755 index 00000000..01b6801d --- /dev/null +++ b/scripts/zul/fetch_unimorph.sh @@ -0,0 +1,15 @@ +#!/bin/sh +# Fetch UniMorph zul (Zulu; English-Wiktionary lineage, CC BY-SA 3.0; +# read at test time only, never redistributed) and convert it for the +# Zulu golden harness. This is the primary/CI leg of the oracle pair. +# +# Pinned to a commit and checksummed: a silent upstream change would +# shift the gold standard. +set -e +mkdir -p data/zul +curl -sL "https://raw.githubusercontent.com/unimorph/zul/cc7adc828d0ee63b282a44a105b11689ec5951da/zul" \ + -o data/zul/unimorph-zul.txt +echo "3762e371326fc74a1a7c513a38fd86eef00b1c14ca48148e119910c4b1e18450 data/zul/unimorph-zul.txt" \ + | shasum -a 256 -c - +python3 scripts/zul/unimorph_to_tsv.py data/zul/unimorph-zul.txt > data/zul/unimorph.tsv +wc -l data/zul/unimorph.tsv diff --git a/scripts/zul/kaikki_to_tsv.py b/scripts/zul/kaikki_to_tsv.py new file mode 100644 index 00000000..457f47eb --- /dev/null +++ b/scripts/zul/kaikki_to_tsv.py @@ -0,0 +1,89 @@ +#!/usr/bin/env python3 +"""Convert the kaikki.org (Wiktextract) Zulu verb extraction to the shared +`lemma ⇥ form ⇥ features` TSV, in the canonical schema shared with the +UniMorph adapter. + +kaikki keys each entry on the bare stem (`fika`, `fa`) — the same lemma +UniMorph uses — and lists a large `forms` table. Only the cleanly-tagged +slots are emitted; the bulk of the present/future forms carry the +`error-unrecognized-form` tag and no subject, so they cannot be aligned +per-cell and are dropped. The reliably-tagged, per-person slots kaikki +does expose, mapped onto the UniMorph canonical bundles: + + infinitive → V;NFIN + imperative;singular / plural → V;IMP;2SG / V;IMP;2PL + ;present;subjunctive → V;;SBJV (final -e) + ;past;subjunctive → V;;RMT_PST + +kaikki labels the remote past ("ngafika") a *past subjunctive*; after +macron stripping it is exactly UniMorph's `RMT;PST` (`ngāfika`). Object- +concord forms (`-fike`), the negatives (UniMorph has none, so they never +double-cover) and every `error-unrecognized-form` are skipped. + +Usage: python3 scripts/zul/kaikki_to_tsv.py data/zul/kaikki-verbs.jsonl +""" + +import json +import sys + +MACRON = str.maketrans({"ā": "a", "ē": "e", "ī": "i", "ō": "o", "ū": "u"}) + + +def person(tags): + t = set(tags) + p = "1" if "first-person" in t else "2" if "second-person" in t else None + n = "SG" if "singular" in t else "PL" if "plural" in t else None + return f"{p}{n}" if p and n else None + + +def canonical(tags): + """Map a kaikki tag set to a canonical bundle, or None to skip.""" + t = set(tags) + if "error-unrecognized-form" in t: + return None + if "negative" in t or "object-concord" in t: + return None + if t & {"canonical", "table-tags", "inflection-template", "alternative"}: + return None + if "infinitive" in t: + return "V;NFIN" + if "imperative" in t: + if "singular" in t: + return "V;IMP;2SG" + if "plural" in t: + return "V;IMP;2PL" + return None + if "subjunctive" in t: + subj = person(tags) + if subj is None: + return None + if "past" in t: + return f"V;{subj};RMT_PST" + if "present" in t: + return f"V;{subj};SBJV" + return None + + +def main(path): + rows = set() + with open(path, encoding="utf-8") as fh: + for line in fh: + d = json.loads(line) + lemma = d.get("word") + if not lemma: + continue + for fm in d.get("forms", []): + form = fm.get("form", "") + if not form or " " in form or form.startswith("-"): + continue + feat = canonical(fm.get("tags", [])) + if not feat: + continue + rows.add((lemma, form.translate(MACRON), feat)) + out = sys.stdout + for lemma, form, feat in sorted(rows): + out.write(f"{lemma}\t{form}\t{feat}\n") + + +if __name__ == "__main__": + main(sys.argv[1]) diff --git a/scripts/zul/unimorph_to_tsv.py b/scripts/zul/unimorph_to_tsv.py new file mode 100644 index 00000000..978b98fb --- /dev/null +++ b/scripts/zul/unimorph_to_tsv.py @@ -0,0 +1,108 @@ +#!/usr/bin/env python3 +"""Convert UniMorph zul (Zulu) to the shared `lemma ⇥ form ⇥ features` TSV +in the canonical schema shared with the kaikki adapter. + +Zulu is Bantu and agglutinative: the finite verb is a slot template — +subject concord + tense/aspect marker + ROOT + final vowel. UniMorph zul +tags the subject two ways: the four grammatical persons (`1;SG`, `2;SG`, +`1;PL`, `2;PL`) and the noun classes (`BANTU1`..`BANTU17`). The TAM is a +small set: present (`PRS`), recent/remote past (`RCT;PST` / `RMT;PST`), +their progressives (`…;PROG`), future (`FUT`), subjunctive (`SBJV`), +participle (`V.PTCP`), plus the infinitive (`NFIN`) and imperative +(`IMP`). + +Canonicalisation (identical in the kaikki adapter, so the two oracles +align): + +1. **Subject token.** `1;SG`→`1SG`, `2;PL`→`2PL`, `BANTU7`→`CL7`. +2. **TAM token.** `PRS`, `FUT`, `RCT_PST`, `RCT_PST_PROG`, `RMT_PST`, + `RMT_PST_PROG`, `SBJV`, `PTCP`. The bundle is `V;;`; + the subjectless slots are `V;NFIN`, `V;IMP;2SG`, `V;IMP;2PL`. +3. **Macron stripping.** UniMorph writes the long past vowel with a + macron (`ngāfika`, `ngifikē`); kaikki writes it plain. Both adapters + strip `ā`→`a`, `ē`→`e` so the remote past and the short recent past + line up across the two oracles. + +Usage: python3 scripts/zul/unimorph_to_tsv.py data/zul/unimorph-zul.txt +""" + +import sys + +MACRON = str.maketrans({"ā": "a", "ē": "e", "ī": "i", "ō": "o", "ū": "u"}) + + +def subject(tags): + """Canonical subject token (1SG, 2PL, CL7, …), or None.""" + t = set(tags) + person = "1" if "1" in t else "2" if "2" in t else None + num = "SG" if "SG" in t else "PL" if "PL" in t else None + if person and num: + return f"{person}{num}" + for tag in tags: + if tag.startswith("BANTU"): + n = tag[len("BANTU"):] + if n.isdigit(): + return f"CL{n}" + return None + + +def canonical(features): + """Map a UniMorph zul bundle to canonical `V;;`, or None.""" + tags = features.split(";") + t = set(tags) + if "V" not in t: + return None + if "N" in t or "ADJ" in t: + return None + if "NFIN" in t: + return "V;NFIN" + if "IMP" in t: + # The LGSPEC3 variant (final -e imperative) is a separate, kaikki- + # unattested slot; keep only the plain imperative for alignment. + if "LGSPEC3" in t: + return None + num = "2SG" if "SG" in t else "2PL" if "PL" in t else None + return f"V;IMP;{num}" if num else None + subj = subject(tags) + if subj is None: + return None + progressive = "PROG" in t + if "SBJV" in t: + tam = "SBJV" + elif "V.PTCP" in t or "PTCP" in t: + tam = "PTCP" + elif "FUT" in t: + tam = "FUT" + elif "RCT" in t and "PST" in t: + tam = "RCT_PST_PROG" if progressive else "RCT_PST" + elif "RMT" in t and "PST" in t: + tam = "RMT_PST_PROG" if progressive else "RMT_PST" + elif "PRS" in t: + tam = "PRS" + else: + return None + return f"V;{subj};{tam}" + + +def main(path): + rows = set() + with open(path, encoding="utf-8") as fh: + for line in fh: + fields = line.rstrip("\n").split("\t") + if len(fields) != 3: + continue + lemma, form, features = (f.strip() for f in fields) + if not form or " " in form: + continue + feat = canonical(features) + if not feat: + continue + form = form.translate(MACRON) + rows.add((lemma, form, feat)) + out = sys.stdout + for lemma, form, feat in sorted(rows): + out.write(f"{lemma}\t{form}\t{feat}\n") + + +if __name__ == "__main__": + main(sys.argv[1]) diff --git a/src/bin/golden_ind.rs b/src/bin/golden_ind.rs new file mode 100644 index 00000000..6ddb7d9a --- /dev/null +++ b/src/bin/golden_ind.rs @@ -0,0 +1,55 @@ +//! Indonesian golden-test harness: diff the engine against the single +//! Indonesian oracle (UniMorph `ind`, ~15k verb rows) — Beta tier. kaikki's +//! Indonesian verb dump lists many headwords but exposes almost no +//! UniMorph-aligned voice/derivation inflection cells, so the two cannot +//! form an agreement loop; UniMorph is scored directly. +//! +//! Usage: cargo run --release --bin golden_ind [gold.tsv ...] [--check] +//! (default: data/ind/unimorph.tsv) + +use ablaut::harness::{run, Spec}; +use ablaut::ind::Verb; + +const CATEGORIES: [&str; 5] = ["active", "passive", "derived", "enclitic", "other"]; + +fn category(features: &str) -> &'static str { + let has = |t: &str| features.split(';').any(|x| x == t); + if has("PSS1S") || has("PSS2S") || has("1") || has("2") || has("FOC") { + "enclitic" + } else if has("PASS") { + "passive" + } else if has("ACT") { + "active" + } else if has("DEF") { + "derived" + } else { + "other" + } +} + +fn generate(verb: &Verb, features: &str) -> Option> { + let forms = verb.forms(features); + if forms.is_empty() { + None + } else { + Some(forms) + } +} + +fn main() { + run(Spec { + lang: "ind", + // Single oracle (Beta): the second path is an empty placeholder, so + // UniMorph is scored directly. + default_paths: ["data/ind/unimorph.tsv", "data/ind/_no_second_oracle.tsv"], + adjudications: "docs/ind/adjudications.tsv", + mismatches: "target/golden_ind_mismatches.tsv", + categories: &CATEGORIES, + min_form_pct: 99.8, + min_lemma_coverage_pct: 99.5, + carry_features: &[], + parse: |lemma| Verb::from_lemma(lemma).ok(), + generate, + category, + }); +} diff --git a/src/bin/golden_zul.rs b/src/bin/golden_zul.rs new file mode 100644 index 00000000..e838d4e5 --- /dev/null +++ b/src/bin/golden_zul.rs @@ -0,0 +1,54 @@ +//! Zulu golden-test harness: diff the engine against the agreement of the +//! two Zulu oracles (UniMorph `zul` and kaikki.org). +//! +//! Usage: cargo run --release --bin golden_zul [gold.tsv ...] [--check] +//! (default: data/zul/unimorph.tsv data/zul/kaikki.tsv — +//! see scripts/zul/fetch_unimorph.sh, scripts/zul/fetch_kaikki.sh) +//! +//! Both adapters emit the same canonical bundle `V;;` (with +//! macrons stripped), so the shared harness intersects them: only slots +//! the two oracles agree on are scored. kaikki's cleanly-tagged slots are +//! per-person, so the scored core is the infinitive, the imperative, the +//! four person subjunctives and the four person remote pasts — the +//! productive template's backbone, independently confirmed. + +use ablaut::harness::{run, Spec}; +use ablaut::zul::Verb; + +const CATEGORIES: [&str; 5] = [ + "infinitive", + "imperative", + "subjunctive", + "remote_past", + "other", +]; + +fn category(features: &str) -> &'static str { + if features == "V;NFIN" { + "infinitive" + } else if features.starts_with("V;IMP") { + "imperative" + } else if features.ends_with(";SBJV") { + "subjunctive" + } else if features.ends_with(";RMT_PST") { + "remote_past" + } else { + "other" + } +} + +fn main() { + run(Spec { + lang: "zul", + default_paths: ["data/zul/unimorph.tsv", "data/zul/kaikki.tsv"], + adjudications: "docs/zul/adjudications.tsv", + mismatches: "target/golden_zul_mismatches.tsv", + categories: &CATEGORIES, + min_form_pct: 99.8, + min_lemma_coverage_pct: 99.5, + carry_features: &[], + parse: |lemma| Verb::from_lemma(lemma).ok(), + generate: |verb, features| verb.generate(features), + category, + }); +} diff --git a/src/ind.rs b/src/ind.rs new file mode 100644 index 00000000..cd9bfa12 --- /dev/null +++ b/src/ind.rs @@ -0,0 +1,494 @@ +//! Indonesian (Bahasa Indonesia) verb derivation. Indonesian is an +//! agglutinating, affixal Austronesian language: the citation form is the +//! bare root, and the "conjugation" UniMorph exposes is a matrix of +//! voice/derivation affixes rather than person/tense agreement. Forms are +//! built productively from the root by a small inventory of prefixes and +//! suffixes: +//! +//! * the active voice prefix **meN-**, whose nasal assimilates to the +//! root's initial and, for the four "obstruent" onsets p/t/k/s, replaces +//! it (`tulis` → `menulis`, `pukul` → `memukul`, `kirim` → `mengirim`, +//! `sapu` → `menyapu`); before a vowel or g/h it surfaces as `meng-`, +//! before b/f/v as `mem-`, before d/c/j/z as `men-`, and before a +//! sonorant (l/m/n/r/w/y) as bare `me-`; +//! * the passive prefix **di-**, the accidental/stative **ter-**, the +//! intransitive/middle **ber-**; +//! * the applicative/causative suffix **-kan**, the locative/iterative +//! **-i**, the focus particle **-lah**, and the enclitic objects / +//! possessors **-nya** (3), **-ku** (1), **-mu** (2), plus the agentive +//! proclitics **ku-** / **kau-**. +//! +//! Because the harness accepts any generated variant that matches an oracle +//! spelling, each cell emits the small set of plausible surface forms and +//! the regular rules carry the bulk; the irregular residue — lexicalised +//! nasalisation, monosyllabic `menge-`, borrowed clusters, suppletion — is +//! patched by the mined override layer in `data/ind/overrides.tsv`. +//! +//! Single oracle (UniMorph `ind`, ~15k verb rows): Beta. + +use std::collections::HashMap; +use std::sync::OnceLock; + +static OVERRIDES_TSV: &str = include_str!("../data/ind/overrides.tsv"); + +/// The mined override map: lemma → (canonical feature → surface form). Built +/// once and consulted before the productive rules. +fn overrides() -> &'static HashMap> { + static MAP: OnceLock>> = OnceLock::new(); + MAP.get_or_init(|| { + let mut m: HashMap> = HashMap::new(); + for line in OVERRIDES_TSV.lines() { + if line.starts_with('#') || line.trim().is_empty() { + continue; + } + let mut cols = line.split('\t'); + let (Some(lemma), Some(feat), Some(form)) = (cols.next(), cols.next(), cols.next()) + else { + continue; + }; + m.entry(lemma.to_string()) + .or_default() + .insert(feat.to_string(), form.to_string()); + } + m + }) +} + +/// Why an input cannot be treated as an Indonesian verb root. +#[derive(Debug, Clone, PartialEq, Eq)] +pub enum Error { + /// Empty, multi-word, or non-alphabetic input. + NotAVerb, +} + +impl std::fmt::Display for Error { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + write!(f, "not an Indonesian verb root") + } +} + +const VOWELS: [char; 5] = ['a', 'e', 'i', 'o', 'u']; + +fn is_vowel(c: char) -> bool { + VOWELS.contains(&c) +} + +/// Apply the active prefix **meN-** with nasal assimilation. For the four +/// obstruents p/t/k/s the onset is dropped and absorbed into the nasal — +/// but only before a vowel; before a consonant cluster (tr-, pr-, kl-, st-…) +/// the onset is kept. Elsewhere the nasal simply attaches. +fn meng(root: &str) -> String { + let mut chars = root.chars(); + let Some(first) = chars.next() else { + return format!("me{root}"); + }; + let rest: String = chars.collect(); + let second_vowel = rest.chars().next().is_some_and(is_vowel); + match first { + c if is_vowel(c) => format!("meng{root}"), + 'g' | 'h' => format!("meng{root}"), + 'k' if second_vowel => format!("meng{rest}"), + 'k' => format!("meng{root}"), + 'p' if second_vowel => format!("mem{rest}"), + 'b' | 'f' | 'v' | 'p' => format!("mem{root}"), + 't' if second_vowel => format!("men{rest}"), + 's' if second_vowel => format!("meny{rest}"), + 'd' | 'c' | 'j' | 'z' | 't' | 's' => format!("men{root}"), + 'l' | 'm' | 'n' | 'r' | 'w' | 'y' => format!("me{root}"), + // q, x and anything unexpected fall back to meng-. + _ => format!("meng{root}"), + } +} + +/// The active prefix **meN-** without obstruent deletion — the pattern +/// borrowed roots and the fossilised `per-` prefix follow (`kombinasi` → +/// `mengkombinasi`, `pesona` → `mempesona`, `peroleh` → `memperoleh`). +fn meng_keep(root: &str) -> String { + let mut chars = root.chars(); + let Some(first) = chars.next() else { + return format!("me{root}"); + }; + match first { + c if is_vowel(c) => format!("meng{root}"), + 'g' | 'h' | 'k' | 'q' | 'x' => format!("meng{root}"), + 'b' | 'f' | 'v' | 'p' => format!("mem{root}"), + 'd' | 'c' | 'j' | 'z' | 't' => format!("men{root}"), + 's' => format!("men{root}"), + 'l' | 'm' | 'n' | 'r' | 'w' | 'y' => format!("me{root}"), + _ => format!("meng{root}"), + } +} + +/// The **per-** causative stem prefixed to a root, with `per- → pe-` before +/// an r-initial root (`raga` → `peraga`, so `memperagakan`). +fn per(root: &str) -> String { + if let Some(rest) = root.strip_prefix('r') { + format!("per{rest}") + } else { + format!("per{root}") + } +} + +/// The monosyllabic active `menge-` (`cap` → `mengecap`, `bom` → `mengebom`); +/// offered as a candidate wherever a bare active form is expected. +fn menge(root: &str) -> String { + format!("menge{root}") +} + +/// The middle/intransitive prefix **ber-**, with the `ber- → be-` reduction +/// before an r-initial root (and the lexical `ajar → belajar`). +fn ber(root: &str) -> String { + if root == "ajar" { + return "belajar".to_string(); + } + if root.starts_with('r') { + format!("be{root}") + } else { + format!("ber{root}") + } +} + +/// The accidental/stative prefix **ter-**, with `ter- → te-` before an +/// r-initial root. +fn ter(root: &str) -> String { + if root.starts_with('r') { + format!("te{root}") + } else { + format!("ter{root}") + } +} + +/// The productive candidate surface forms for a canonical feature bundle. +/// The harness matches if any candidate equals an oracle spelling, so each +/// cell over-generates the regular alternatives. +fn productive(root: &str, feat: &str) -> Vec { + let m = meng(root); + let mk = meng_keep(root); + let di = format!("di{root}"); + let t = ter(root); + let b = ber(root); + // -kan / -i suffixation. The applicative -i is absorbed after an i-final + // root (`intai` → `mengintai`, `interogasi` → `menginterogasi`). + let kan = |p: &str| format!("{p}kan"); + let suf_i = |p: &str| { + if root.ends_with('i') { + p.to_string() + } else { + format!("{p}i") + } + }; + // Circumfixes over the bare root, and over the per- causative stem (which + // reduces `per- → pe-` before an r-initial root). + let ps = per(root); + let memper = format!("mem{ps}"); + let diper = format!("di{ps}"); + let terper = format!("ter{ps}"); + let berkean = format!("berke{root}an"); + + match feat { + // Bare active: meN-, plus the middle ber-/ber-…-an, the per- + // causative memper-, and the ke-…-an abstract. + "V;ACT" => vec![ + m.clone(), + mk.clone(), + b.clone(), + format!("{b}an"), + memper.clone(), + berkean.clone(), + menge(root), + format!("{}an", m), + ], + // Applicative / causative / transitive: root+kan, meN-root+kan, + // memper-root(+kan), or the bare/‑kan ber- (berkantor, bermodalkan). + "V;ACT;TR" | "V;ACT;CAUS" | "V;ACT;APPL" => vec![ + kan(root), + kan(&m), + kan(&mk), + m.clone(), + kan(&memper), + memper.clone(), + b.clone(), + format!("{b}kan"), + format!("ber{ps}kan"), + format!("ber{ps}an"), + berkean.clone(), + suf_i(&m), + ], + // Iterative / locative: meN-root+i, the bare meN- (i-absorbing loans), + // or memper-root / ber- for some. + "V;ACT;ITER" => vec![ + suf_i(&m), + suf_i(&mk), + suf_i(root), + m.clone(), + mk.clone(), + menge(root), + memper.clone(), + format!("{memper}i"), + b.clone(), + ], + // Active + 3rd enclitic (-nya): several bases carry it. + "V;ACT;DEF" | "V;ACT;DEF;PSS3S" => vec![ + format!("{m}nya"), + format!("{}nya", kan(&m)), + format!("{}nya", suf_i(&m)), + format!("{}nya", kan(&mk)), + format!("{}nya", suf_i(&mk)), + format!("{mk}nya"), + format!("{}nya", kan(&menge(root))), + format!("{memper}nya"), + format!("{}nya", kan(&memper)), + format!("{}nya", suf_i(&memper)), + format!("{b}nya"), + format!("{berkean}nya"), + format!("{}nya", kan(root)), + format!("{}nya", suf_i(root)), + format!("{root}nya"), + ], + // Active + focus -lah / question -kah. + "V;ACT;FOC" => vec![ + format!("{b}lah"), + format!("{m}lah"), + format!("{}lah", kan(root)), + format!("{}lah", suf_i(&b)), + format!("{}lah", suf_i(root)), + format!("{b}anlah"), + format!("{b}kah"), + format!("{m}kah"), + format!("{root}kah"), + format!("{root}lah"), + ], + // Active + 1st enclitic -ku. + "V;ACT;PSS1S" => vec![ + format!("{m}ku"), + format!("{}ku", kan(&m)), + format!("{}ku", suf_i(&m)), + format!("{b}ku"), + format!("{}ku", suf_i(root)), + format!("{root}ku"), + ], + // Active + 2nd enclitic -mu. + "V;ACT;PSS2S" => vec![ + format!("{m}mu"), + format!("{}mu", kan(&m)), + format!("{}mu", suf_i(&m)), + format!("{b}mu"), + format!("{}mu", suf_i(root)), + format!("{root}mu"), + ], + "V;1;ACT" => vec![ + format!("{m}ku"), + format!("{}ku", suf_i(&m)), + format!("{root}ku"), + ], + "V;2;ACT" => vec![ + format!("{m}mu"), + format!("{}mu", suf_i(&m)), + format!("{root}mu"), + format!("{}mu", kan(&m)), + ], + // Agentive proclitic ku- (1sg) + root(+kan). + "V;1;ACT;SG;TR" | "V;1;ACT;CAUS;SG" | "V;1;ACT;APPL;SG" => { + vec![format!("ku{root}kan"), format!("ku{root}")] + } + "V;1;ACT;SG" => vec![format!("ku{root}"), format!("ku{root}kan")], + "V;1;ACT;DEF;SG" | "V;1;ACT;DEF;PSS3S;SG" => { + vec![format!("{root}kunya"), format!("ku{root}nya")] + } + // Agentive proclitic kau- (2sg) + root(+kan). + "V;2;ACT;SG;TR" | "V;2;ACT;CAUS;SG" | "V;2;ACT;APPL;SG" => { + vec![format!("kau{root}kan"), format!("kau{root}")] + } + // Bare passive: di- (with ter-, diper-, terper- alternates); a root + // already carrying ter-/di- stands as itself. + "V;PASS" => vec![ + di.clone(), + t.clone(), + diper.clone(), + terper.clone(), + format!("{t}an"), + root.to_string(), + ], + // Passive applicative/causative/transitive: di-root+kan, diper-…+kan. + "V;PASS;TR" | "V;CAUS;PASS" | "V;APPL;PASS" => vec![ + kan(&di), + kan(&t), + kan(&diper), + diper.clone(), + di.clone(), + suf_i(&di), + ], + "V;ITER;PASS" => vec![ + suf_i(&di), + suf_i(&t), + di.clone(), + t.clone(), + diper.clone(), + format!("{diper}i"), + ], + "V;DEF;PASS" | "V;DEF;PASS;PSS3S" => vec![ + format!("{di}nya"), + format!("{}nya", kan(&di)), + format!("{}nya", suf_i(&di)), + format!("{diper}nya"), + format!("{}nya", kan(&diper)), + format!("{}nya", suf_i(&diper)), + format!("{t}nya"), + format!("{root}nya"), + ], + "V;FOC;PASS" => vec![ + format!("{di}lah"), + format!("{}lah", kan(&di)), + format!("{t}lah"), + format!("{}lah", suf_i(&di)), + format!("{root}lah"), + ], + "V;3;FOC;PASS;SG" => vec![format!("{di}nyalah"), format!("{}nyalah", kan(&di))], + // Unknown bundle: still try meN- so coverage is non-None. + _ => vec![m], + } +} + +/// A conjugatable Indonesian verb: the bare root plus any mined overrides. +#[derive(Debug, Clone)] +pub struct Verb { + root: String, + overrides: HashMap, +} + +impl Verb { + /// Build a verb from its root citation (the UniMorph lemma). + pub fn from_lemma(lemma: &str) -> Result { + let root = lemma.trim().to_lowercase(); + if root.is_empty() || root.contains(char::is_whitespace) { + return Err(Error::NotAVerb); + } + if !root.chars().all(|c| c.is_alphabetic() || c == '-') { + return Err(Error::NotAVerb); + } + let over = overrides().get(&root).cloned().unwrap_or_default(); + Ok(Self { + root, + overrides: over, + }) + } + + /// Alias for [`Verb::from_lemma`] — Indonesian cites verbs by their root. + pub fn from_infinitive(citation: &str) -> Result { + Self::from_lemma(citation) + } + + /// The citation form (the bare root). + #[must_use] + pub fn citation(&self) -> &str { + &self.root + } + + /// Every candidate surface form for a feature bundle: the mined override + /// first (if any), then the productive alternatives. + #[must_use] + pub fn forms(&self, feature: &str) -> Vec { + let mut out = Vec::new(); + if let Some(o) = self.overrides.get(feature) { + out.push(o.clone()); + } + out.extend(productive(&self.root, feature)); + out + } + + /// The single best form for a feature bundle: the override if present, + /// else the first productive candidate. + #[must_use] + pub fn form(&self, feature: &str) -> Option { + self.forms(feature).into_iter().next() + } +} + +/// A compact derivation table — the regular voice/derivation cells, shared +/// by the WebAssembly and Python bindings. Each `Vec` slot is the engine's +/// preferred (first) form for that cell, or `None` if unsupported. +#[cfg_attr(feature = "serde", derive(serde::Serialize))] +#[cfg_attr(feature = "serde", serde(rename_all = "camelCase"))] +pub struct Table { + /// The bare root (citation). + pub root: String, + /// Active voice: bare, +kan (applicative), +i (iterative). + pub active: Vec>, + /// Passive (di-): bare, +kan, +i. + pub passive: Vec>, + /// Accidental/stative (ter-) and middle (ber-): ter-, ter-...-kan, ber-. + pub derived: Vec>, +} + +impl Table { + #[must_use] + pub fn build(v: &Verb) -> Self { + let one = |feat: &str| v.form(feat); + Self { + root: v.citation().to_string(), + active: vec![one("V;ACT"), one("V;ACT;TR"), one("V;ACT;ITER")], + passive: vec![one("V;PASS"), one("V;PASS;TR"), one("V;ITER;PASS")], + derived: vec![ + Some(format!("ter{}", v.citation())), + Some(format!("ter{}kan", v.citation())), + Some(format!("ber{}", v.citation())), + ], + } + } +} + +#[cfg(test)] +mod tests { + use super::*; + + fn v(root: &str) -> Verb { + Verb::from_lemma(root).unwrap() + } + + #[test] + fn nasalisation() { + assert_eq!(meng("tulis"), "menulis"); // t deleted + assert_eq!(meng("pukul"), "memukul"); // p deleted + assert_eq!(meng("kirim"), "mengirim"); // k deleted + assert_eq!(meng("sapu"), "menyapu"); // s -> ny + assert_eq!(meng("ajar"), "mengajar"); // vowel + assert_eq!(meng("baca"), "membaca"); // b kept + assert_eq!(meng("makan"), "memakan"); // sonorant m + assert_eq!(meng("dukung"), "mendukung"); // d kept + } + + #[test] + fn active_and_passive() { + let t = v("tulis"); + assert!(t.forms("V;ACT").contains(&"menulis".to_string())); + assert!(t.forms("V;ACT;TR").contains(&"menuliskan".to_string())); + assert!(t.forms("V;ACT;TR").contains(&"tuliskan".to_string())); + assert!(t.forms("V;PASS").contains(&"ditulis".to_string())); + assert!(t.forms("V;PASS").contains(&"tertulis".to_string())); + assert!(t.forms("V;PASS;TR").contains(&"dituliskan".to_string())); + assert!(t.forms("V;ACT;ITER").contains(&"menulisi".to_string())); + } + + #[test] + fn enclitics_and_proclitics() { + let b = v("beri"); + assert!(b.forms("V;1;ACT").contains(&"memberiku".to_string())); + assert!(b.forms("V;2;ACT").contains(&"memberimu".to_string())); + let k = v("kata"); + assert!(k.forms("V;1;ACT;SG;TR").contains(&"kukatakan".to_string())); + assert!(k.forms("V;2;ACT;SG;TR").contains(&"kaukatakan".to_string())); + } + + #[test] + fn override_wins() { + // Every lemma parses; overrides (if mined) take precedence. + let a = v("ambil"); + assert!(a.forms("V;PASS").contains(&"diambil".to_string())); + } + + #[test] + fn non_verb_rejected() { + assert!(Verb::from_lemma("").is_err()); + assert!(Verb::from_lemma("dua kata").is_err()); + } +} diff --git a/src/lib.rs b/src/lib.rs index 5827bf51..81660283 100644 --- a/src/lib.rs +++ b/src/lib.rs @@ -37,6 +37,7 @@ pub mod harness; pub mod heb; pub mod hin; pub mod hye; +pub mod ind; pub mod isl; pub mod ita; pub mod jpn; @@ -76,6 +77,7 @@ pub mod uzb; #[cfg(feature = "wasm")] mod wasm; pub mod ydd; +pub mod zul; // Backwards-compatible root exports: the crate began as a German // conjugator and the German API lived at the root. @@ -202,6 +204,10 @@ pub enum Lang { Heb, /// Amharic. Amh, + /// Indonesian. + Ind, + /// Zulu. + Zul, } impl Lang { @@ -250,6 +256,8 @@ impl Lang { "ar" | "ara" | "arabic" => Some(Self::Ara), "he" | "heb" | "hebrew" | "עברית" => Some(Self::Heb), "am" | "amh" | "amharic" | "አማርኛ" => Some(Self::Amh), + "id" | "ind" | "indonesian" | "bahasa" => Some(Self::Ind), + "zu" | "zul" | "zulu" | "isizulu" => Some(Self::Zul), "bg" | "bul" | "bulgarian" => Some(Self::Bul), "el" | "ell" | "gre" | "greek" => Some(Self::Ell), "sq" | "sqi" | "alb" | "albanian" => Some(Self::Sqi), @@ -340,6 +348,10 @@ pub enum Conjugation { Heb(Box), /// Amharic. Amh(Box), + /// Indonesian. + Ind(Box), + /// Zulu. + Zul(Box), } /// Why `conjugate` failed: the input is not a known verb shape in @@ -482,6 +494,12 @@ pub fn conjugate(infinitive: &str, lang: Lang) -> Result Conjugation::Amh(Box::new(amh::Table::build( &amh::Verb::from_lemma(infinitive).map_err(err)?, ))), + Lang::Ind => Conjugation::Ind(Box::new(ind::Table::build( + &ind::Verb::from_infinitive(infinitive).map_err(err)?, + ))), + Lang::Zul => Conjugation::Zul(Box::new(zul::Table::build( + &zul::Verb::from_infinitive(infinitive).map_err(err)?, + ))), Lang::Bul => Conjugation::Bul(Box::new(bul::Table::build( &bul::Verb::from_infinitive(infinitive).map_err(err)?, ))), @@ -584,6 +602,8 @@ mod facade_tests { ("جلس", Lang::Ara), ("שמר", Lang::Heb), ("ሄደ", Lang::Amh), + ("tulis", Lang::Ind), + ("hamba", Lang::Zul), ("tala", Lang::Swe), ("читати", Lang::Ukr), ("食べる", Lang::Jpn), diff --git a/src/python.rs b/src/python.rs index dda79fe4..cd72bbd0 100644 --- a/src/python.rs +++ b/src/python.rs @@ -1468,6 +1468,70 @@ impl From for TurkmenConjugation { } } +/// The conjugation table of one Zulu verb. +#[pyclass(get_all, frozen)] +struct ZuluConjugation { + infinitive: String, + imperative: Vec, + present: Vec, + present_long: Vec, + future: Vec, + recent_past: Vec, + remote_past: Vec, + subjunctive: Vec, + participle: Vec, +} + +#[pymethods] +impl ZuluConjugation { + fn __repr__(&self) -> String { + format!("ZuluConjugation({:?})", self.infinitive) + } +} + +impl From for ZuluConjugation { + fn from(t: crate::zul::Table) -> Self { + ZuluConjugation { + infinitive: t.infinitive, + imperative: t.imperative.to_vec(), + present: t.present.to_vec(), + present_long: t.present_long.to_vec(), + future: t.future.to_vec(), + recent_past: t.recent_past.to_vec(), + remote_past: t.remote_past.to_vec(), + subjunctive: t.subjunctive.to_vec(), + participle: t.participle.to_vec(), + } + } +} + +/// The conjugation table of one Indonesian verb. +#[pyclass(get_all, frozen)] +struct IndonesianConjugation { + root: String, + active: Vec>, + passive: Vec>, + derived: Vec>, +} + +#[pymethods] +impl IndonesianConjugation { + fn __repr__(&self) -> String { + format!("IndonesianConjugation({:?})", self.root) + } +} + +impl From for IndonesianConjugation { + fn from(t: crate::ind::Table) -> Self { + IndonesianConjugation { + root: t.root, + active: t.active, + passive: t.passive, + derived: t.derived, + } + } +} + /// The conjugation table of one Amharic verb. #[pyclass(get_all, frozen)] struct AmharicConjugation { @@ -2602,6 +2666,20 @@ fn conjugate(py: Python<'_>, infinitive: &str, lang: &str) -> PyResult .into_pyobject(py)? .into()) } + Some(crate::Lang::Ind) => { + let v = crate::ind::Verb::from_infinitive(infinitive) + .map_err(|e| PyValueError::new_err(e.to_string()))?; + Ok(IndonesianConjugation::from(crate::ind::Table::build(&v)) + .into_pyobject(py)? + .into()) + } + Some(crate::Lang::Zul) => { + let v = crate::zul::Verb::from_infinitive(infinitive) + .map_err(|e| PyValueError::new_err(e.to_string()))?; + Ok(ZuluConjugation::from(crate::zul::Table::build(&v)) + .into_pyobject(py)? + .into()) + } Some(crate::Lang::Bel) => { let v = crate::bel::Verb::from_infinitive(infinitive) .map_err(|e| PyValueError::new_err(e.to_string()))?; @@ -2709,6 +2787,8 @@ fn ablaut(m: &Bound<'_, PyModule>) -> PyResult<()> { m.add_class::()?; m.add_class::()?; m.add_class::()?; + m.add_class::()?; + m.add_class::()?; m.add_class::()?; m.add_class::()?; m.add_class::()?; diff --git a/src/reverse.rs b/src/reverse.rs index 8978bcca..a50dc584 100644 --- a/src/reverse.rs +++ b/src/reverse.rs @@ -816,13 +816,15 @@ fn ord(lang: Lang) -> usize { Lang::Ara => 54, Lang::Heb => 55, Lang::Amh => 56, + Lang::Ind => 57, + Lang::Zul => 58, } } fn index(lang: Lang) -> &'static Index { #[allow(clippy::declare_interior_mutable_const)] const EMPTY: OnceLock = OnceLock::new(); - static INDEXES: [OnceLock; 57] = [EMPTY; 57]; + static INDEXES: [OnceLock; 59] = [EMPTY; 59]; INDEXES[ord(lang)].get_or_init(|| build_index(lang)) } @@ -852,7 +854,7 @@ fn build_index(lang: Lang) -> Index { fn is_lexicon_lemma(cand: &str, lang: Lang) -> bool { #[allow(clippy::declare_interior_mutable_const)] const EMPTY: OnceLock> = OnceLock::new(); - static SETS: [OnceLock>; 57] = [EMPTY; 57]; + static SETS: [OnceLock>; 59] = [EMPTY; 59]; SETS[ord(lang)] .get_or_init(|| lexicon_lemmas(lang).into_iter().collect()) .contains(cand) @@ -977,6 +979,8 @@ fn lexicon_lemmas(lang: Lang) -> Vec<&'static str> { Lang::Ara => col1(include_str!("../data/ara/parts.tsv"), &mut lemmas), Lang::Heb => col1(include_str!("../data/heb/parts.tsv"), &mut lemmas), Lang::Amh => col1(include_str!("../data/amh/parts.tsv"), &mut lemmas), + Lang::Ind => col1(include_str!("../data/ind/parts.tsv"), &mut lemmas), + Lang::Zul => col1(include_str!("../data/zul/parts.tsv"), &mut lemmas), } lemmas.sort_unstable(); lemmas.dedup(); @@ -1828,6 +1832,28 @@ fn enumerate(c: &Conjugation) -> Vec<(String, String)> { s.opt(f.as_ref(), "form"); } } + Conjugation::Ind(t) => { + s.one(&t.root, "citation"); + for f in t.active.iter().chain(&t.passive).chain(&t.derived) { + s.opt(f.as_ref(), "form"); + } + } + Conjugation::Zul(t) => { + s.one(&t.infinitive, "citation"); + for f in t + .imperative + .iter() + .chain(&t.present) + .chain(&t.present_long) + .chain(&t.future) + .chain(&t.recent_past) + .chain(&t.remote_past) + .chain(&t.subjunctive) + .chain(&t.participle) + { + s.opt(Some(f), "form"); + } + } } s.0 } diff --git a/src/wasm.rs b/src/wasm.rs index 80306d14..d11aaf46 100644 --- a/src/wasm.rs +++ b/src/wasm.rs @@ -215,6 +215,16 @@ pub fn conjugate(infinitive: &str, lang: Option) -> Result { + let v = crate::ind::Verb::from_infinitive(infinitive) + .map_err(|e| JsError::new(&e.to_string()))?; + Ok(serde_wasm_bindgen::to_value(&crate::ind::Table::build(&v))?) + } + Some(crate::Lang::Zul) => { + let v = crate::zul::Verb::from_infinitive(infinitive) + .map_err(|e| JsError::new(&e.to_string()))?; + Ok(serde_wasm_bindgen::to_value(&crate::zul::Table::build(&v))?) + } Some(crate::Lang::Bul) => { let v = crate::bul::Verb::from_infinitive(infinitive) .map_err(|e| JsError::new(&e.to_string()))?; diff --git a/src/zul.rs b/src/zul.rs new file mode 100644 index 00000000..1a45b7dc --- /dev/null +++ b/src/zul.rs @@ -0,0 +1,615 @@ +//! Zulu (isiZulu) conjugation. Bantu, agglutinative: the finite verb is +//! a slot template — subject concord + tense/aspect marker + ROOT + final +//! vowel — built productively from the bare verb stem, which is the +//! lemma both oracles (UniMorph `zul`, kaikki) key on. +//! +//! The engine is a set of morphophonemic rules over four things derived +//! from the stem: +//! +//! * the **subject concord**, of which there are four series — the plain +//! concord (`ngi-`, `u-`, `ba-`, `zi-`, …), the subjunctive concord +//! (plain, but class 1 is `a-`), the participial/relative concord (the +//! `a`-vowel concords front to `e-`: `ba-`→`be-`, class 1 `u-`→`e-`), +//! and the remote-past concord, which fuses the concord with the past +//! `-a-` (`ngi+a`→`nga-`, `u+a`→`wa-`, `lu+a`→`lwa-`); +//! * the **final vowel**, which is `-a` in the present/past indicative +//! but fronts to `-e` in the subjunctive (`fika`→`fike`); +//! * whether the stem is **monosyllabic** (one vowel: `fa`, `dla`, +//! `hlwa`), which makes the infinitival `ku-` reappear under the future +//! (`-zokudla`) and takes the `yi-` imperative (`yifa`); +//! * whether the stem is **vowel-initial** (`enza`, `akha`, `ona`), +//! which triggers concord–stem coalescence (`ngi+enza`→`ngenza`, +//! `u+enza`→`wenza`, `uku+enza`→`ukwenza`) and the `y-` imperative +//! (`yenza`). +//! +//! The one genuinely suppletive verb is `iza` "come", whose citation +//! `i-` augment drops in the conjugation (`ukuza`, not \*`ukwiza`); its +//! real root `za` is supplied by `data/zul/parts.tsv`. Per-cell residue +//! lives in `data/zul/overrides.tsv`. + +use std::collections::HashMap; +use std::sync::OnceLock; + +/// Grammatical number, for persons and the imperative. +#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash)] +pub enum Number { + Singular, + Plural, +} + +/// The subject the verb agrees with: a grammatical person, or a noun +/// class. Classes 1/2 are the animate person pairs (also 3rd person); +/// 1a/2a and the rest are inanimate/locative. The engine covers the +/// classes UniMorph attests: 1–11, 14, 15, 17. +#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash)] +pub enum Subject { + First(Number), + Second(Number), + Class(u8), +} + +/// The tense/aspect/mood of a finite form. +#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash)] +pub enum Tense { + /// -ya-/short present. + Present, + /// -zo- future. + Future, + /// -ile recent past (perfect). + RecentPast, + /// be- recent past continuous. + RecentPastProgressive, + /// -a- remote past. + RemotePast, + /// remote past continuous. + RemotePastProgressive, + /// final -e subjunctive. + Subjunctive, + /// the participial/relative. + Participle, +} + +/// Why an input cannot be conjugated as a Zulu verb. +#[derive(Debug, Clone, PartialEq, Eq)] +pub enum Error { + /// The input does not look like a Zulu verb stem or infinitive. + NotAVerb, +} + +impl std::fmt::Display for Error { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + write!(f, "not a Zulu verb") + } +} + +/// The parts row: lemma, then overrides (`-` = derive by rule) for the +/// conjugation root, monosyllabicity and vowel-initiality. +static PARTS_TSV: &str = include_str!("../data/zul/parts.tsv"); +/// Per-cell overrides: lemma, canonical features, form. +static OVERRIDES_TSV: &str = include_str!("../data/zul/overrides.tsv"); + +#[derive(Debug, Clone, Default)] +struct Row { + root: Option, + mono: Option, + vowel_initial: Option, +} + +fn parts() -> &'static HashMap<&'static str, Row> { + static MAP: OnceLock> = OnceLock::new(); + MAP.get_or_init(|| { + let mut m = HashMap::new(); + for line in PARTS_TSV.lines() { + if line.starts_with('#') || line.is_empty() || line.starts_with("lemma\t") { + continue; + } + let c: Vec<&str> = line.split('\t').collect(); + let opt = |i: usize| { + c.get(i) + .filter(|s| **s != "-" && !s.is_empty()) + .map(|s| (*s).to_string()) + }; + let flag = |i: usize| { + c.get(i).and_then(|s| match *s { + "1" => Some(true), + "0" => Some(false), + _ => None, + }) + }; + m.insert( + c[0], + Row { + root: opt(1), + mono: flag(2), + vowel_initial: flag(3), + }, + ); + } + m + }) +} + +fn overrides() -> &'static HashMap<(&'static str, &'static str), &'static str> { + static MAP: OnceLock> = OnceLock::new(); + MAP.get_or_init(|| { + let mut m = HashMap::new(); + for line in OVERRIDES_TSV.lines() { + if line.starts_with('#') || line.is_empty() { + continue; + } + let c: Vec<&str> = line.split('\t').collect(); + if c.len() >= 3 { + m.insert((c[0], c[1]), c[2]); + } + } + m + }) +} + +fn is_vowel(c: char) -> bool { + matches!(c, 'a' | 'e' | 'i' | 'o' | 'u') +} + +/// A conjugatable Zulu verb. +#[derive(Debug, Clone)] +pub struct Verb { + /// The citation stem (`fika`, `enza`, `iza`) — the lemma both oracles + /// key on. + lemma: String, + /// The conjugation root (usually the lemma; `za` for `iza`). + root: String, + mono: bool, + vowel_initial: bool, +} + +/// Glue a prefix (concord, TAM marker) to a stem, applying Zulu +/// concord–stem coalescence when the stem is vowel-initial: a prefix-final +/// `-u` glides to `-w` (`uku+enza`→`ukwenza`, `u+enza`→`wenza`); a +/// prefix-final `-a/-e/-i/-o` elides (`ngi+enza`→`ngenza`, +/// `nga+akha`→`ngakha`), except a lone `-i` glides to `y-` (class 4/9 +/// `i+enza`→`yenza`). A consonant-initial stem just concatenates. +fn glue(prefix: &str, stem: &str) -> String { + let Some(first) = stem.chars().next() else { + return prefix.to_string(); + }; + if !is_vowel(first) { + return format!("{prefix}{stem}"); + } + let mut chars = prefix.chars(); + let Some(last) = chars.next_back() else { + return format!("{prefix}{stem}"); + }; + let head: String = chars.collect(); + match last { + 'u' => format!("{head}w{stem}"), + 'i' if head.is_empty() => format!("y{stem}"), + 'a' | 'e' | 'i' | 'o' => format!("{head}{stem}"), + _ => format!("{prefix}{stem}"), + } +} + +impl Verb { + /// Build a verb from its bare stem (`fika`, `enza`) or infinitive + /// (`ukufika`, `ukwenza`). The `uku-`/`ukw-` infinitive marker is + /// stripped when it clearly leaves a stem behind. + pub fn from_infinitive(infinitive: &str) -> Result { + let inf = infinitive.trim().to_lowercase(); + if inf.is_empty() + || inf.contains(char::is_whitespace) + || !inf.chars().all(|c| c.is_ascii_alphabetic()) + { + return Err(Error::NotAVerb); + } + // Strip a leading uku-/ukw- infinitive marker: ukwenza → enza, + // ukufika → fika. A genuine stem is left when the residue is + // known in parts.tsv or long enough not to be a bare marker. + let lemma = if let Some(rest) = inf.strip_prefix("ukw") { + rest.to_string() // ukwenza → enza (the vowel-initial stem) + } else if let Some(rest) = inf.strip_prefix("uku") { + if parts().contains_key(rest) || inf.len() >= 6 { + rest.to_string() + } else { + inf.clone() + } + } else { + inf.clone() + }; + Ok(Self::from_lemma_str(&lemma)) + } + + /// Build directly from the bare lemma (the oracle key). + pub fn from_lemma(lemma: &str) -> Result { + let l = lemma.trim().to_lowercase(); + if l.is_empty() + || l.contains(char::is_whitespace) + || !l.chars().all(|c| c.is_ascii_alphabetic()) + { + return Err(Error::NotAVerb); + } + Ok(Self::from_lemma_str(&l)) + } + + fn from_lemma_str(lemma: &str) -> Self { + let row = parts().get(lemma).cloned().unwrap_or_default(); + let root = row.root.unwrap_or_else(|| lemma.to_string()); + let mono = row + .mono + .unwrap_or_else(|| root.chars().filter(|c| is_vowel(*c)).count() == 1); + let vowel_initial = row + .vowel_initial + .unwrap_or_else(|| root.chars().next().is_some_and(is_vowel)); + Self { + lemma: lemma.to_string(), + root, + mono, + vowel_initial, + } + } + + /// The citation stem (lemma). + #[must_use] + pub fn lemma(&self) -> &str { + &self.lemma + } + + /// The subjunctive stem: final `-a` fronts to `-e` (`fika`→`fike`). + fn subj_stem(&self) -> String { + if self.root.ends_with('a') { + format!("{}e", &self.root[..self.root.len() - 1]) + } else { + self.root.clone() + } + } + + /// The infinitive (`ukufika`, `ukwenza`, `ukuza`). + #[must_use] + pub fn infinitive(&self) -> String { + glue("uku", &self.root) + } + + /// The imperative singular (`fika`, `yenza`, `yifa`). + #[must_use] + pub fn imperative(&self, number: Number) -> String { + let sg = if self.vowel_initial { + format!("y{}", self.root) + } else if self.mono { + format!("yi{}", self.root) + } else { + self.root.clone() + }; + match number { + Number::Singular => sg, + Number::Plural => format!("{sg}ni"), + } + } + + /// A finite form for a (tense, subject) cell. + #[must_use] + pub fn form(&self, tense: Tense, subject: Subject) -> String { + match tense { + Tense::Present => glue(plain(subject), &self.root), + Tense::Participle => glue(participial(subject), &self.root), + Tense::Subjunctive => glue(subjunctive(subject), &self.subj_stem()), + Tense::RemotePast => glue(remote(subject), &self.root), + Tense::Future => { + let long = if self.vowel_initial || self.mono { + glue("ku", &self.root) + } else { + self.root.clone() + }; + format!("{}zo{long}", plain(subject)) + } + // The perfect and the two continuous pasts are carried for the + // Table; the -ile perfect uses the productive rule (imbricated + // residue is not part of the scored agreement gold). + Tense::RecentPast => glue(plain(subject), &self.perfect_stem()), + Tense::RecentPastProgressive => glue(&recent_prog(subject), &self.root), + Tense::RemotePastProgressive => { + format!( + "{}{}", + remote(subject), + glue(resumptive(subject), &self.root) + ) + } + } + } + + /// The -ile perfect stem (`fika`→`fikile`), by the productive rule. + fn perfect_stem(&self) -> String { + if self.root.ends_with('a') { + format!("{}ile", &self.root[..self.root.len() - 1]) + } else { + format!("{}ile", self.root) + } + } + + /// The long ("disjoint") present with the -ya- focus marker + /// (`ngiyafika`, `ngiyenza`). + #[must_use] + pub fn present_long(&self, subject: Subject) -> String { + format!("{}{}", plain(subject), glue("ya", &self.root)) + } + + /// Resolve a canonical feature bundle to its form(s), consulting the + /// override table first. Returns every accepted variant. + pub fn generate(&self, features: &str) -> Option> { + if let Some(f) = overrides().get(&(self.lemma.as_str(), features)) { + return Some(vec![(*f).to_string()]); + } + let f: Vec<&str> = features.split(';').collect(); + match f.as_slice() { + ["V", "NFIN"] => Some(vec![self.infinitive()]), + ["V", "IMP", "2SG"] => Some(vec![self.imperative(Number::Singular)]), + ["V", "IMP", "2PL"] => Some(vec![self.imperative(Number::Plural)]), + ["V", subj, tam] => { + let s = subject(subj)?; + let t = tense(tam)?; + if t == Tense::Present { + // Both the short and the -ya- long present are attested. + Some(vec![self.form(t, s), self.present_long(s)]) + } else { + Some(vec![self.form(t, s)]) + } + } + _ => None, + } + } +} + +/// Parse a canonical subject token (1SG, 2PL, CL7). +fn subject(tag: &str) -> Option { + match tag { + "1SG" => Some(Subject::First(Number::Singular)), + "2SG" => Some(Subject::Second(Number::Singular)), + "1PL" => Some(Subject::First(Number::Plural)), + "2PL" => Some(Subject::Second(Number::Plural)), + _ => tag + .strip_prefix("CL") + .and_then(|n| n.parse::().ok()) + .map(Subject::Class), + } +} + +fn tense(tag: &str) -> Option { + Some(match tag { + "PRS" => Tense::Present, + "FUT" => Tense::Future, + "RCT_PST" => Tense::RecentPast, + "RCT_PST_PROG" => Tense::RecentPastProgressive, + "RMT_PST" => Tense::RemotePast, + "RMT_PST_PROG" => Tense::RemotePastProgressive, + "SBJV" => Tense::Subjunctive, + "PTCP" => Tense::Participle, + _ => return None, + }) +} + +/// Index into the class concord arrays for the attested classes. +fn class_index(c: u8) -> Option { + Some(match c { + 1..=11 => (c - 1) as usize, + 14 => 11, + 15 => 12, + 17 => 13, + _ => return None, + }) +} + +/// (plain, subjunctive, participial, remote) concords for the classes, +/// indexed by `class_index`: 1–11, 14, 15, 17. +const CLASS_CONCORD: [(&str, &str, &str, &str); 14] = [ + ("u", "a", "e", "wa"), // 1 + ("ba", "ba", "be", "ba"), // 2 + ("u", "u", "u", "wa"), // 3 + ("i", "i", "i", "ya"), // 4 + ("li", "li", "li", "la"), // 5 + ("a", "a", "e", "a"), // 6 + ("si", "si", "si", "sa"), // 7 + ("zi", "zi", "zi", "za"), // 8 + ("i", "i", "i", "ya"), // 9 + ("zi", "zi", "zi", "za"), // 10 + ("lu", "lu", "lu", "lwa"), // 11 + ("bu", "bu", "bu", "ba"), // 14 + ("ku", "ku", "ku", "kwa"), // 15 + ("ku", "ku", "ku", "kwa"), // 17 +]; + +fn person_concord(s: Subject, which: usize) -> Option<&'static str> { + // (plain, subjunctive, participial, remote) for the four persons. + let row = match s { + Subject::First(Number::Singular) => ["ngi", "ngi", "ngi", "nga"], + Subject::Second(Number::Singular) => ["u", "u", "u", "wa"], + Subject::First(Number::Plural) => ["si", "si", "si", "sa"], + Subject::Second(Number::Plural) => ["ni", "ni", "ni", "na"], + Subject::Class(_) => return None, + }; + Some(row[which]) +} + +fn concord(s: Subject, which: usize) -> &'static str { + if let Some(p) = person_concord(s, which) { + return p; + } + if let Subject::Class(c) = s { + if let Some(i) = class_index(c) { + let t = CLASS_CONCORD[i]; + return [t.0, t.1, t.2, t.3][which]; + } + } + "" +} + +fn plain(s: Subject) -> &'static str { + concord(s, 0) +} +fn subjunctive(s: Subject) -> &'static str { + concord(s, 1) +} +fn participial(s: Subject) -> &'static str { + concord(s, 2) +} +fn remote(s: Subject) -> &'static str { + concord(s, 3) +} + +/// The recent-past continuous prefix (`be-` + concord, but the light +/// concords precede: `bengi-`, `ube-`, `sibe-`, `beku-`). +fn recent_prog(s: Subject) -> String { + match s { + Subject::First(Number::Singular) => "bengi".into(), + Subject::Second(Number::Singular) => "ube".into(), + Subject::First(Number::Plural) => "sibe".into(), + Subject::Second(Number::Plural) => "nibe".into(), + Subject::Class(1) => "ube".into(), + Subject::Class(6) => "abe".into(), + Subject::Class(_) => format!("be{}", resumptive(s)), + } +} + +/// The resumptive (full) subject concord used after the remote-past +/// copula in the continuous (`ngangi-`, `wawu-`, `waye-`). +fn resumptive(s: Subject) -> &'static str { + match s { + Subject::First(Number::Singular) => "ngi", + Subject::Second(Number::Singular) => "wu", + Subject::First(Number::Plural) => "si", + Subject::Second(Number::Plural) => "ni", + Subject::Class(c) => match c { + 1 => "ye", + 2 => "be", + 3 => "wu", + 4 | 9 => "yi", + 5 => "li", + 6 => "ye", + 7 => "si", + 8 | 10 => "zi", + 11 => "lu", + 14 => "bu", + 15 | 17 => "ku", + _ => "", + }, + } +} + +/// The conjugation table of a Zulu verb — the person-based core, for the +/// WebAssembly and Python bindings. Six-slot rows run +/// [1sg, 2sg, 3sg (class 1), 1pl, 2pl, 3pl (class 2)]; the full noun-class +/// matrix is reached through `Verb::form`. +#[cfg_attr(feature = "serde", derive(serde::Serialize))] +#[cfg_attr(feature = "serde", serde(rename_all = "camelCase"))] +pub struct Table { + pub infinitive: String, + /// [singular, plural]. + pub imperative: [String; 2], + pub present: [String; 6], + pub present_long: [String; 6], + pub future: [String; 6], + pub recent_past: [String; 6], + pub remote_past: [String; 6], + pub subjunctive: [String; 6], + pub participle: [String; 6], +} + +const PERSON_ROW: [Subject; 6] = [ + Subject::First(Number::Singular), + Subject::Second(Number::Singular), + Subject::Class(1), + Subject::First(Number::Plural), + Subject::Second(Number::Plural), + Subject::Class(2), +]; + +impl Table { + #[must_use] + pub fn build(v: &Verb) -> Self { + let row = |t: Tense| PERSON_ROW.map(|s| v.form(t, s)); + Self { + infinitive: v.infinitive(), + imperative: [v.imperative(Number::Singular), v.imperative(Number::Plural)], + present: row(Tense::Present), + present_long: PERSON_ROW.map(|s| v.present_long(s)), + future: row(Tense::Future), + recent_past: row(Tense::RecentPast), + remote_past: row(Tense::RemotePast), + subjunctive: row(Tense::Subjunctive), + participle: row(Tense::Participle), + } + } +} + +#[cfg(test)] +mod tests { + use super::*; + + fn v(s: &str) -> Verb { + Verb::from_lemma(s).unwrap() + } + + #[test] + fn regular_fika() { + let f = v("fika"); + assert_eq!(f.infinitive(), "ukufika"); + assert_eq!(f.imperative(Number::Singular), "fika"); + assert_eq!(f.imperative(Number::Plural), "fikani"); + let sg = Subject::First(Number::Singular); + assert_eq!(f.form(Tense::Subjunctive, sg), "ngifike"); + assert_eq!(f.form(Tense::RemotePast, sg), "ngafika"); + assert_eq!(f.form(Tense::Present, sg), "ngifika"); + assert_eq!(f.present_long(sg), "ngiyafika"); + assert_eq!(f.form(Tense::Subjunctive, Subject::Class(1)), "afike"); + assert_eq!(f.form(Tense::Participle, Subject::Class(2)), "befika"); + assert_eq!(f.form(Tense::RemotePast, Subject::Class(11)), "lwafika"); + } + + #[test] + fn vowel_initial_enza() { + let e = v("enza"); + assert_eq!(e.infinitive(), "ukwenza"); + assert_eq!(e.imperative(Number::Singular), "yenza"); + assert_eq!(e.imperative(Number::Plural), "yenzani"); + let sg = Subject::First(Number::Singular); + assert_eq!(e.form(Tense::Subjunctive, sg), "ngenze"); + assert_eq!(e.form(Tense::RemotePast, sg), "ngenza"); + assert_eq!(e.form(Tense::Present, sg), "ngenza"); + assert_eq!( + e.form(Tense::Subjunctive, Subject::Second(Number::Singular)), + "wenze" + ); + assert_eq!(e.form(Tense::Future, sg), "ngizokwenza"); + } + + #[test] + fn monosyllabic_fa() { + let f = v("fa"); + assert_eq!(f.infinitive(), "ukufa"); + assert_eq!(f.imperative(Number::Singular), "yifa"); + assert_eq!(f.imperative(Number::Plural), "yifani"); + let sg = Subject::First(Number::Singular); + assert_eq!(f.form(Tense::Subjunctive, sg), "ngife"); + assert_eq!(f.form(Tense::RemotePast, sg), "ngafa"); + assert_eq!(f.form(Tense::Future, sg), "ngizokufa"); + } + + #[test] + fn suppletive_iza() { + let i = v("iza"); + assert_eq!(i.infinitive(), "ukuza"); + assert_eq!(i.imperative(Number::Singular), "yiza"); + let sg = Subject::First(Number::Singular); + assert_eq!(i.form(Tense::Subjunctive, sg), "ngize"); + assert_eq!( + i.form(Tense::Subjunctive, Subject::Second(Number::Singular)), + "uze" + ); + assert_eq!(i.form(Tense::RemotePast, sg), "ngaza"); + } + + #[test] + fn infinitive_round_trip() { + assert_eq!(Verb::from_infinitive("ukufika").unwrap().lemma(), "fika"); + assert_eq!(Verb::from_infinitive("ukwenza").unwrap().lemma(), "enza"); + assert_eq!(Verb::from_infinitive("fika").unwrap().lemma(), "fika"); + assert!(Verb::from_infinitive("").is_err()); + assert!(Verb::from_infinitive("two words").is_err()); + } +}