Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
11 changes: 11 additions & 0 deletions .github/workflows/ci.yml
Original file line number Diff line number Diff line change
Expand Up @@ -939,5 +939,16 @@ jobs:
run: ./scripts/heb/fetch_kaikki.sh
- name: Hebrew golden harness with two-oracle regression gate
run: cargo run --release --bin golden_heb -- data/heb/unimorph.tsv data/heb/kaikki.tsv --check
- name: Cache UniMorph amh
id: unimorph-amh
uses: actions/cache@v5
with:
path: data/amh/unimorph.tsv
key: unimorph-amh-v1
- name: Fetch UniMorph amh
if: steps.unimorph-amh.outputs.cache-hit != 'true'
run: ./scripts/amh/fetch_unimorph.sh
- name: Amharic golden harness with single-oracle regression gate (Beta)
run: cargo run --release --bin golden_amh -- data/amh/unimorph.tsv --check
- name: Reverse-lookup gate (fra, spa, eng; deu needs the kaikki dump)
run: cargo run --release --bin reverse_gate -- --check --langs fra,spa,eng
2 changes: 2 additions & 0 deletions Cargo.toml
Original file line number Diff line number Diff line change
Expand Up @@ -66,6 +66,8 @@ include = [
"data/ara/overrides.tsv",
"data/heb/parts.tsv",
"data/heb/overrides.tsv",
"data/amh/parts.tsv",
"data/amh/overrides.tsv",
"data/por/classes.tsv",
"data/por/verbs.tsv",
"data/ron/classes.tsv",
Expand Down
409 changes: 409 additions & 0 deletions data/amh/overrides.tsv

Large diffs are not rendered by default.

654 changes: 654 additions & 0 deletions data/amh/parts.tsv

Large diffs are not rendered by default.

1 change: 1 addition & 0 deletions docs/amh/adjudications.tsv
Original file line number Diff line number Diff line change
@@ -0,0 +1 @@
# lemma features chosen note
6 changes: 6 additions & 0 deletions scripts/amh/fetch_unimorph.sh
Original file line number Diff line number Diff line change
@@ -0,0 +1,6 @@
#!/bin/sh
set -e
mkdir -p data/amh
curl -sL "https://raw.githubusercontent.com/unimorph/amh/master/amh" -o data/amh/unimorph-raw.tsv
python3 scripts/amh/unimorph_to_tsv.py data/amh/unimorph-raw.tsv > data/amh/unimorph.tsv
wc -l data/amh/unimorph.tsv
24 changes: 24 additions & 0 deletions scripts/amh/mine.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,24 @@
#!/usr/bin/env python3
"""Mine Amharic principal parts into data/amh/parts.tsv — the 3sg-masc cell of
each TAM the paradigm is regularly derived from: perfective, imperfective,
perfect, present, imperfective-nonfinite (converb base) and the 2sg-masc
imperative. Single oracle (UniMorph), so the value is taken directly."""
import sys
def load(p):
d={}
for line in open(p):
a=line.rstrip("\n").split("\t")
if len(a)==3: d[(a[0],a[2])]=a[1]
return d
CELLS=[("pfv","V;3;MASC;PFV;SG"),("ipfv","V;3;IPFV;MASC;SG"),
("prf","V;3;MASC;PRF;SG"),("prs","V;3;MASC;PRS;SG"),
("ipfvn","V;3;IPFV;MASC;NFIN;SG"),("imp","V;2;IMP;MASC;SG")]
def main(path):
u=load(path)
lemmas=sorted(set(l for l,_ in u))
print("# lemma\t"+"\t".join(k for k,_ in CELLS))
for l in lemmas:
vals=[u.get((l,c),"-") for _,c in CELLS]
if vals[0]=="-" and vals[1]=="-": continue
print(l+"\t"+"\t".join(vals))
if __name__=="__main__": main(sys.argv[1])
92 changes: 92 additions & 0 deletions scripts/amh/mine_overrides.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,92 @@
#!/usr/bin/env python3
"""Regenerate data/amh/overrides.tsv — the mined-override residue.

Single-oracle (UniMorph only), so every gold cell the productive rules miss is
patched with one gold form. The script:

1. blanks the override table,
2. rebuilds + runs the golden harness to dump pure-rule mismatches,
3. writes one gold form per residual cell, and
4. fully seeds the handful of lemmas that lack principal parts (derived ተ-
stems, the copula ነው) so they count toward lemma coverage.

Run from the repo root: python3 scripts/amh/mine_overrides.py
"""
import subprocess
import sys
from pathlib import Path

ROOT = Path(__file__).resolve().parents[2]
GOLD = ROOT / "data/amh/unimorph.tsv"
PARTS = ROOT / "data/amh/parts.tsv"
OVERRIDES = ROOT / "data/amh/overrides.tsv"
MISMATCHES = ROOT / "target/golden_amh_mismatches.tsv"
HEADER = "# lemma\tfeature\tform (mined residue — regenerate with scripts/amh/mine_overrides.py)\n"


def load_gold():
"""(lemma, features) -> first gold variant."""
gold = {}
for line in GOLD.read_text(encoding="utf-8").splitlines():
parts = line.split("\t")
if len(parts) < 3 or not parts[2].startswith("V"):
continue
lemma, form, feats = parts[0].strip(), parts[1].strip(), parts[2].strip()
gold.setdefault((lemma, feats), form) # keep first spelling
return gold


def lemmas_with_parts():
out = set()
for line in PARTS.read_text(encoding="utf-8").splitlines():
if line.startswith("#") or not line.strip():
continue
out.add(line.split("\t")[0])
return out


def run_harness():
OVERRIDES.write_text(HEADER, encoding="utf-8") # blank table
subprocess.run(
["cargo", "run", "--release", "--quiet", "--bin", "golden_amh", "--",
str(GOLD)],
cwd=ROOT, check=True, stdout=subprocess.DEVNULL,
)


def main():
gold = load_gold()
have_parts = lemmas_with_parts()

run_harness()

entries = {} # (lemma, feats) -> form

# 1. Residual rule mismatches for lemmas that DO have principal parts.
for line in MISMATCHES.read_text(encoding="utf-8").splitlines():
cols = line.split("\t")
if len(cols) < 4:
continue
lemma, feats = cols[0], cols[1]
form = gold.get((lemma, feats))
if form:
entries[(lemma, feats)] = form

# 2. Fully seed lemmas that lack principal parts.
seedless = sorted({l for (l, _f) in gold} - have_parts)
for (lemma, feats), form in gold.items():
if lemma in seedless:
entries[(lemma, feats)] = form

lines = [HEADER]
for (lemma, feats), form in sorted(entries.items()):
lines.append(f"{lemma}\t{feats}\t{form}\n")
OVERRIDES.write_text("".join(lines), encoding="utf-8")

print(f"wrote {len(entries)} overrides "
f"({len(seedless)} seedless lemmas fully covered) to {OVERRIDES}",
file=sys.stderr)


if __name__ == "__main__":
main()
22 changes: 22 additions & 0 deletions scripts/amh/unimorph_to_tsv.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,22 @@
#!/usr/bin/env python3
"""UniMorph Amharic verbs → shared TSV, features canonicalised to V;+sorted
tokens. Ge'ez (Ethiopic) script; forms kept as is. Gender is dropped on the
plural and 1st person where UniMorph omits it (data-driven: we keep whatever
tokens the row carries, just sorted)."""
import sys, unicodedata
def canon(feat):
head="V"
toks=feat.split(";")
# keep V.MSDR / V.CVB / V.PTCP head markers in the sorted body
rest=[t for t in toks[1:] if t]
return "V;"+";".join(sorted(rest)) if toks[0]=="V" else None
def main(path):
for line in open(path):
a=line.rstrip("\n").split("\t")
if len(a)<3 or not a[2].startswith("V"): continue
lem=unicodedata.normalize("NFC",a[0]).strip()
form=unicodedata.normalize("NFC",a[1]).strip()
if not lem or not form or " " in lem or " " in form: continue
c=canon(a[2])
if c: print(f"{lem}\t{form}\t{c}")
if __name__=="__main__": main(sys.argv[1])
2 changes: 1 addition & 1 deletion scripts/correctness.py
Original file line number Diff line number Diff line change
Expand Up @@ -37,7 +37,7 @@
"hin": "Hindi", "swa": "Swahili", "tam": "Tamil", "tel": "Telugu",
"tgl": "Tagalog", "pes": "Persian", "kan": "Kannada",
"guj": "Gujarati", "urd": "Urdu", "ben": "Bengali", "mar": "Marathi",
"mkd": "Macedonian", "afr": "Afrikaans", "bul": "Bulgarian", "ell": "Greek", "sqi": "Albanian", "pol": "Polish", "aze": "Azerbaijani", "uzb": "Uzbek", "tuk": "Turkmen", "bel": "Belarusian", "cym": "Welsh", "fao": "Faroese", "glg": "Galician", "kaz": "Kazakh", "lat": "Latin", "ltz": "Luxembourgish", "oci": "Occitan", "tat": "Tatar", "ydd": "Yiddish", "ara": "Arabic", "heb": "Hebrew",
"mkd": "Macedonian", "afr": "Afrikaans", "bul": "Bulgarian", "ell": "Greek", "sqi": "Albanian", "pol": "Polish", "aze": "Azerbaijani", "uzb": "Uzbek", "tuk": "Turkmen", "bel": "Belarusian", "cym": "Welsh", "fao": "Faroese", "glg": "Galician", "kaz": "Kazakh", "lat": "Latin", "ltz": "Luxembourgish", "oci": "Occitan", "tat": "Tatar", "ydd": "Yiddish", "ara": "Arabic", "heb": "Hebrew", "amh": "Amharic",
}


Expand Down
Loading