diff --git a/.github/workflows/ci-ml.yml b/.github/workflows/ci-ml.yml index 92ecfc1..2ad7445 100644 --- a/.github/workflows/ci-ml.yml +++ b/.github/workflows/ci-ml.yml @@ -91,7 +91,9 @@ jobs: sudo apt-get update # libgl1 replaces libgl1-mesa-glx on Ubuntu 24.04+ sudo apt-get install -y \ + libegl1 \ libgl1 \ + libusb-1.0-0 \ libglib2.0-0 \ libgeos-dev \ libproj-dev @@ -105,6 +107,11 @@ jobs: pip install -e ".[dev]" pip install python-docx + # Open3D loads EGL during import, even when only reading PLY on CPU. + # Report missing shared libraries before the same import breaks many tests. + - name: Check Open3D runtime libraries + run: python -c "import open3d; print(open3d.__version__)" + # photogrammetry/ used to be listed here. It has no tracked files - the # photo path was dropped and its modules deleted - so on a fresh checkout # the directory does not exist and ruff exits with E902 "cannot find the diff --git a/AGENTS.md b/AGENTS.md index c60e74f..d1023c8 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -20,7 +20,7 @@ backend, web, สถานะจริง, known bugs, roadmap. **อ่าน ด้วย `tlsep` baseline แยก **ลำต้น(wood)/ใบ(leaf)** → วัด **DBH + ความสูง** → คำนวณ **ชีวมวล→carbon stock→CO₂e estimate** จาก `species_db.csv` หรือ Chave fallback พร้อม provenance ส่วน PointNet++, photogrammetry, marketplace และ certification ต้องรายงานตามสถานะ Experimental/Planned -สร้างเพื่อแข่ง **NSC 2026 หมวด 14 (อุดมศึกษา)** +เป้าหมายปัจจุบัน: **ถูกวิทยาศาสตร์ ตีพิมพ์ได้ และใช้งานได้จริง** — ดู CLAUDE.md --- @@ -31,7 +31,7 @@ apps/web/ Next.js 14 (App Router, TS, Tailwind, shadcn) — landing + das services/api/ FastAPI (SQLAlchemy async, asyncpg, Pydantic v2, Alembic) — REST + async-job worker services/ml/ Point-cloud pipeline — tlsep default, PointNet++ Experimental, 8-step pipeline + allometric docs/ เอกสาร (PROJECT_SPEC.md, ml/, learning/, decisions/, superpowers/) -proposal/ NSC proposal +proposal/ เอกสารข้อเสนอฉบับเดิม (historical) memory/ project memory ``` @@ -136,7 +136,7 @@ Mean IoU `0.613`, accuracy `0.831` และ held-out loader ถูกใช้ ## Preferences ของผู้ใช้ - ตอบ **ภาษาไทย** (technical terms EN ได้) -- โฟกัส "ทำให้กรรมการ NSC ว้าว" — Deep Tech + visual storytelling +- **ความถูกต้องมาก่อนความน่าประทับใจ** — กรอบ "ทำให้กรรมการว้าว" เลิกใช้แล้ว - **อย่า over-engineer** — prototype ที่เสร็จ > vision สมบูรณ์แต่ไม่เสร็จ - ทุก decision ที่มีค่าใช้จ่าย → "นักศึกษาจ่ายไหวไหม" - verify ก่อนเคลม — รันจริง/ดูผลจริง อย่าเดา diff --git a/CLAUDE.md b/CLAUDE.md index 63d6b1f..5aa299b 100644 --- a/CLAUDE.md +++ b/CLAUDE.md @@ -62,10 +62,10 @@ **โน้ตบุ๊ก Acer Predator PHN16-71 ค้างตายสนิท 19 ครั้ง** (15 ครั้งใน 3 วัน 21-23 ส.ค.) ไม่มี bugcheck ไม่มี dump ไม่มี WHEA -**สาเหตุ (วัดได้แล้ว 23 ส.ค.): ไฟ AC หลุดเข้าหลุดออก + แบตเสื่อม 72% ที่ 53 cycles** +**บันทึกการตรวจ 23 ส.ค.: ไฟ AC หลุดเข้าหลุดออก + แบตเสื่อม 72% ที่ 53 cycles** Event 105 บันทึก AC หลุด/กลับ 4 ครั้งใน 10 วินาที (20:00:01-20:00:11) แล้วเครื่องตาย 20:13:03 -RAM ตัดออกแล้วด้วยการวัด (Memory Diagnostic 23 ส.ค. 13:24 = no errors) -**ไม่ใช่ซอฟต์แวร์ ไม่ใช่ workload** — รายละเอียดใน memory `machine-freeze-root-cause` +Memory Diagnostic 23 ส.ค. 13:24 รายงาน no errors ผลครั้งเดียวไม่ได้ตัดความเป็นไปได้ของปัญหา RAM หรือซอฟต์แวร์ทั้งหมด +รายละเอียดเดิมอยู่ใน memory `machine-freeze-root-cause` รอบตรวจ PR นี้ไม่ได้ตรวจ hardware ใหม่ จึงไม่ยืนยันว่าสาเหตุถูกพิสูจน์ครบหรือแก้ปัญหาแล้ว - **ห้ามรัน full test suite บนเครื่องนี้** (`pytest tests/` ชุดเต็มใช้ ~29 นาที และเคยทิ้ง scratch ไว้ 93 GB จนเครื่องอืด) — ให้ GitHub Actions รับไป diff --git a/README.md b/README.md index e02f978..ca89521 100644 --- a/README.md +++ b/README.md @@ -4,9 +4,8 @@ ### ประเมินชีวมวล คาร์บอน และ CO₂e ของต้นไม้จาก 3D point cloud พร้อมหลักฐานที่ตรวจสอบย้อนกลับได้ -NSC 2026 หมวด 14 · Evidence-gated ML · 3D visual verification +Evidence-gated ML · Destructively validated · 3D visual verification -[![NSC 2026](https://img.shields.io/badge/NSC-2026-2D6A4F)](https://www.nstda.or.th/sims) [![License: MIT](https://img.shields.io/github/license/Remote55/treeq?color=52B788)](LICENSE) [![CI · ML](https://github.com/Remote55/treeq/actions/workflows/ci-ml.yml/badge.svg)](https://github.com/Remote55/treeq/actions/workflows/ci-ml.yml) [![CI · API](https://github.com/Remote55/treeq/actions/workflows/ci-api.yml/badge.svg)](https://github.com/Remote55/treeq/actions/workflows/ci-api.yml) @@ -190,7 +189,7 @@ apps/web/ Next.js landing, dashboard และ 3D viewer services/api/ FastAPI synchronous analyze endpoint services/ml/ 8-stage point-cloud pipeline, training และ evaluation docs/ master spec, ML evidence, capability matrix และ decisions -proposal/ เอกสารข้อเสนอโครงงาน NSC +proposal/ เอกสารข้อเสนอฉบับเดิม (historical — เก็บไว้เพื่อ trace การตัดสินใจ) scripts/ truth sync และ report builder ``` diff --git a/apps/web/src/app/layout.tsx b/apps/web/src/app/layout.tsx index 67e615e..27786b0 100644 --- a/apps/web/src/app/layout.tsx +++ b/apps/web/src/app/layout.tsx @@ -11,17 +11,17 @@ export const metadata: Metadata = { template: '%s | TreeQ Carbon Platform', }, description: - 'แพลตฟอร์มประเมินชีวมวล คาร์บอน และ CO₂e จาก 3D point cloud พร้อมหลักฐานที่ตรวจสอบย้อนกลับได้ — NSC 2026', + 'แพลตฟอร์มประเมินชีวมวล คาร์บอน และ CO₂e จาก 3D point cloud พร้อมหลักฐานที่ตรวจสอบย้อนกลับได้ทุกตัวเลข', keywords: [ 'TreeQ Carbon Platform', 'Carbon Stock Estimate', 'LiDAR', 'Point Cloud', - 'Deep Learning', + 'Terrestrial Laser Scanning', 'Tree Biomass', 'Allometric Equation', 'TGO', - 'NSC 2026', + 'T-VER', 'Climate Tech', 'Sustainable Innovation', ], diff --git a/apps/web/src/app/page.test.tsx b/apps/web/src/app/page.test.tsx index 3582cad..8c0490d 100644 --- a/apps/web/src/app/page.test.tsx +++ b/apps/web/src/app/page.test.tsx @@ -88,6 +88,60 @@ describe('Landing evidence contract', () => { expect(markup).not.toMatch(/PointNet\+\+[^<]*Default/); }); + // The accuracy panel showed one diameter error: 0.90 cm, from 65 isolated + // trees in Belgium. That number is true and it is not the one this product + // is about. TreeQ is aimed at tropical forest; the tropical cohort has been + // measured — 61 trees felled and weighed in Cameroon — and the panel had no + // way to say so, because scripts/sync_truth.py never emitted the block. + // + // A visitor reading 0.90 cm was reading a temperate figure as the accuracy + // of a tropical product, and could not tell that the shipped gate refuses + // 33 of the 60 measurable trees in the cohort that does apply. + describe('the tropical cohort is not hidden behind the temperate one', () => { + const cameroon = CORE_DEMO_EVIDENCE.validation.cameroon61; + + it('states the tropical diameter error beside the temperate one', () => { + const markup = renderToStaticMarkup(); + + expect(markup).toContain(cameroon.dbhGateAppliedMaeCm.toFixed(2)); + expect(markup).toContain( + CORE_DEMO_EVIDENCE.validation.demol65.dbhMaeCm.toFixed(2), + ); + }); + + it('says how many trees the gate refused', () => { + const markup = renderToStaticMarkup(); + + // Without both counts the headline is an average over an unnamed + // subset, and 27 of 60 reads as 60 of 60. + expect(markup).toContain(String(cameroon.gatePassedTrees)); + expect(markup).toContain(String(cameroon.gateRefusedTrees)); + }); + + it('names the cohort as tropical and destructively harvested', () => { + const markup = renderToStaticMarkup(); + + expect(markup).toContain('เขตร้อน'); + }); + + it('quotes no accuracy figure that is not in the generated evidence', () => { + const markup = renderToStaticMarkup(); + + // The ungated mean error is cohort-specific, not an upper error bound. + expect(markup).not.toContain('1.37 ซม. จากต้นไม้ 60'); + }); + + it('distinguishes the cohort, measurable subset and ungated mean without inventing refusal causes', () => { + const markup = renderToStaticMarkup(); + + expect(markup).toContain(`ทั้งหมด ${cameroon.trees} ต้น`); + expect(markup).toContain(`วัดได้ ${cameroon.treesMeasured} ต้น`); + expect(markup).toContain('ไม่ใช่ขอบบนของความผิดพลาด'); + expect(markup).not.toContain('ส่วนใหญ่เป็นต้นใหญ่ที่มีพูพอน'); + expect(markup).not.toContain('ขอบบนของขั้นวัดขนาด'); + }); + }); + it('renders the five evidence-led editorial beats', () => { const markup = renderToStaticMarkup(); @@ -206,4 +260,4 @@ describe('Landing evidence contract', () => { expect(offenders).toEqual([]); }); -}); \ No newline at end of file +}); diff --git a/apps/web/src/app/page.tsx b/apps/web/src/app/page.tsx index 7f3278b..b8126ce 100644 --- a/apps/web/src/app/page.tsx +++ b/apps/web/src/app/page.tsx @@ -17,7 +17,7 @@ import { decimate, parsePly } from '@/lib/ply-loader'; import { LIMITS_LABEL_TH } from '@/lib/upload-limits'; const { baseline, candidate } = CORE_DEMO_EVIDENCE; -const { demol65, pointnetIndependent } = CORE_DEMO_EVIDENCE.validation; +const { demol65, cameroon61, pointnetIndependent } = CORE_DEMO_EVIDENCE.validation; // The separation quality that describes what ships: the default backend, // measured on a cohort it never trained on. The Wan held-out numbers belong to // the candidate and to a split that also picked its best epoch. @@ -59,7 +59,11 @@ const JOURNEY = [ technical: [ 'วัดเส้นผ่านศูนย์กลางลำต้นที่ระดับ 1.3 เมตร', 'คำนวณปริมาตรจากแบบจำลองทรงกระบอก QSM', - `ค่าคลาดเคลื่อนเฉลี่ย ${demol65.dbhMaeCm.toFixed(2)} ซม. จากต้นไม้จริง 65 ต้น`, + `ค่าคลาดเคลื่อนเฉลี่ย ${demol65.dbhMaeCm.toFixed(2)} ซม. จากต้นไม้เขตอบอุ่น 65 ต้น`, + // The tropical figure belongs beside the temperate one, not instead of + // it: the cohorts answer different questions and both are quoted. + `${cameroon61.dbhGateAppliedMaeCm.toFixed(2)} ซม. จากต้นไม้เขตร้อนที่โค่นและชั่งจริง ` + + `(${cameroon61.gatePassedTrees} ต้นที่ผ่านเกณฑ์คุณภาพการวัด)`, ], }, { @@ -195,7 +199,7 @@ useEffect(() => { - TREEQ CARBON / NSC 2026 + TREEQ CARBON / ตรวจสอบย้อนกลับได้ทุกตัวเลข @@ -544,11 +548,50 @@ useEffect(() => { /> + {/* Two cohorts, because they answer different questions and the + tropical one is the question this product is for. The panel used to + carry only the Belgian figure, which is a true number about + temperate isolated trees presented as the accuracy of a platform + aimed at tropical forest. */}
+
+ + +
+
+ +
+ + {/* The refusal count is not a caveat, it is half the headline. An + average over the trees that passed says nothing about the trees + that did not, and 27 of 60 reads as 60 of 60 without this. */} +
+ +
+ +
+
@@ -566,11 +609,11 @@ useEffect(() => { TreeQ Carbon Platform
- Prototype for NSC 2026 · หมวด 14 + Prototype · ผลลัพธ์เป็นค่าประมาณ ไม่ใช่คาร์บอนเครดิตที่ผ่านการรับรอง ); -} \ No newline at end of file +} diff --git a/apps/web/src/generated/core-demo-evidence.ts b/apps/web/src/generated/core-demo-evidence.ts index e531063..c284b1c 100644 --- a/apps/web/src/generated/core-demo-evidence.ts +++ b/apps/web/src/generated/core-demo-evidence.ts @@ -23,6 +23,19 @@ export const CORE_DEMO_EVIDENCE = { dbhMaeCm: 0.898318, volumeMapePct: 11.520556, }, + cameroon61: { + trees: 61, + treesMeasured: 60, + dbhGateAppliedMaeCm: 1.369868, + gatePassedTrees: 27, + gateRefusedTrees: 33, + dbhMaeCmSmallStems: 1.683709, + dbhMaeCmSmallStemsN: 31, + dbhMaeCm: 11.254575, + chaveRouteBApePctMedian: 13.999223, + tverRouteBApePctMedian: 20.850468, + chaveMeasurementSharePctMedian: 5.800567, + }, pointnetIndependent: { verdict: "FAIL_METRICS", baseline: { diff --git a/docs/PROJECT_SPEC.md b/docs/PROJECT_SPEC.md index c912c43..760d039 100644 --- a/docs/PROJECT_SPEC.md +++ b/docs/PROJECT_SPEC.md @@ -14,6 +14,8 @@ - PointNet++: **Experimental**, not promoted; reviewed evidence never changes the default automatically. - Wan 2021 held-out: Wood IoU `0.418`, Leaf IoU `0.808`, Mean IoU `0.613`, accuracy `0.831`. The held-out loader was also used for best-epoch selection. - Demol isolated-tree validation (65 trees): DBH MAE `0.898318 cm`; Volume MAPE `11.520556%`. This is not an eight-stage or carbon validation. +- Cameroon destructive tropical validation (61 trees, 60 measurable): DBH MAE `1.369868 cm` over the `27` trees the shipped gate passes, `33` refused; ungated DBH MAE `11.254575 cm` over all `60` measurable trees when forced to answer. This cohort mean is not an upper bound on individual-tree error. +- Cameroon field-geometry allometric AGB, scored against harvested mass: Chave 2014 median APE `13.999223%` against T-VER `20.850468%`. The Chave median paired APE difference (cloud geometry minus field geometry) is `5.800567` percentage points, not a causal error decomposition. These clouds arrive leaf-stripped and are single trees, so this validates neither stage 5 nor stages 1-4, and Cameroon is not Thailand. - Independent PointNet review: verdict `FAIL_METRICS`; candidate/baseline external macro Wood IoU `0.23728726507501768`/`0.1958779956856453`. - Independent downstream candidate/baseline: DBH MAE `1.1591405814498605`/`1.1339476465903928` cm; Height MAE `0.9508502244897976`/`0.5433234000000015` m; Volume MAPE `21.74924193798788`/`18.928262273343613`%; measurable trees `49`/`65`. - Deterministic core demo: `3` trees, `1036.09 kg C`, `3798.99 kg CO2e`; analyzed commit `8cf3058c1f61` with a clean worktree. diff --git a/docs/ml/PIPELINE.md b/docs/ml/PIPELINE.md index dba7222..445a201 100644 --- a/docs/ml/PIPELINE.md +++ b/docs/ml/PIPELINE.md @@ -11,6 +11,8 @@ - PointNet++: **Experimental**, not promoted; reviewed evidence never changes the default automatically. - Wan 2021 held-out: Wood IoU `0.418`, Leaf IoU `0.808`, Mean IoU `0.613`, accuracy `0.831`. The held-out loader was also used for best-epoch selection. - Demol isolated-tree validation (65 trees): DBH MAE `0.898318 cm`; Volume MAPE `11.520556%`. This is not an eight-stage or carbon validation. +- Cameroon destructive tropical validation (61 trees, 60 measurable): DBH MAE `1.369868 cm` over the `27` trees the shipped gate passes, `33` refused; ungated DBH MAE `11.254575 cm` over all `60` measurable trees when forced to answer. This cohort mean is not an upper bound on individual-tree error. +- Cameroon field-geometry allometric AGB, scored against harvested mass: Chave 2014 median APE `13.999223%` against T-VER `20.850468%`. The Chave median paired APE difference (cloud geometry minus field geometry) is `5.800567` percentage points, not a causal error decomposition. These clouds arrive leaf-stripped and are single trees, so this validates neither stage 5 nor stages 1-4, and Cameroon is not Thailand. - Independent PointNet review: verdict `FAIL_METRICS`; candidate/baseline external macro Wood IoU `0.23728726507501768`/`0.1958779956856453`. - Independent downstream candidate/baseline: DBH MAE `1.1591405814498605`/`1.1339476465903928` cm; Height MAE `0.9508502244897976`/`0.5433234000000015` m; Volume MAPE `21.74924193798788`/`18.928262273343613`%; measurable trees `49`/`65`. - Deterministic core demo: `3` trees, `1036.09 kg C`, `3798.99 kg CO2e`; analyzed commit `8cf3058c1f61` with a clean worktree. diff --git a/docs/ml/WHAT_CI_DOES_NOT_CHECK.md b/docs/ml/WHAT_CI_DOES_NOT_CHECK.md index 7afde92..b354439 100644 --- a/docs/ml/WHAT_CI_DOES_NOT_CHECK.md +++ b/docs/ml/WHAT_CI_DOES_NOT_CHECK.md @@ -19,6 +19,7 @@ against trees that were cut down and weighed: | `test_published_evidence_is_current.py` | whether the accuracy figures in the proposal and on the dashboard are still what the pipeline produces — see [DEMOL_EVIDENCE_CHAIN.md](DEMOL_EVIDENCE_CHAIN.md) | | `test_dbh_bias_by_species.py` (part) | the control for the bark finding: whether an independent QSM under-reads the same trees — see [DBH_BIAS_AND_BARK.md](DBH_BIAS_AND_BARK.md) | | `test_cameroon_eval.py` (part) | the cohort loader against the real 1.29 GB archive: the 61-tree keying that excludes `ID_56`, min-Z normalization and the point cap, seeded-sample determinism, and that the five irregularly-formatted clouds parse and are flagged as repaired; the Chave-on-the-tape wiring smoke test and the Chave-vs-T-VER route B comparison, both costed from the real cohort's tape DBH and felled height — see [CAMEROON_EVIDENCE_CHAIN.md](CAMEROON_EVIDENCE_CHAIN.md) | +| `test_cameroon_evidence_is_current.py` (part) | whether the tropical figures on the dashboard and in the truth block still re-derive from the 61 trees — the Cameroon counterpart of the row above it, and the same limitation | They skip because `services/ml/data/raw/zenodo_belgium/` and `services/ml/woodleaf_pn2.pt` are not in git — point clouds for 65 trees and a @@ -59,6 +60,19 @@ cannot cross the gap is anything that has to re-run the pipeline over the point clouds. The three assertions in that file which need the cohort itself still skip, and they are listed in the table above. +`docs/evidence/cameroon_61/result.json` is committed for the same reason, and +`scripts/sync_truth.py`'s `validate_cameroon` re-hashes it and compares every +one of the twenty-five figures the manifest publishes against it on each +`--check`. That runs on CI and cannot skip, so the manifest can no longer drift +from its own artefact. It says nothing about whether the artefact still matches +the trees — that is `test_cameroon_evidence_is_current.py`, and it skips here +for the usual reason. + +Until both existed, the tropical block was the least guarded evidence in the +repository: `load_manifest` did not require it, nothing re-hashed it, and no +test held the two to each other, while the 65-tree temperate cohort had all +three. The newest and strongest evidence had the weakest gate. + ## What this means in practice - **CI protects the code, not the measurement.** Ruff, mypy, the unit and diff --git a/docs/ml/WOODLEAF_RESULTS.md b/docs/ml/WOODLEAF_RESULTS.md index a79500a..b4ff200 100644 --- a/docs/ml/WOODLEAF_RESULTS.md +++ b/docs/ml/WOODLEAF_RESULTS.md @@ -14,6 +14,8 @@ - PointNet++: **Experimental**, not promoted; reviewed evidence never changes the default automatically. - Wan 2021 held-out: Wood IoU `0.418`, Leaf IoU `0.808`, Mean IoU `0.613`, accuracy `0.831`. The held-out loader was also used for best-epoch selection. - Demol isolated-tree validation (65 trees): DBH MAE `0.898318 cm`; Volume MAPE `11.520556%`. This is not an eight-stage or carbon validation. +- Cameroon destructive tropical validation (61 trees, 60 measurable): DBH MAE `1.369868 cm` over the `27` trees the shipped gate passes, `33` refused; ungated DBH MAE `11.254575 cm` over all `60` measurable trees when forced to answer. This cohort mean is not an upper bound on individual-tree error. +- Cameroon field-geometry allometric AGB, scored against harvested mass: Chave 2014 median APE `13.999223%` against T-VER `20.850468%`. The Chave median paired APE difference (cloud geometry minus field geometry) is `5.800567` percentage points, not a causal error decomposition. These clouds arrive leaf-stripped and are single trees, so this validates neither stage 5 nor stages 1-4, and Cameroon is not Thailand. - Independent PointNet review: verdict `FAIL_METRICS`; candidate/baseline external macro Wood IoU `0.23728726507501768`/`0.1958779956856453`. - Independent downstream candidate/baseline: DBH MAE `1.1591405814498605`/`1.1339476465903928` cm; Height MAE `0.9508502244897976`/`0.5433234000000015` m; Volume MAPE `21.74924193798788`/`18.928262273343613`%; measurable trees `49`/`65`. - Deterministic core demo: `3` trees, `1036.09 kg C`, `3798.99 kg CO2e`; analyzed commit `8cf3058c1f61` with a clean worktree. diff --git a/scripts/sync_truth.py b/scripts/sync_truth.py index ae33d46..cf36331 100644 --- a/scripts/sync_truth.py +++ b/scripts/sync_truth.py @@ -77,6 +77,45 @@ def _require_sha256(value: Any, label: str) -> None: DEMOL_RESULT_PATH = "docs/evidence/demol_65/result.json" +#: The Cameroon figures this manifest publishes, mapped to the key each one has +#: in the artefact that derives them. +#: +#: Almost all of them are named the same on both sides. `trees` is the one that +#: is not: the manifest names the cohort the way the other validation blocks +#: name theirs, and `services/ml/scripts/derive_cameroon_evidence.py` writes it +#: as `cohort_size`. Mapping rather than assuming equality is what lets the +#: comparison below cover every published number instead of the subset whose +#: names happen to line up. +CAMEROON_PUBLISHED_FIELDS = { + "trees": "cohort_size", + "trees_measured": "trees_measured", + "trees_excluded": "trees_excluded", + "dbh_gate_applied_mae_cm": "dbh_gate_applied_mae_cm", + "gate_passed_trees": "gate_passed_trees", + "gate_refused_trees": "gate_refused_trees", + "gate_min_dbh_fit_quality": "gate_min_dbh_fit_quality", + "dbh_mae_cm_small_stems": "dbh_mae_cm_small_stems", + "dbh_mae_cm_small_stems_n": "dbh_mae_cm_small_stems_n", + "dbh_mae_cm": "dbh_mae_cm", + "dbh_bias_cm": "dbh_bias_cm", + "dbh_mae_vs_reference_cm": "dbh_mae_vs_reference_cm", + "dbh_bias_vs_reference_cm": "dbh_bias_vs_reference_cm", + "height_mae_m": "height_mae_m", + "height_bias_m": "height_bias_m", + "volume_mape_pct": "volume_mape_pct", + "volume_vs_reference_qsm_mape_pct": "volume_vs_reference_qsm_mape_pct", + "chave_route_a_ape_pct_median": "chave_route_a_ape_pct_median", + "chave_route_b_ape_pct_median": "chave_route_b_ape_pct_median", + "chave_measurement_share_pct_median": "chave_measurement_share_pct_median", + "tver_route_a_ape_pct_median": "tver_route_a_ape_pct_median", + "tver_route_b_ape_pct_median": "tver_route_b_ape_pct_median", + "tver_measurement_share_pct_median": "tver_measurement_share_pct_median", + "chave_vs_tver_route_b_chave_closer_count": "chave_vs_tver_route_b_chave_closer_count", + "chave_vs_tver_route_b_tver_closer_count": "chave_vs_tver_route_b_tver_closer_count", +} + +CAMEROON_RESULT_PATH = "docs/evidence/cameroon_61/result.json" + #: Documents that quote accuracy figures in hand-written prose. #: #: The TREEQ_TRUTH block is regenerated from the manifest, so the numbers inside @@ -263,6 +302,68 @@ def validate_demol(block: Any, *, repo_root: str | Path | None) -> None: ) +def validate_cameroon(block: Any, *, repo_root: str | Path | None) -> None: + """Check the published Cameroon figures against the artefact that derived them. + + Written to the same standard as `validate_demol`, because until it existed + the tropical block was held to none. `load_manifest` did not require it, + nothing re-hashed `docs/evidence/cameroon_61/result.json`, and no test + compared the two -- so the only cohort in this repository that has been cut + down and weighed, and the only one that reaches the allometric stage at all, + was the least guarded evidence in it. + + The asymmetry mattered because `published_figure_values` reads this block to + decide whether a figure quoted in prose is current. A manifest number that + had drifted from its own artefact would have been used to certify documents + quoting the drifted number. + """ + if not isinstance(block, dict): + raise ValueError("validation.cameroon_61 must be an object") + _require_keys( + block, + {"result_path", "result_sha256", *CAMEROON_PUBLISHED_FIELDS}, + "validation.cameroon_61", + ) + if block["result_path"] != CAMEROON_RESULT_PATH: + raise ValueError( + f"validation.cameroon_61 result_path must be {CAMEROON_RESULT_PATH}" + ) + _require_sha256(block["result_sha256"], "validation.cameroon_61 result_sha256") + + if repo_root is None: + # Structure only, matching validate_demol: sync() always supplies + # repo_root, so the comparison below runs on every `--check`. + return + + result_file = Path(repo_root) / CAMEROON_RESULT_PATH + if not result_file.is_file(): + raise ValueError( + f"{CAMEROON_RESULT_PATH} is missing; the published figures have no source" + ) + raw = result_file.read_bytes() + digest = hashlib.sha256(raw).hexdigest() + if digest != block["result_sha256"]: + raise ValueError( + f"{CAMEROON_RESULT_PATH} has changed since it was reviewed " + f"(recorded {block['result_sha256']}, found {digest})" + ) + + metrics = json.loads(raw.decode("utf-8")).get("metrics") + if not isinstance(metrics, dict): + raise ValueError(f"{CAMEROON_RESULT_PATH} has no metrics block") + disagreeing = sorted( + manifest_key + for manifest_key, metrics_key in CAMEROON_PUBLISHED_FIELDS.items() + if block[manifest_key] != metrics.get(metrics_key) + ) + if disagreeing: + raise ValueError( + "validation.cameroon_61 disagrees with the derived result for " + f"{disagreeing}; re-run derive_cameroon_evidence.py rather than " + "editing the manifest" + ) + + def load_manifest( path: str | Path, *, repo_root: str | Path | None = None ) -> dict[str, Any]: @@ -317,12 +418,19 @@ def load_manifest( raise ValueError("promotion evidence cannot auto-promote PointNet++") validation = manifest["validation"] - _require_keys(validation, {"wan_held_out", "demol_65"}, "validation") + # cameroon_61 is required, not optional. It was optional, which meant the + # tropical cohort could be dropped from the manifest and every gate would + # still pass -- and the figures the site publishes would silently revert to + # the temperate ones. + _require_keys( + validation, {"wan_held_out", "demol_65", "cameroon_61"}, "validation" + ) wan = validation["wan_held_out"] for name, expected in EXPECTED_WAN.items(): if wan.get(name) != expected: raise ValueError(f"Wan held-out {name} must equal {expected}") validate_demol(validation["demol_65"], repo_root=repo_root) + validate_cameroon(validation["cameroon_61"], repo_root=repo_root) independent = validation.get("pointnet_independent") if independent is not None: if repo_root is None: @@ -430,6 +538,7 @@ def render_typescript(manifest: dict[str, Any]) -> str: """Render the immutable subset used by the Next.js UI.""" wan = manifest["validation"]["wan_held_out"] demol = manifest["validation"]["demol_65"] + cameroon = manifest["validation"]["cameroon_61"] core = manifest["core_demo"] candidate = manifest["candidate"] independent = manifest["validation"].get("pointnet_independent") @@ -460,6 +569,25 @@ def render_typescript(manifest: dict[str, Any]) -> str: f" dbhMaeCm: {demol['dbh_mae_cm']},", f" volumeMapePct: {demol['volume_mape_pct']},", " },", + # The tropical cohort, and the three numbers that have to travel + # together. dbhGateAppliedMaeCm is what a user is handed; it is an + # average over gatePassedTrees of gatePassedTrees + gateRefusedTrees, + # and dbhMaeCm is what the same stage produces when forced to answer + # for every tree. Publishing the first alone is how 27 of 60 reads + # as 60 of 60. + " cameroon61: {", + f" trees: {cameroon['trees']},", + f" treesMeasured: {cameroon['trees_measured']},", + f" dbhGateAppliedMaeCm: {cameroon['dbh_gate_applied_mae_cm']},", + f" gatePassedTrees: {cameroon['gate_passed_trees']},", + f" gateRefusedTrees: {cameroon['gate_refused_trees']},", + f" dbhMaeCmSmallStems: {cameroon['dbh_mae_cm_small_stems']},", + f" dbhMaeCmSmallStemsN: {cameroon['dbh_mae_cm_small_stems_n']},", + f" dbhMaeCm: {cameroon['dbh_mae_cm']},", + f" chaveRouteBApePctMedian: {cameroon['chave_route_b_ape_pct_median']},", + f" tverRouteBApePctMedian: {cameroon['tver_route_b_ape_pct_median']},", + f" chaveMeasurementSharePctMedian: {cameroon['chave_measurement_share_pct_median']},", + " },", *( [ " pointnetIndependent: {", @@ -503,6 +631,7 @@ def render_truth_block(manifest: dict[str, Any]) -> str: """Render a compact human-readable snapshot for controlled documents.""" wan = manifest["validation"]["wan_held_out"] demol = manifest["validation"]["demol_65"] + cameroon = manifest["validation"]["cameroon_61"] core = manifest["core_demo"] candidate = manifest["candidate"] independent = manifest["validation"].get("pointnet_independent") @@ -531,6 +660,26 @@ def render_truth_block(manifest: dict[str, Any]) -> str: f"`{demol['dbh_mae_cm']} cm`; Volume MAPE " f"`{demol['volume_mape_pct']}%`. This is not an eight-stage or carbon validation." ), + ( + f"- Cameroon destructive tropical validation " + f"({cameroon['trees']} trees, {cameroon['trees_measured']} measurable): " + f"DBH MAE `{cameroon['dbh_gate_applied_mae_cm']} cm` over the " + f"`{cameroon['gate_passed_trees']}` trees the shipped gate passes, " + f"`{cameroon['gate_refused_trees']}` refused; " + f"ungated DBH MAE `{cameroon['dbh_mae_cm']} cm` over all " + f"`{cameroon['trees_measured']}` measurable trees when forced to answer. " + "This cohort mean is not an upper bound on individual-tree error." + ), + ( + f"- Cameroon field-geometry allometric AGB, scored against harvested mass: Chave 2014 " + f"median APE `{cameroon['chave_route_b_ape_pct_median']}%` against " + f"T-VER `{cameroon['tver_route_b_ape_pct_median']}%`. The Chave " + "median paired APE difference (cloud geometry minus field geometry) is " + f"`{cameroon['chave_measurement_share_pct_median']}` percentage points, " + "not a causal error decomposition. These clouds arrive leaf-stripped and are single " + "trees, so this validates neither stage 5 nor stages 1-4, and " + "Cameroon is not Thailand." + ), *( [ ( diff --git a/scripts/tests/test_review_pointnet_evidence.py b/scripts/tests/test_review_pointnet_evidence.py index 3276536..1ca3bd6 100644 --- a/scripts/tests/test_review_pointnet_evidence.py +++ b/scripts/tests/test_review_pointnet_evidence.py @@ -13,6 +13,8 @@ import pytest from scripts.review_pointnet_evidence import _load_json, _validate_result, import_reviewed_result from scripts.sync_truth import ( + CAMEROON_PUBLISHED_FIELDS, + CAMEROON_RESULT_PATH, CONTROLLED_DOCS, DEMOL_PUBLISHED_FIELDS, DEMOL_RESULT_PATH, @@ -35,6 +37,18 @@ } | {"trees": 65} DEMOL_ARTEFACT = json.dumps({"metrics": DEMOL_METRICS}, sort_keys=True).encode("utf-8") DEMOL_SHA256 = hashlib.sha256(DEMOL_ARTEFACT).hexdigest() + +#: The tropical block is required by load_manifest, so this fixture carries one +#: too. Nothing here tests it -- these are the PointNet review's tests -- it +#: just has to be present and self-consistent for the manifest to load. +CAMEROON_METRICS: dict[str, object] = { + metrics_key: round(index * 0.4, 6) + for index, metrics_key in enumerate(CAMEROON_PUBLISHED_FIELDS.values(), start=1) +} +CAMEROON_ARTEFACT = json.dumps({"metrics": CAMEROON_METRICS}, sort_keys=True).encode( + "utf-8" +) +CAMEROON_SHA256 = hashlib.sha256(CAMEROON_ARTEFACT).hexdigest() EXTERNAL_IDS = tuple(f"tree-{index:02d}" for index in range(1, 11)) review_pointnet_evidence = sys.modules[_validate_result.__module__] @@ -110,6 +124,14 @@ def _manifest() -> dict[str, object]: "result_sha256": DEMOL_SHA256, **DEMOL_METRICS, }, + "cameroon_61": { + "result_path": CAMEROON_RESULT_PATH, + "result_sha256": CAMEROON_SHA256, + **{ + manifest_key: CAMEROON_METRICS[metrics_key] + for manifest_key, metrics_key in CAMEROON_PUBLISHED_FIELDS.items() + }, + }, }, "capabilities": [ { @@ -370,6 +392,9 @@ def reviewed_repo(tmp_path: Path) -> tuple[Path, Path, Path]: demol_artefact = repo / DEMOL_RESULT_PATH demol_artefact.parent.mkdir(parents=True, exist_ok=True) demol_artefact.write_bytes(DEMOL_ARTEFACT) + cameroon_artefact = repo / CAMEROON_RESULT_PATH + cameroon_artefact.parent.mkdir(parents=True, exist_ok=True) + cameroon_artefact.write_bytes(CAMEROON_ARTEFACT) manifest_path = repo / "docs/evidence/core_demo_manifest.json" _write(manifest_path, _manifest()) _git(repo, "add", ".") diff --git a/scripts/tests/test_sync_truth.py b/scripts/tests/test_sync_truth.py index b893916..bce151c 100644 --- a/scripts/tests/test_sync_truth.py +++ b/scripts/tests/test_sync_truth.py @@ -9,6 +9,8 @@ import pytest from scripts.sync_truth import ( + CAMEROON_PUBLISHED_FIELDS, + CAMEROON_RESULT_PATH, DEMOL_PUBLISHED_FIELDS, DEMOL_RESULT_PATH, FIGURE_PROSE_DOCS, @@ -16,6 +18,7 @@ load_manifest, missing_evidence_paths, render_capability_matrix, + render_truth_block, render_typescript, replace_truth_block, stale_figures_in_text, @@ -29,6 +32,15 @@ for index, field in enumerate(DEMOL_PUBLISHED_FIELDS, start=1) } | {"trees": 65} +#: The same idea for the tropical cohort. CAMEROON_PUBLISHED_FIELDS maps the +#: manifest's name for each figure to the artefact's, because one of them +#: differs -- the manifest calls the cohort size `trees` and the artefact calls +#: it `cohort_size` -- so the fixture is keyed by the artefact's names. +CAMEROON_METRICS: dict[str, object] = { + metrics_key: round(index * 0.4, 6) + for index, metrics_key in enumerate(CAMEROON_PUBLISHED_FIELDS.values(), start=1) +} + CURRENT_CLAIM_DOCS = ( Path("README.md"), Path("AGENTS.md"), @@ -127,6 +139,31 @@ def _unsupported_wan_positive_claims(prose: str) -> tuple[str, ...]: ) +def _write_artefact(root: Path, relative: str, metrics: dict[str, object]) -> str: + """Write a derivation artefact under `root` and return its SHA-256. + + Both cohorts publish through the same shape -- a `metrics` object beside a + pinned hash -- so both test classes build their fixtures with this. + """ + artefact = root / relative + artefact.parent.mkdir(parents=True, exist_ok=True) + payload = json.dumps({"metrics": metrics}).encode("utf-8") + artefact.write_bytes(payload) + return hashlib.sha256(payload).hexdigest() + + +def _cameroon_block() -> dict[str, object]: + """The manifest block, named the manifest's way, valued the artefact's.""" + return { + "result_path": CAMEROON_RESULT_PATH, + "result_sha256": "5" * 64, + **{ + manifest_key: CAMEROON_METRICS[metrics_key] + for manifest_key, metrics_key in CAMEROON_PUBLISHED_FIELDS.items() + }, + } + + def _manifest(tmp_path: Path) -> Path: path = tmp_path / "manifest.json" path.write_text( @@ -158,6 +195,7 @@ def _manifest(tmp_path: Path) -> Path: "result_sha256": "4" * 64, **DEMOL_METRICS, }, + "cameroon_61": _cameroon_block(), }, "capabilities": [ { @@ -202,15 +240,20 @@ class TestThePublishedDemolFiguresHaveASource: @staticmethod def _with_artefact(tmp_path: Path, metrics: dict[str, object] | None = None) -> Path: - """A manifest beside a derivation artefact it correctly cites.""" + """A manifest beside a derivation artefact it correctly cites. + + The tropical artefact is written too, because `cameroon_61` is required + the same way this block is; without it every case below would fail on + the wrong cohort. `validate_demol` runs first, so the failures these + tests assert on are still Demol's. + """ path = _manifest(tmp_path) - artefact = tmp_path / DEMOL_RESULT_PATH - artefact.parent.mkdir(parents=True, exist_ok=True) - payload = json.dumps({"metrics": metrics or DEMOL_METRICS}).encode("utf-8") - artefact.write_bytes(payload) + demol_sha = _write_artefact(tmp_path, DEMOL_RESULT_PATH, metrics or DEMOL_METRICS) + cameroon_sha = _write_artefact(tmp_path, CAMEROON_RESULT_PATH, CAMEROON_METRICS) data = json.loads(path.read_text(encoding="utf-8")) - data["validation"]["demol_65"]["result_sha256"] = hashlib.sha256(payload).hexdigest() + data["validation"]["demol_65"]["result_sha256"] = demol_sha + data["validation"]["cameroon_61"]["result_sha256"] = cameroon_sha path.write_text(json.dumps(data), encoding="utf-8") return path @@ -284,6 +327,139 @@ def test_every_published_field_is_compared_not_just_the_quoted_ones( load_manifest(path, repo_root=root) +class TestThePublishedCameroonFiguresHaveASource: + """The tropical block had no validator at all. + + `validate_demol` re-hashes its artefact and compares every published field + against it. `cameroon_61` had neither: `load_manifest` did not require the + block, nothing re-hashed `docs/evidence/cameroon_61/result.json`, and no + test held the two to each other. The only thing reading the block was + `published_figure_values`, which uses it to spot stale prose -- so a + manifest figure that had drifted from its own artefact would have been used + to certify documents quoting the drifted number. + + That left the newest and strongest evidence in the repository -- 61 trees + cut down and weighed, the only cohort that checks the allometric stage -- + guarded less than the 65-tree cohort it supersedes for tropical claims. + """ + + @staticmethod + def _with_artefacts( + tmp_path: Path, metrics: dict[str, object] | None = None + ) -> Path: + """A manifest beside both derivation artefacts, correctly cited.""" + path = _manifest(tmp_path) + demol_sha = _write_artefact(tmp_path, DEMOL_RESULT_PATH, DEMOL_METRICS) + cameroon_sha = _write_artefact( + tmp_path, CAMEROON_RESULT_PATH, metrics or CAMEROON_METRICS + ) + + data = json.loads(path.read_text(encoding="utf-8")) + data["validation"]["demol_65"]["result_sha256"] = demol_sha + data["validation"]["cameroon_61"]["result_sha256"] = cameroon_sha + path.write_text(json.dumps(data), encoding="utf-8") + return path + + def test_a_manifest_matching_its_artefact_loads(self, tmp_path: Path): + path = self._with_artefacts(tmp_path) + + assert load_manifest(path, repo_root=tmp_path)["validation"]["cameroon_61"] + + def test_a_manifest_with_no_tropical_block_is_refused(self, tmp_path: Path): + """The block was optional. A cohort that can be dropped from the + manifest without a gate noticing is a cohort the gate does not hold.""" + path = self._with_artefacts(tmp_path) + data = json.loads(path.read_text(encoding="utf-8")) + del data["validation"]["cameroon_61"] + path.write_text(json.dumps(data), encoding="utf-8") + + with pytest.raises(ValueError, match="validation missing required keys"): + load_manifest(path, repo_root=tmp_path) + + def test_a_figure_with_no_artefact_behind_it_is_refused(self, tmp_path: Path): + path = _manifest(tmp_path) + _write_artefact(tmp_path, DEMOL_RESULT_PATH, DEMOL_METRICS) + data = json.loads(path.read_text(encoding="utf-8")) + data["validation"]["demol_65"]["result_sha256"] = hashlib.sha256( + json.dumps({"metrics": DEMOL_METRICS}).encode("utf-8") + ).hexdigest() + path.write_text(json.dumps(data), encoding="utf-8") + + with pytest.raises(ValueError, match="have no source"): + load_manifest(path, repo_root=tmp_path) + + def test_a_block_that_cites_nothing_is_refused(self, tmp_path: Path): + path = self._with_artefacts(tmp_path) + data = json.loads(path.read_text(encoding="utf-8")) + del data["validation"]["cameroon_61"]["result_path"] + path.write_text(json.dumps(data), encoding="utf-8") + + with pytest.raises(ValueError, match="missing required keys"): + load_manifest(path, repo_root=tmp_path) + + def test_a_block_citing_some_other_file_is_refused(self, tmp_path: Path): + path = self._with_artefacts(tmp_path) + data = json.loads(path.read_text(encoding="utf-8")) + data["validation"]["cameroon_61"]["result_path"] = "docs/evidence/elsewhere.json" + path.write_text(json.dumps(data), encoding="utf-8") + + with pytest.raises(ValueError, match="result_path must be"): + load_manifest(path, repo_root=tmp_path) + + def test_a_hand_edited_figure_is_refused(self, tmp_path: Path): + path = self._with_artefacts(tmp_path) + data = json.loads(path.read_text(encoding="utf-8")) + data["validation"]["cameroon_61"]["dbh_gate_applied_mae_cm"] = 0.1 + path.write_text(json.dumps(data), encoding="utf-8") + + with pytest.raises( + ValueError, + match=r"disagrees with the derived result.*dbh_gate_applied_mae_cm", + ): + load_manifest(path, repo_root=tmp_path) + + def test_a_rewritten_artefact_is_refused(self, tmp_path: Path): + path = self._with_artefacts(tmp_path) + artefact = tmp_path / CAMEROON_RESULT_PATH + artefact.write_bytes( + json.dumps( + {"metrics": {**CAMEROON_METRICS, "dbh_gate_applied_mae_cm": 0.1}} + ).encode("utf-8") + ) + + with pytest.raises(ValueError, match="has changed since it was reviewed"): + load_manifest(path, repo_root=tmp_path) + + def test_every_published_field_is_compared_not_just_the_quoted_ones( + self, tmp_path: Path + ): + """Six of the twenty-five reach `published_figure_values`. The rest -- + the gate counts, the two allometric routes, the measurement share -- + are quoted in CAMEROON_EVIDENCE_CHAIN.md and on the site, and a block is + only as sourced as its least-checked number.""" + for manifest_key in CAMEROON_PUBLISHED_FIELDS: + root = tmp_path / manifest_key + root.mkdir() + path = self._with_artefacts(root) + data = json.loads(path.read_text(encoding="utf-8")) + data["validation"]["cameroon_61"][manifest_key] = "tampered" + path.write_text(json.dumps(data), encoding="utf-8") + + with pytest.raises( + ValueError, match=rf"disagrees with the derived result.*{manifest_key}" + ): + load_manifest(path, repo_root=root) + + def test_the_checked_in_manifest_agrees_with_its_committed_artefact(self): + """Not a fixture: the real manifest against the real 61-tree result.""" + manifest = load_manifest( + Path("docs/evidence/core_demo_manifest.json"), repo_root=Path.cwd() + ) + + assert manifest["validation"]["cameroon_61"]["gate_passed_trees"] == 27 + assert manifest["validation"]["cameroon_61"]["gate_refused_trees"] == 33 + + def test_manifest_rejects_promoted_pointnet_without_gate(tmp_path: Path): path = _manifest(tmp_path) data = json.loads(path.read_text(encoding="utf-8")) @@ -345,6 +521,93 @@ def test_generated_outputs_contain_exact_truth(tmp_path: Path): assert "Stub" in matrix +class TestTheTropicalCohortReachesTheGeneratedSurfaces: + """The site published the temperate figure as the accuracy of the product. + + `render_typescript` emitted `wanHeldOut`, `demol65` and + `pointnetIndependent` and stopped. `render_truth_block` emitted the first + two. So `apps/web/src/generated/core-demo-evidence.ts` had no tropical + figure to show, the landing page's accuracy panel read `0.90 cm` from the + 65 Belgian trees, and the truth block in docs/PROJECT_SPEC.md said nothing + about the cohort that was cut down and weighed. + + The number a visitor saw was true of temperate isolated trees. The product + is aimed at tropical forest, the tropical cohort had been measured, and the + gate that exists to stop documents drifting from evidence could not help, + because the evidence never reached the document. + """ + + def test_the_typescript_the_site_reads_carries_the_tropical_cohort( + self, tmp_path: Path + ): + data = load_manifest(_manifest(tmp_path)) + + typescript = render_typescript(data) + + assert "cameroon61" in typescript + assert ( + f"dbhGateAppliedMaeCm: {CAMEROON_METRICS['dbh_gate_applied_mae_cm']}" + in typescript + ) + + def test_the_typescript_carries_how_many_trees_the_gate_refused( + self, tmp_path: Path + ): + """The headline is an average over the trees that passed. Without the + refused count beside it, 27 of 60 looks like 60 of 60.""" + data = load_manifest(_manifest(tmp_path)) + + typescript = render_typescript(data) + + assert f"gatePassedTrees: {CAMEROON_METRICS['gate_passed_trees']}" in typescript + assert ( + f"gateRefusedTrees: {CAMEROON_METRICS['gate_refused_trees']}" in typescript + ) + + def test_the_typescript_carries_the_ungated_mean_error(self, tmp_path: Path): + """The all-measurable mean error is not a bound on individual error.""" + data = load_manifest(_manifest(tmp_path)) + + typescript = render_typescript(data) + + assert f"dbhMaeCm: {CAMEROON_METRICS['dbh_mae_cm']}" in typescript + + def test_the_truth_block_states_the_tropical_result(self, tmp_path: Path): + data = load_manifest(_manifest(tmp_path)) + + block = render_truth_block(data) + + assert "Cameroon" in block + assert str(CAMEROON_METRICS["dbh_gate_applied_mae_cm"]) in block + + def test_ungated_mean_error_is_not_presented_as_an_upper_bound(self, tmp_path: Path): + block = render_truth_block(load_manifest(_manifest(tmp_path))) + + assert "ungated DBH MAE" in block + assert "not an upper bound" in block + assert "ceiling and not the error" not in block + + def test_paired_error_difference_is_not_a_causal_measurement_share(self, tmp_path: Path): + block = render_truth_block(load_manifest(_manifest(tmp_path))) + + assert "median paired APE difference" in block + assert "percentage points" in block + assert "not a causal error decomposition" in block + assert "measurement contributing" not in block + + def test_the_truth_block_states_what_the_tropical_cohort_does_not_validate( + self, tmp_path: Path + ): + """These clouds arrive leaf-stripped and are single trees, so stages + 1-4 and stage 5 are untouched by them. A truth block that gave the + figure without that would be the drift it exists to prevent.""" + data = load_manifest(_manifest(tmp_path)) + + block = render_truth_block(data) + + assert "leaf-stripped" in block + + def test_truth_block_requires_exactly_one_marker_pair(): source = ( "before\n" @@ -622,6 +885,46 @@ def test_nothing_still_refers_to_the_deleted_mobile_app(): assert not path.exists(), f"{path} is back; the guard it needs is not" +#: Surfaces a visitor or a new reader meets first, where the project has to say +#: what it currently is. +#: +#: NSC 2026 was the competition this repository was built for. It was not won, +#: and CLAUDE.md records that the framing is retired: the goal is now correct +#: science that can be published and used. The badge, the hero eyebrow, the +#: page metadata and the footer went on saying otherwise for weeks after, +#: because nothing checked. Historical documents and the proposal keep the +#: reference -- rewriting those would be falsifying a record -- so this is +#: scoped to the surfaces that speak in the present tense. +CURRENT_FRAMING_SURFACES = ( + Path("README.md"), + Path("AGENTS.md"), + Path("apps/web/src/app/page.tsx"), + Path("apps/web/src/app/layout.tsx"), + Path("apps/web/src/app/demo/page.tsx"), +) + + +@pytest.mark.parametrize("path", CURRENT_FRAMING_SURFACES) +def test_no_current_surface_still_claims_the_competition(path: Path): + text = path.read_text(encoding="utf-8") + + assert "NSC 2026" not in text, ( + f"{path} still presents the project as an NSC 2026 entry. The " + "competition was not won and the framing is retired; see CLAUDE.md." + ) + + +def test_the_retired_framing_is_still_recorded_where_it_belongs(): + """The opposite failure: erasing it everywhere. + + docs/DOCUMENT_STATUS.md classifies the proposal and the historical + documents as records of decisions taken. Deleting the competition from + those would be rewriting what happened, which this repository treats as a + worse fault than an out-of-date badge. + """ + assert "NSC" in Path("docs/DOCUMENT_STATUS.md").read_text(encoding="utf-8") + + CORE_DEMO_PROSE_DOCS = (Path("README.md"), Path("docs/PROJECT_SPEC.md")) #: "1036.09 kg C" / "3798.99 kg CO₂e", written into a sentence by hand rather diff --git a/services/api/Dockerfile b/services/api/Dockerfile index 3d169d3..330f024 100644 --- a/services/api/Dockerfile +++ b/services/api/Dockerfile @@ -94,11 +94,14 @@ ENV PYTHONDONTWRITEBYTECODE=1 \ TREEQ_GIT_DIRTY=${GIT_DIRTY} \ ML_DIR=/app/services/ml -# libgl1 and libgomp1 are open3d's; curl is what the HEALTHCHECK calls. +# Open3D needs EGL, OpenGL, OpenMP and USB libraries even for CPU PLY import. +# curl is for HEALTHCHECK. # libpq5, libgeos-c1v5 and libproj25 went with the database layer. RUN apt-get update && apt-get install -y --no-install-recommends \ + libegl1 \ libgl1 \ libgomp1 \ + libusb-1.0-0 \ curl \ && rm -rf /var/lib/apt/lists/* @@ -111,6 +114,10 @@ WORKDIR /app COPY --from=builder /usr/local/lib/python3.11/site-packages /usr/local/lib/python3.11/site-packages COPY --from=builder /usr/local/bin /usr/local/bin +# Readiness imports the orchestrator but Open3D is loaded only on a PLY request. +# Fail the image build here if any of its native dependencies are missing. +RUN python -c "import open3d; print(open3d.__version__)" + # API COPY --chown=app:app services/api/app/ ./app/ diff --git a/services/ml/tests/test_cameroon_evidence_is_current.py b/services/ml/tests/test_cameroon_evidence_is_current.py new file mode 100644 index 0000000..a5c0b4c --- /dev/null +++ b/services/ml/tests/test_cameroon_evidence_is_current.py @@ -0,0 +1,139 @@ +"""The published tropical figures still reproduce from the cohort. + +`docs/evidence/cameroon_61/result.json` is the derived source for every +tropical number this project publishes: the diameter error on the trees the +gate passes, the count it refuses, and the two allometric routes scored +against mass that was actually weighed. Those numbers reach the manifest, the +truth block in docs/PROJECT_SPEC.md and the landing page's accuracy panel. + +`scripts/sync_truth.py` holds the manifest to this artefact byte for byte, and +that check runs on CI. What it cannot check is the step before: whether the +artefact still reproduces from the 61 trees. That needs the 1.29 GB archive, +so it lives here and skips where the archive is absent -- the same shape, and +the same limitation, as test_published_evidence_is_current.py for Demol. + +See docs/ml/WHAT_CI_DOES_NOT_CHECK.md. +""" + +from __future__ import annotations + +import json +import sys +from pathlib import Path +from typing import Any + +import pytest + +ML_ROOT = Path(__file__).resolve().parent.parent +ARCHIVE = ML_ROOT / "data" / "raw" / "dryad_cameroon" / "Trees" +ARTEFACT = ML_ROOT.parent.parent / "docs" / "evidence" / "cameroon_61" / "result.json" + +needs_archive = pytest.mark.skipif( + not (ARCHIVE / "database.xls").is_file(), + reason="Cameroon archive not present: see docs/ml/CAMEROON_EVIDENCE_CHAIN.md", +) + +#: How far a fresh run may sit from the published artefact before it is stale. +#: +#: Not a precision claim, and read the same way as the Demol file's: the +#: protocol fixes the 61-tree keying, the 20,000-point cap and the seed, so an +#: unchanged pipeline reproduces the artefact exactly. The tolerance exists so +#: a refactor or a platform float difference does not turn the build red. +#: +#: The gated figure gets the tightest band of the three because it is the one a +#: user is handed. It is an average over 27 trees, so a single tree moving in or +#: out of the gate shifts it by more than a refactor ever should -- which is +#: precisely the change this file exists to catch. +TOLERANCES = { + "dbh_gate_applied_mae_cm": 0.05, + "dbh_mae_cm": 0.20, + "height_mae_m": 0.10, + "volume_mape_pct": 2.00, + "chave_route_b_ape_pct_median": 0.50, + "tver_route_b_ape_pct_median": 0.50, +} + +#: Counts, not measurements. A tolerance would be meaningless: the gate either +#: passes the same trees or it does not, and if it does not, every figure above +#: is an average over a different population than the published one. +EXACT_FIELDS = ( + "cohort_size", + "trees_measured", + "trees_excluded", + "gate_passed_trees", + "gate_refused_trees", + "dbh_mae_cm_small_stems_n", +) + + +@pytest.fixture(scope="module") +def published() -> dict[str, Any]: + return json.loads(ARTEFACT.read_text(encoding="utf-8"))["metrics"] + + +@pytest.fixture(scope="module") +def measured() -> dict[str, Any]: + """A fresh run of the derivation the published artefact came from. + + Imports the script rather than reimplementing it, for the reason the Demol + file gives: a test that recomputed these statistics its own way would pass + while the script that writes the published file was broken. + """ + sys.path.insert(0, str(ML_ROOT / "scripts")) + try: + from derive_cameroon_evidence import derive + finally: + sys.path.pop(0) + + return derive(archive_root=ARCHIVE)["metrics"] + + +@needs_archive +@pytest.mark.parametrize("field", sorted(TOLERANCES)) +def test_the_published_figures_have_not_drifted(field, published, measured): + drift = measured[field] - published[field] + + assert abs(drift) <= TOLERANCES[field], ( + f"{field}: published {published[field]}, measured {measured[field]} " + f"({drift:+.4f}). Re-run scripts/derive_cameroon_evidence.py, repin the " + "manifest from it, and re-run sync_truth.py --write." + ) + + +@needs_archive +@pytest.mark.parametrize("field", EXACT_FIELDS) +def test_the_cohort_and_the_gate_still_partition_it_the_same_way( + field, published, measured +): + assert measured[field] == published[field], ( + f"{field}: published {published[field]}, measured {measured[field]}. " + "The gate is passing a different set of trees, so every averaged " + "figure in this artefact is an average over a different population." + ) + + +@needs_archive +def test_every_published_field_still_reproduces(published, measured): + """The six in TOLERANCES are the quoted ones. An artefact is only as + current as its least-checked number, so the rest are compared too.""" + stale = sorted( + field + for field, value in published.items() + if isinstance(value, (int, float)) + and field not in TOLERANCES + and field not in EXACT_FIELDS + and field in measured + and abs(measured[field] - value) > 0.01 + ) + + assert not stale, ( + f"{stale} no longer reproduce. Run " + "scripts/derive_cameroon_evidence.py --check for the full comparison." + ) + + +def test_the_artefact_the_manifest_cites_is_present(): + """Runs everywhere, archive or not. sync_truth.py's validate_cameroon + re-hashes this file on every --check; if it is gone, that check has + nothing to compare the manifest against.""" + assert ARTEFACT.is_file(), f"{ARTEFACT} is missing"