From 44ff90e767b2589223bcd002e611bd70f168936c Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?=E6=9B=BE=E7=AB=8B=E5=AE=8F?= <曾立宏@buaa.edu.cn> Date: Thu, 20 Aug 2026 14:02:53 +0800 Subject: [PATCH 1/4] feat(standards): Phase 1 lossless canonical standard layer + alignment audit MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Build the ORIGINAL classification/grading standards (finance guide Excel, shougang 关基 catalog Excel) into a lossless, auditable canonical form that restores real hierarchy depth and grading columns the legacy standards_map digests dropped. Adds Phase 0 semantics doc + Phase 1 design/migration note. - src/agent/standards: contracts (StandardCategory/CanonicalStandard, standard_data_level kept separate from sample data_level), raw xlsx readers, deterministic builders, read-only sample<->standard alignment - script/standard/cli: regenerable data/standards/*.standard.json + artifacts/generated/provenance/* (finance/shougang/infra alignments, summary) - alignment facts: finance 529 matched / 2 outliers (AMONEY, HXTRADENO) / 37 unresolved; shougang 18,393/18,393 (100%); infra 64/64; guanji_dict lost B3-6 中厚板作业计划 (documented, not fixed) - tests/standards: 26 hermetic + real-xlsx integration tests (raw missing => skip) - no training behavior change: no changes to prompts/parser/reward/SFT/RL, samples, registry/corpus, or data_level-as-target --- .../finance_standard_alignment.json | 268 ++++++++++++++ .../provenance/infra_standard_alignment.json | 252 +++++++++++++ .../shougang_standard_alignment.json | 66 ++++ .../provenance/standard_build_summary.json | 155 ++++++++ docs/design/data_level_design.md | 168 +++++++++ docs/design/phase1_canonical_standard.md | 135 +++++++ script/standard/__init__.py | 0 script/standard/cli.py | 295 +++++++++++++++ src/agent/standards/__init__.py | 58 +++ src/agent/standards/align.py | 157 ++++++++ src/agent/standards/build.py | 348 ++++++++++++++++++ src/agent/standards/contracts.py | 255 +++++++++++++ src/agent/standards/sources.py | 197 ++++++++++ tests/standards/test_align.py | 88 +++++ tests/standards/test_build_finance.py | 92 +++++ tests/standards/test_build_shougang.py | 83 +++++ tests/standards/test_contracts.py | 99 +++++ tests/standards/test_real_xlsx.py | 131 +++++++ 18 files changed, 2847 insertions(+) create mode 100644 artifacts/generated/provenance/finance_standard_alignment.json create mode 100644 artifacts/generated/provenance/infra_standard_alignment.json create mode 100644 artifacts/generated/provenance/shougang_standard_alignment.json create mode 100644 artifacts/generated/provenance/standard_build_summary.json create mode 100644 docs/design/data_level_design.md create mode 100644 docs/design/phase1_canonical_standard.md create mode 100644 script/standard/__init__.py create mode 100644 script/standard/cli.py create mode 100644 src/agent/standards/__init__.py create mode 100644 src/agent/standards/align.py create mode 100644 src/agent/standards/build.py create mode 100644 src/agent/standards/contracts.py create mode 100644 src/agent/standards/sources.py create mode 100644 tests/standards/test_align.py create mode 100644 tests/standards/test_build_finance.py create mode 100644 tests/standards/test_build_shougang.py create mode 100644 tests/standards/test_contracts.py create mode 100644 tests/standards/test_real_xlsx.py diff --git a/artifacts/generated/provenance/finance_standard_alignment.json b/artifacts/generated/provenance/finance_standard_alignment.json new file mode 100644 index 0000000..5c9217b --- /dev/null +++ b/artifacts/generated/provenance/finance_standard_alignment.json @@ -0,0 +1,268 @@ +{ + "mismatched_samples": [ + { + "category_id": "finance:业务.交易信息.交易基本信息", + "field": "AMONEY", + "name": "交易基本信息", + "sample_level": "L3", + "sample_path": [ + "业务", + "交易信息", + "交易通用信息", + "交易基本信息" + ], + "source_row": 133, + "standard_level": "L2", + "standard_raw_level": "2" + }, + { + "category_id": "finance:业务.账户信息.基本信息", + "field": "HXTRADENO", + "name": "基本信息", + "sample_level": "L3", + "sample_path": [ + "业务", + "账户信息", + "基本信息" + ], + "source_row": 51, + "standard_level": "L2", + "standard_raw_level": "2" + } + ], + "resolved_match_rate": 0.9962, + "sample_counts": { + "matched": 529, + "mismatched": 2, + "resolved": 531, + "standard_level_unavailable": 0, + "standard_missing": 0, + "total": 568, + "unresolved": 37 + }, + "sample_missing_standard_categories": [ + "finance:业务.交易信息.保险收费信息", + "finance:业务.交易信息.保险金信托给付信息", + "finance:业务.合约协议.交易类中间业务信息", + "finance:业务.合约协议.代理类中间业务信息", + "finance:业务.合约协议.保单基本信息", + "finance:业务.合约协议.保单责任信息", + "finance:业务.合约协议.信托产品信息", + "finance:业务.合约协议.信托募集信息", + "finance:业务.合约协议.信托运用重要信息", + "finance:业务.合约协议.债权转让合同信息", + "finance:业务.合约协议.催收信息", + "finance:业务.合约协议.其他类中间业务信息", + "finance:业务.合约协议.咨询顾问类中间业务信息", + "finance:业务.合约协议.固有业务信息", + "finance:业务.合约协议.垫款信息", + "finance:业务.合约协议.展期信息", + "finance:业务.合约协议.托管业务信息", + "finance:业务.合约协议.投资理财信息", + "finance:业务.合约协议.投资银行业务信息", + "finance:业务.合约协议.担保信息", + "finance:业务.合约协议.担保承诺类中间业务信息", + "finance:业务.合约协议.授信信息", + "finance:业务.合约协议.放还款信息", + "finance:业务.合约协议.核保信息", + "finance:业务.合约协议.签约信息", + "finance:业务.合约协议.自营资金投资信息", + "finance:业务.合约协议.贷款业务重要信息", + "finance:业务.合约协议.资产信息", + "finance:业务.合约协议.赔付结果信息", + "finance:业务.合约协议.违约信息", + "finance:业务.合约协议.逾期信息", + "finance:业务.合约协议.销售渠道费用信息", + "finance:业务.账户信息.介质信息", + "finance:业务.账户信息.冻结信息", + "finance:业务.账户信息.特有账户信息", + "finance:业务.金融监管和服务.LEI数据发布信息", + "finance:业务.金融监管和服务.LEI注册信息", + "finance:业务.金融监管和服务.个人征信机构管理信息", + "finance:业务.金融监管和服务.事件公告信息", + "finance:业务.金融监管和服务.事项审批信息", + "finance:业务.金融监管和服务.事项申请信息", + "finance:业务.金融监管和服务.企业征信机构管理信息", + "finance:业务.金融监管和服务.信息沟通信息", + "finance:业务.金融监管和服务.信用体系建设管理信息", + "finance:业务.金融监管和服务.信用评级信息", + "finance:业务.金融监管和服务.分类考核评级信息", + "finance:业务.金融监管和服务.咨询投诉管理信息", + "finance:业务.金融监管和服务.国际合作信息", + "finance:业务.金融监管和服务.失效居民身份证核查信息", + "finance:业务.金融监管和服务.宣传培训信息", + "finance:业务.金融监管和服务.居民身份证核查信息", + "finance:业务.金融监管和服务.异议反馈信息", + "finance:业务.金融监管和服务.征信维权信息", + "finance:业务.金融监管和服务.成果评审信息", + "finance:业务.金融监管和服务.房地产市场分析信息", + "finance:业务.金融监管和服务.房地产数据处理信息", + "finance:业务.金融监管和服务.房地产调查信息", + "finance:业务.金融监管和服务.报表处理信息", + "finance:业务.金融监管和服务.指标分析信息", + "finance:业务.金融监管和服务.指标管理信息", + "finance:业务.金融监管和服务.机构业务申请基本信息", + "finance:业务.金融监管和服务.机构业务许可证书管理信息", + "finance:业务.金融监管和服务.机构年检信息", + "finance:业务.金融监管和服务.案例管理信息", + "finance:业务.金融监管和服务.汇率信息", + "finance:业务.金融监管和服务.消费者教育信息", + "finance:业务.金融监管和服务.热点汇总信息", + "finance:业务.金融监管和服务.电子公告管理信息", + "finance:业务.金融监管和服务.监督检查信息", + "finance:业务.金融监管和服务.舆情采集信息", + "finance:业务.金融监管和服务.行政监管信息", + "finance:业务.金融监管和服务.调查协查信息", + "finance:业务.金融监管和服务.金融信用数据库管理信息", + "finance:业务.金融监管和服务.问卷调查信息", + "finance:业务.金融监管和服务.非居民身份信息核查信息", + "finance:客户.个人.个人信贷信息", + "finance:客户.个人.个人健康生理信息", + "finance:客户.个人.个人党政信息", + "finance:客户.个人.个人司法信息", + "finance:客户.个人.个人地理位置信息", + "finance:客户.个人.个人就学信息", + "finance:客户.个人.个人职业信息", + "finance:客户.个人.个人财产信息", + "finance:客户.个人.个人资质证书信息", + "finance:客户.个人.个人间关系信息", + "finance:客户.个人.交易类标签信息", + "finance:客户.个人.价值标签信息", + "finance:客户.个人.公私间关系信息", + "finance:客户.个人.关系标签信息", + "finance:客户.个人.基础标签信息", + "finance:客户.个人.弱隐私生物特征信息", + "finance:客户.个人.强隐私生物特征信息", + "finance:客户.个人.签约标签信息", + "finance:客户.个人.营销服务标签信息", + "finance:客户.个人.行为信息", + "finance:客户.个人.行为标签信息", + "finance:客户.个人.风险标签信息", + "finance:客户.单位.交易类标签信息", + "finance:客户.单位.价值标签信息", + "finance:客户.单位.企业信贷信息", + "finance:客户.单位.企业司法信息", + "finance:客户.单位.企业工商信息", + "finance:客户.单位.企业税务信息", + "finance:客户.单位.传统鉴别信息", + "finance:客户.单位.公私间关系信息", + "finance:客户.单位.单位联系信息", + "finance:客户.单位.单位财务信息", + "finance:客户.单位.单位间关系信息", + "finance:客户.单位.基础标签信息", + "finance:客户.单位.签约标签信息", + "finance:客户.单位.股东信息", + "finance:客户.单位.股东重要信息", + "finance:客户.单位.营销标签信息", + "finance:客户.单位.行为信息", + "finance:客户.单位.行为标签信息", + "finance:客户.单位.风险标签信息", + "finance:监管.数据报送.监管指标上报信息", + "finance:监管.数据报送.监管明细数据上报信息", + "finance:监管.数据报送.金融统计信息", + "finance:监管.数据收取.审计信息", + "finance:监管.数据收取.统计分析信息", + "finance:监管.数据收取.评价、处罚与违规信息", + "finance:监管.数据收取.预警信息", + "finance:经营管理.技术管理.信息资产管理信息", + "finance:经营管理.技术管理.办公软件资源", + "finance:经营管理.技术管理.安全管理信息", + "finance:经营管理.技术管理.开发信息", + "finance:经营管理.技术管理.测试信息", + "finance:经营管理.技术管理.系统运维信息", + "finance:经营管理.技术管理.规划信息", + "finance:经营管理.技术管理.质量管理信息", + "finance:经营管理.综合管理.一般员工信息(公开)", + "finance:经营管理.综合管理.业务发展规划信息", + "finance:经营管理.综合管理.业绩信息", + "finance:经营管理.综合管理.人力需求规划信息", + "finance:经营管理.综合管理.人员招聘报名信息", + "finance:经营管理.综合管理.人员招聘考试信息", + "finance:经营管理.综合管理.企事业财务管理信息", + "finance:经营管理.综合管理.信息披露信息", + "finance:经营管理.综合管理.党务纪检信息", + "finance:经营管理.综合管理.公文信息", + "finance:经营管理.综合管理.内部审计信息", + "finance:经营管理.综合管理.内部资金往来信息", + "finance:经营管理.综合管理.分类信息", + "finance:经营管理.综合管理.利率管理信息", + "finance:经营管理.综合管理.合规信息", + "finance:经营管理.综合管理.员工信息(非公开)", + "finance:经营管理.综合管理.品牌战略规划信息", + "finance:经营管理.综合管理.国库管理信息", + "finance:经营管理.综合管理.培训与资质信息", + "finance:经营管理.综合管理.基建财务管理信息", + "finance:经营管理.综合管理.基本信息(公开)", + "finance:经营管理.综合管理.基本信息(非公开)", + "finance:经营管理.综合管理.层级信息", + "finance:经营管理.综合管理.岗位角色信息", + "finance:经营管理.综合管理.工会信息", + "finance:经营管理.综合管理.市场营销规划信息", + "finance:经营管理.综合管理.技能信息", + "finance:经营管理.综合管理.支付清算汇总信息", + "finance:经营管理.综合管理.支付清算鉴别信息", + "finance:经营管理.综合管理.档案管理信息", + "finance:经营管理.综合管理.法务信息", + "finance:经营管理.综合管理.税务信息", + "finance:经营管理.综合管理.章程制度信息", + "finance:经营管理.综合管理.管理会计信息", + "finance:经营管理.综合管理.薪资信息", + "finance:经营管理.综合管理.证件信息", + "finance:经营管理.综合管理.财务会计信息", + "finance:经营管理.综合管理.财务支出信息", + "finance:经营管理.综合管理.资产负债管理信息", + "finance:经营管理.综合管理.资金渠道流通汇总信息", + "finance:经营管理.综合管理.资金规划信息", + "finance:经营管理.综合管理.邮件信息", + "finance:经营管理.营销服务.产品重要信息", + "finance:经营管理.营销服务.分类信息", + "finance:经营管理.营销服务.基本信息", + "finance:经营管理.营销服务.市场营销信息(公开)", + "finance:经营管理.营销服务.市场营销信息(非公开)", + "finance:经营管理.营销服务.新产品(项目)研发信息", + "finance:经营管理.营销服务.服务管理信息", + "finance:经营管理.营销服务.渠道管理信息", + "finance:经营管理.营销服务.特征信息", + "finance:经营管理.营销服务.第三方代理渠道信息(公开)", + "finance:经营管理.营销服务.管理信息", + "finance:经营管理.营销服务.线上自有渠道信息(公开)", + "finance:经营管理.营销服务.线下自有渠道信息(公开)", + "finance:经营管理.营销服务.营销管理信息", + "finance:经营管理.运营管理.单证入库信息", + "finance:经营管理.运营管理.单证发放信息", + "finance:经营管理.运营管理.单证日常管理信息", + "finance:经营管理.运营管理.单证核销信息", + "finance:经营管理.运营管理.单证设计信息", + "finance:经营管理.运营管理.参数/指标运维信息", + "finance:经营管理.运营管理.合作内容信息", + "finance:经营管理.运营管理.合作单位基本信息", + "finance:经营管理.运营管理.合作单位联系人信息", + "finance:经营管理.运营管理.客户及监管相关音影像信息", + "finance:经营管理.运营管理.技术安防信息", + "finance:经营管理.运营管理.日常管理相关音影像信息", + "finance:经营管理.运营管理.柜面服务信息", + "finance:经营管理.运营管理.物理安防信息", + "finance:经营管理.运营管理.电话服务信息", + "finance:经营管理.运营管理.网络服务信息", + "finance:经营管理.风险管理信息.风险偏好不定量指标", + "finance:经营管理.风险管理信息.风险偏好偏离信息", + "finance:经营管理.风险管理信息.风险偏好定最指标", + "finance:经营管理.风险管理信息.风险抵御水平信息", + "finance:经营管理.风险管理信息.风险控制信息", + "finance:经营管理.风险管理信息.风险监测信息", + "finance:经营管理.风险管理信息.风险缓释信息", + "finance:经营管理.风险管理信息.风险计量信息", + "finance:经营管理.风险管理信息.风险识别信息", + "finance:经营管理.风险管理信息.黑名单信息" + ], + "standard": "finance", + "standard_categories": 233, + "standard_categories_observed": 20, + "standard_categories_unobserved": 213, + "standard_level_unavailable_samples": [], + "standard_missing_samples": [], + "unresolved_by_status": { + "missing_leaf": 34, + "path_mismatch": 3 + } +} diff --git a/artifacts/generated/provenance/infra_standard_alignment.json b/artifacts/generated/provenance/infra_standard_alignment.json new file mode 100644 index 0000000..1a8edee --- /dev/null +++ b/artifacts/generated/provenance/infra_standard_alignment.json @@ -0,0 +1,252 @@ +{ + "mismatched_samples": [], + "resolved_match_rate": 1.0, + "sample_counts": { + "matched": 64, + "mismatched": 0, + "resolved": 64, + "standard_level_unavailable": 0, + "standard_missing": 0, + "total": 64, + "unresolved": 0 + }, + "sample_missing_standard_categories": [ + "A1-1-1", + "A1-1-2", + "A1-1-4", + "B1-2", + "B1-3-1", + "B1-3-2", + "B1-3-3", + "B1-4-1", + "B1-4-2", + "B1-4-3", + "B1-4-4", + "B1-5", + "B2-1-1", + "B2-1-2", + "B2-2-1", + "B2-2-2", + "B2-2-3", + "B2-2-4", + "B2-2-5", + "B2-2-6", + "B3-1-1", + "B3-1-2", + "B3-1-3", + "B3-1-4", + "B3-1-5", + "B3-2-1", + "B3-2-2", + "B3-3", + "B3-4", + "B3-5", + "B3-6", + "B3-6-1", + "B3-6-2", + "B3-7-1", + "B3-7-2", + "B4-1-1", + "B4-1-2", + "B4-1-3", + "B4-1-4", + "B4-1-5", + "B4-1-6", + "B4-1-7", + "B4-2", + "B4-3-1", + "B5-1-1", + "B5-1-2", + "B5-1-3", + "B5-1-4", + "B5-1-5", + "B5-1-6", + "B5-1-7", + "B5-1-8", + "B5-1-9", + "B5-2-1", + "B5-2-10", + "B5-2-2", + "B5-2-3", + "B5-2-4", + "B5-2-5", + "B5-2-6", + "B5-2-7", + "B5-2-8", + "B5-2-9", + "B6-1-1", + "B6-1-10", + "B6-1-11", + "B6-1-12", + "B6-1-13", + "B6-1-2", + "B6-1-3", + "B6-1-4", + "B6-1-5", + "B6-1-6", + "B6-1-7", + "B6-1-8", + "B6-1-9", + "B6-2", + "C1-1-1", + "C1-1-2", + "C1-1-3", + "C1-2-1", + "C1-2-10", + "C1-2-2", + "C1-2-3", + "C1-2-4", + "C1-2-5", + "C1-2-6", + "C1-2-7", + "C1-2-8", + "C1-2-9", + "C1-3-1", + "C1-3-2", + "C1-3-3", + "C1-3-4", + "C1-3-5", + "C1-3-6", + "C1-4-1", + "C1-4-2", + "C1-4-3", + "C1-4-4", + "C1-4-5", + "C1-4-6", + "C1-5", + "C2-1-1", + "C2-1-10", + "C2-1-11", + "C2-1-2", + "C2-1-3", + "C2-1-4", + "C2-1-5", + "C2-1-6", + "C2-1-7", + "C2-1-8", + "C2-1-9", + "C2-2-1", + "C2-2-10", + "C2-2-11", + "C2-2-12", + "C2-2-13", + "C2-2-14", + "C2-2-15", + "C2-2-16", + "C2-2-17", + "C2-2-2", + "C2-2-3", + "C2-2-4", + "C2-2-5", + "C2-2-6", + "C2-2-7", + "C2-2-8", + "C2-2-9", + "C2-3-1", + "C2-3-2", + "C2-3-3", + "C2-3-4", + "C2-3-5", + "C2-3-6", + "C2-3-7", + "C3-1-1", + "C3-1-2", + "C3-2-1", + "C3-2-2", + "C3-2-3", + "C3-2-4", + "C3-2-6", + "C3-2-7", + "C3-2-8", + "C3-3-1", + "C3-3-2", + "C3-3-3", + "C3-4-1", + "C3-4-2", + "C3-4-3", + "C3-4-4", + "C4-1", + "C5-1-1", + "C5-1-2", + "C5-1-3", + "C5-2-1", + "C5-2-2", + "C5-2-3", + "C5-2-4", + "C5-2-5", + "C6-1-1", + "C6-1-2", + "C6-1-3", + "C6-1-4", + "C6-2-1", + "C6-2-2", + "C6-2-3", + "C6-2-4", + "C6-2-5", + "C6-2-6", + "C6-2-7", + "C6-3-1", + "C7-1-1", + "C7-1-2", + "C7-1-3", + "C7-1-4", + "C7-2-1", + "C7-2-2", + "C7-2-3", + "C7-3-1", + "C7-3-2", + "C7-3-3", + "C7-3-4", + "C7-3-5", + "C7-4-1", + "C7-4-2", + "C7-4-3", + "C7-5-1", + "C7-5-2", + "C8-1-1", + "C8-1-2", + "C8-1-3", + "C8-1-4", + "C8-1-5", + "C8-1-6", + "C8-1-7", + "C8-2-1", + "C8-2-2", + "C8-3-1", + "C8-3-2", + "C8-3-3", + "C8-4-1", + "C8-5-1", + "C8-6-1", + "C8-6-2", + "C8-6-3", + "C8-6-4", + "C8-6-5", + "C8-7-1", + "C8-7-10", + "C8-7-2", + "C8-7-3", + "C8-7-4", + "C8-7-5", + "C8-7-6", + "C8-7-7", + "C8-7-8", + "C8-7-9", + "C8-8-1", + "C8-8-2", + "C8-8-3", + "C8-8-4", + "C8-8-5", + "C8-8-6", + "C8-8-7", + "C8-8-8", + "C8-9-1" + ], + "standard": "shougang", + "standard_categories": 234, + "standard_categories_observed": 4, + "standard_categories_unobserved": 230, + "standard_level_unavailable_samples": [], + "standard_missing_samples": [], + "unresolved_by_status": {} +} diff --git a/artifacts/generated/provenance/shougang_standard_alignment.json b/artifacts/generated/provenance/shougang_standard_alignment.json new file mode 100644 index 0000000..1d639fa --- /dev/null +++ b/artifacts/generated/provenance/shougang_standard_alignment.json @@ -0,0 +1,66 @@ +{ + "mismatched_samples": [], + "resolved_match_rate": 1.0, + "sample_counts": { + "matched": 18393, + "mismatched": 0, + "resolved": 18393, + "standard_level_unavailable": 0, + "standard_missing": 0, + "total": 19415, + "unresolved": 1022 + }, + "sample_missing_standard_categories": [ + "B1-2", + "B1-4-2", + "B1-4-3", + "B1-4-4", + "B1-5", + "B3-1-3", + "B3-1-4", + "B3-1-5", + "B3-3", + "B3-4", + "B3-5", + "B3-6", + "B3-7-2", + "B4-2", + "B5-1-5", + "B5-1-8", + "B5-1-9", + "B5-2-5", + "B5-2-9", + "B6-2", + "C1-2-10", + "C1-2-4", + "C1-2-9", + "C1-4-6", + "C1-5", + "C3-1-2", + "C3-2-6", + "C3-2-7", + "C3-4-2", + "C3-4-4", + "C4-1", + "C5-1-3", + "C5-2-4", + "C8-1-5", + "C8-7-1", + "C8-7-10", + "C8-7-2", + "C8-7-3", + "C8-7-4", + "C8-7-9", + "C8-8-7", + "C8-8-8" + ], + "standard": "shougang", + "standard_categories": 234, + "standard_categories_observed": 192, + "standard_categories_unobserved": 42, + "standard_level_unavailable_samples": [], + "standard_missing_samples": [], + "unresolved_by_status": { + "placeholder": 1022 + } +} diff --git a/artifacts/generated/provenance/standard_build_summary.json b/artifacts/generated/provenance/standard_build_summary.json new file mode 100644 index 0000000..7ddd026 --- /dev/null +++ b/artifacts/generated/provenance/standard_build_summary.json @@ -0,0 +1,155 @@ +{ + "legacy_information_loss": { + "finance": { + "legacy_dict_entries": 237, + "legacy_dict_path_compression": { + "categories_at_legacy_depth": 1, + "categories_with_path_deeper_than_legacy_L1_L2_leaf": 232, + "note": "legacy financial_standards_dict stored L1-L2-leaf identity strings; the real standard has 三级子类 provenance nodes that the legacy digest dropped" + }, + "legacy_unparseable_level_values": [ + "3 4级", + "l级" + ], + "standard_vs_registry_ids": { + "missing_from_registry": 0, + "registry_ids": 233, + "standard_ids": 233 + } + }, + "shougang": { + "legacy_dict_losses": { + "catalog_categories": 234, + "legacy_dict_entries": 234, + "note": "guanji_dict kept only 'name(code)' + description: real hierarchy path and the 分级 column were dropped (registry path was [] and no class field existed)", + "with_grading_restored": 234, + "with_real_path_restored": 234 + }, + "standard_vs_registry_codes": { + "note": "B3-6 中厚板作业计划 exists in the raw catalog but was dropped by guanji_dict/registry", + "registry_codes": 233, + "registry_only": [], + "standard_codes": 234, + "standard_only": [ + "B3-6" + ] + } + } + }, + "pers_info": { + "alignment_headline": null, + "note": "no confirmed classification/grading standard; the 18-category registry remains dataset-derived and is NOT presented as a canonical standard; no standard_data_level is fabricated", + "standard_source": null, + "status": "missing_or_unknown" + }, + "phase": "phase1-canonical-standard", + "standards": { + "finance": { + "alignment_headline": { + "matched": 529, + "mismatched": 2, + "resolved": 531, + "resolved_match_rate": 0.9962, + "samples_total": 568, + "standard_missing": 0 + }, + "build": { + "aggregated": { + "instances": 4, + "kinds": 1 + }, + "categories_out": 237, + "dataset": "finance", + "entries_read": 237, + "id_strategy": "path", + "issues": [ + { + "detail": "row 150 category '市场营销信息(公开)': raw level 'l' kept as-is, standard_data_level=null (not guessed)", + "kind": "level_unparseable" + }, + { + "detail": "row 168 category '客户及监管相关音影像信息': raw level '3 4' kept as-is, standard_data_level=null (not guessed)", + "kind": "level_unparseable" + } + ], + "level_distribution": { + "''": 2, + "L1": 11, + "L2": 153, + "L3": 61, + "L4": 6 + }, + "source_file": "data/raw/金融行业数据安全分类分级标准指南.xlsx", + "source_sheet": "Table 1", + "standard_name": "金融行业数据安全分类分级标准指南" + }, + "categories": 233, + "standard_source": "finance", + "status": "built" + }, + "infra": { + "alignment_headline": { + "matched": 64, + "mismatched": 0, + "resolved": 64, + "resolved_match_rate": 1.0, + "samples_total": 64, + "standard_missing": 0 + }, + "build": { + "aggregated": { + "instances": 0, + "kinds": 0 + }, + "categories_out": 234, + "dataset": "shougang", + "entries_read": 234, + "id_strategy": "code", + "issues": [], + "level_distribution": { + "L1": 16, + "L2": 170, + "L3": 48 + }, + "source_file": "data/raw/关基-数据分类分级目录.xlsx", + "source_sheet": "数据分类分级", + "standard_name": "首钢京唐数据分类分级目录(关基)" + }, + "categories": 234, + "standard_source": "shougang", + "status": "built" + }, + "shougang": { + "alignment_headline": { + "matched": 18393, + "mismatched": 0, + "resolved": 18393, + "resolved_match_rate": 1.0, + "samples_total": 19415, + "standard_missing": 0 + }, + "build": { + "aggregated": { + "instances": 0, + "kinds": 0 + }, + "categories_out": 234, + "dataset": "shougang", + "entries_read": 234, + "id_strategy": "code", + "issues": [], + "level_distribution": { + "L1": 16, + "L2": 170, + "L3": 48 + }, + "source_file": "data/raw/关基-数据分类分级目录.xlsx", + "source_sheet": "数据分类分级", + "standard_name": "首钢京唐数据分类分级目录(关基)" + }, + "categories": 234, + "standard_source": "shougang", + "status": "built" + } + } +} diff --git a/docs/design/data_level_design.md b/docs/design/data_level_design.md new file mode 100644 index 0000000..371b062 --- /dev/null +++ b/docs/design/data_level_design.md @@ -0,0 +1,168 @@ +# data_level / classification / field_sensitive 角色区分(设计说明 · Phase 0) + +Status: 只读记录(2026-08-20,数据快照 v1 后新增 raw Excel 已核对)。 +原则:只记录仓库内可验证事实,不猜测等级语义;**不改任何代码 / 数据 / 现有训练行为**。 + +**Phase 0 冻结范围**:只冻结三个字段「是什么、来自哪里、当前代码是否训练」。**不决定** `data_level` 最终训不训练——那属于 Stage2 task contract 的决策,本文档保持开放。 + +## 1. 三个概念是什么(角色模型) + +### `classification`(`level_1..level_4`) +- **角色**:sample label(源端标注),并且是**当前**的 training target(经 canonical 解析为 `target.category_id`,Stage1/Stage2 唯一监督目标)。 +- **槽位口径(重要,避免概念混淆)**:`level_1..level_4` 是统一分类槽位。当前 processed 数据把**最终/最小粒度类别**放进 `level_4` 槽位,但不同源数据的**真实分类深度并不统一**: + - pers_info:源数据是单层分类(学籍管理信息…),canonical schema 把它放进 `level_4`;`level_1/2/3` 为空,**不是说源标准真有四级体系**。 + - finance:四层填充(`level_1..level_4`);shougang/infra:四层填充。 + - 因此**不能把 `level_4` 等价成"真实标准的第四层"**;建无损 standard(Phase 1)时按各源的真实深度建模。 + +### `data_level`(`L1..L4`) +- **角色**:**sample label——样本携带的分级标签**。是否同时可被确定为 standard knowledge,**分数据集**: + - finance / shougang / infra:有证据可追溯到分类分级标准/目录(finance↔《金融行业数据安全分类分级标准指南》"最低安全级别参考";shougang/infra↔首钢京唐数据分类分级目录"分级"列,192/192 零冲突)。 + - pers_info:**目前只有 sample label,尚无已确认的 standard knowledge 来源**(源 Excel 直接填 `L1/L2/L3`,仓库内无对应标准文档)。 + - 所以不做笼统定义"`data_level` = standard knowledge 的样本级表现"——pers_info 就是反例。 +- **训练目标状态**:当前实现未进入 training target(`src/agent`、`script/verl`、`script/canonical` 对 `data_level` 零引用,parquet 不导出);**目标设计倾向将 `data_level` 作为与 classification 并列的模型预测目标**,即模型需要根据字段语义、业务上下文和分级规则独立判断安全等级,而不是仅根据 category 查表返回标准等级。最终接口仍待 Stage2 task contract 冻结。 +- **训练目标总口径**: + ```text + 当前实现: + classification → training target + data_level → provenance only + 目标任务方向: + classification → 模型预测 + data_level → 模型独立预测 + standard_data_level → 分级参考 / 审计依据,不直接等同于字段最终等级 + + 最终输出格式、prompt 可见信息和 reward + → 待 Stage2 task contract 冻结 + ``` + +### `sample_data_level` 与 `standard_data_level`(两个变体,必须分开保存) +- **`sample_data_level`** = 原始数据中具体字段的实际分级标签(即 § 上文 `data_level` 的本义),作为训练/评测 gold 候选。 +- **`standard_data_level`** = 分类分级标准给 category 的参考/最低等级(如 finance 标准列原名就叫**"最低安全级别参考"**)。 +- 二者**必须分别保存,不应在 canonicalization 时相互覆盖**:finance 的"最低安全级别参考"本身就不适合直接建模成 `category → 唯一最终 data_level`。 +- 后续 Phase 1 的 canonical standard 建议明确字段名 `standard_data_level`(值如 `"L3"`),**不要**简单命名为 `data_level`,否则日后容易再次与 sample gold 混淆。 + +### `field_sensitive`(是/否) +- **角色**:**独立 field-level 源端标注**,不属于上述两者。 +- 仅存在于 `data/raw/关基设施数据分类分级-不包含训练-用于测试(1).xlsx`(infra 测试来源)的"字段是否敏感"列;mapping 未映射 → 未进入 processed/canonical/parquet。 +- 与 `data_level` **不同步**(64 行:`L3+是=32 / L3+否=14 / L2+是=14 / L2+否=4`),证据明确,故为独立概念。 + +### 三者关系(已验证统计事实) +- `data_level` 与分类深度无关(pers_info 单层分类却分布 L1–L3);在数据上 `data_level` 对叶子类别近似确定性映射(finance 18/20、infra 4/4、pers_info 17/18、shougang 192/192 类别为单一级别,全库仅 4 条跨级样本)。 + +## 2. 三个字段在管线各层的位置 + +| 层 | classification | data_level | field_sensitive | +| --- | --- | --- | --- | +| raw(data/raw/*.xlsx) | 源列:一级子类/一级分类/分类 等 | 源列:数据级别(finance) / 分级(infra,shougang,pers_info),原始值 `1..4`(pers_info 为 `L1..L3`) | 仅 `关基设施…用于测试(1).xlsx` 有"字段是否敏感"列 | +| preprocessing(`script/preprocessing/processor.py`) | `normalize_label` 归一化(剥 `(A1-1-3)` 代码后缀) | `normalize_level`:`1/2/3/4/LEVEL1..4 → L1..L4`,其余直接 raise(仅重命名+别名归一,无转换规则) | **未映射 → 丢弃** | +| canonical(`data/canonical//all.json`) | 原样保留(provenance);另解析出 `target.category_id` 为唯一训练身份 | 原样保留(provenance;已验证 4 数据集 100% 保真) | 不存在 | +| SFT / RL parquet(`data/sft|rl/…`) | 标签 = `target.category_id`;messages 只含 prompt-visible 元数据(默认 field_name/field_description/field_type) | **不导出**(SFT row 与 RL five-field 均无 data_level) | 不存在 | + +关键代码事实: +- `src/agent/task/contracts.py::SampleTarget` 仅含 `leaf_level/leaf_name/category_id/category_path`,不含 data_level。 +- `src/agent/training/common.py::canonical_target`:训练标签唯一来自 `target.category_id`;`classification.level_1..4` 明确 "provenance only, never fallback"。 +- 全仓 grep:`data_level` 在 `src/agent`、`script/verl`、`script/canonical` 中 0 命中;仅 预处理(processor/split/rft_export)、mapping、`script/analysis`(只读统计)触及。 + +## 3. 各数据集数据来源与已知事实 + +### finance(信托核心系统) +- 原始文件:`data/raw/部分金融数据.xlsx`(568 行 ↔ processed 568;按 (table,field) 566/566 data_level 一致);标准文档 `data/raw/金融行业数据安全分类分级标准指南.xlsx`。 +- 分类体系:标准指南 → `financial_standards_dict.json`(233 条)→ finance corpus 233 / registry(path ID)。 +- data_level 来源:`部分金融数据.xlsx`"数据级别"列(原始 1/2/3/4);与标准指南"最低安全级别参考"(1级~4级)**一致**:529/568 精确对齐,37 条可经同域相邻类对齐(如 单位联系人信息 L2 ↔ 单位联系信息 2级),仅 2 条离群(AMONEY、HXTRADENO)。 +- data_level 分布:`L1:25 / L2:460 / L3:76 / L4:7`;L4 全部为"个人身份鉴别信息/传统鉴别信息"。 +- field_sensitive:无。等级语义(1~4 级代表什么):**仓库内无定义文本 → 不猜测**。 + +### shougang(首钢京唐集团) +- 原始文件:`data/raw/关基-数据分类分级目录.xlsx`("首钢京唐数据分类分级目录",234 叶,含**"分级"列** per-leaf `1/2/3`,分布 1:16/2:170/3:48);测试子集导出 `data/raw/关基设施…用于测试(1).xlsx`。 +- 分类体系:目录 → `guanji_dict.json`(234 条,**只保留类别、丢掉了"分级"列**)→ shougang corpus 233 / registry(code ID `A1-1-1`)。 +- data_level 来源:目录"分级"列;shougang 已观测的 192 个 leaf category 中,sample `data_level` 与目录对应 category 的"分级"**192/192 一致、零冲突**(32 个目录类别在数据中未出现)。因此该目录是目前可确认的 `standard_data_level` 来源——但 sample 与 standard **概念上不预设必然相等**(见 §1 两个变体)。 +- data_level 分布:`L1:899 / L2:14,188 / L3:4,328`;占位符 `——` 不可训练(1,022 条)。 +- field_sensitive:主训练数据无;仅 64 行的测试导出文件带该列。等级语义:**无定义**(目录只有数字)。 + +### infra(钢铁基建,= shougang 测试子集) +- 原始文件:`data/raw/关基设施…用于测试(1).xlsx`(64 行);逐条 (db,table,field) 与 infra processed **64/64 一致**(含 data_level);⊂ shougang。data_level 的 `standard_data_level` 来源同样是该目录(经 shougang 复用),不另设独立映射。 +- 分类/registry:同 shougang(registry_source=shougang)。 +- data_level 分布:`L2:18 / L3:46`(原"分级"值 2/3)。 +- field_sensitive:**有**("字段是否敏感",是=46 / 否=18,与分级不同步,见 §1);预处理丢弃。 +- 等级语义:无定义。 + +### pers_info(高校个人信息) +- 原始文件:`data/raw/带分级分类的个人基础信息样本190条.xlsx`(189 行 → 176,去重后;176/176 data_level 一致)。 +- 分类体系:单层 `level_4` 18 类(放入 `level_4` 槽位,见 §1);corpus 来自数据集自身 universe(`build_report.source = null`)。 +- data_level 来源:该 Excel "分级"列直接 `L1/L2/L3`;**仓库无对应标准文档 → 仅有 sample label,无确认的 standard knowledge 来源**。 +- data_level 分布:`L1:31 / L2:98 / L3:47`;1 类跨级(基本信息年级信息和班级信息 L1×1/L2×8)。 +- field_sensitive:无。等级语义:仓库内 `education_dict.json` 为"公开/内部/重要/敏感"四档,与 L1~L3 无法对应且有反例(考核信息=内部数据却标 L3 等)→ 不猜测。 + +## 4. Phase 0 冻结的核心模型 + +```text +classification data_level + sample label sample label + ↓ ↓ +canonical category_id normalized data_level + │ │ + └──────────┬─────────────────┘ + ↓ + proposed Stage2 joint target + category + data_level + (Stage2 contract 待实现 / future target) + + + 原始分类分级标准 + │ + ↓ + canonical standard(Phase 1) + category_id / description / path + standard_data_level + grading_rules(若能获得) + │ + ↓ + 为模型提供分级规则/参考 + + 与 sample data_level 审计 + + +field_sensitive = 独立 field-level annotation,不属于上述两者 +``` + +> **最关键的独立分级设计原则**:`standard_data_level` 是 category 的标准参考等级,**不预设其必然等于每个具体字段的最终 `sample_data_level`**;若目标是独立分级,Stage2 **不应直接把候选类别对应的 gold `standard_data_level` 暴露给模型**,否则 data_level 任务会退化为 category→level 查表。 + +- **Phase 0 冻结**:`data_level` 是什么(sample label 的分级标签)、来自哪里(finance/shougang/infra 可追溯到标准/目录;pers_info 尚无标准来源)、当前代码未训练它(provenance only)。 +- **Phase 0 不决定**:`data_level` 最后训不训练;不把"当前未训练"写成"设计上不训练"。 +- **设计方向(非当前实现,待 Phase 1 起逐步验证)**:从原始分类分级标准构建 canonical standard(含 `standard_data_level`),再与样本 `data_level` 做一致性审计——因此建 standard 时按各源真实分类深度建模,勿把 `level_4` 槽位当作真实标准第四层。 + +## 5. 目标任务方向与前置缺口 + +**目标任务方向:优先采用任务形态 B —— 模型独立判断字段安全等级。** + +```text +Stage1: +field metadata + 全量 leaf categories +→ Top-5 categories + +Stage2: +field metadata ++ Top-5 category descriptions/examples ++ 领域信息 ++ 分级规则/标准说明 +→ category + data_level +``` + +其中: +- `category` 与 `data_level` **都是模型预测目标**。 +- `standard_data_level` 作为标准参考和审计字段保存。 +- **默认不直接把每个候选的 `standard_data_level` 暴露给 Stage2**,否则 data_level 任务会退化为 category→level 查表。 + +**Baseline A(降级为基线):lookup-assisted grading** +向模型提供候选 category 的 `standard_data_level`,用于评估"分类 + 标准读取"任务,并作为 independent grading(Target B)的对照基线。 + +**Target B:independent grading** +不直接给候选 gold level,模型依据字段语义、业务上下文和分级规则自行判级。 + +研究问题由此清晰为: +- **A:会不会选对标准项?** +- **B:会不会真正做分级判断?** + +**其他未决项**: +- holdout 缺口:L4 仅 finance 7 条且全在 train;L1 在多数 val/test 缺失 → 分级泛化评估需处理(补充/重抽样/保持冻结)。 +- `field_sensitive` 是否建设为字段级敏感标签,及其与 data_level 的关系定义。 +- 跨数据集 L1~L4 **不推断同义**:finance / pers_info / shougang 无证据表明相同 `L3` 用同一套业务定义,仅 infra=shougang 同源明确。 +- **研究风险(需专门实验验证)**:per-category `data_level` 近确定性映射(跨级样本全库仅 4 条)——若训练集里几乎每个 category 永远只有一个 level,即使目标是独立分级,模型仍可能学成**隐式 `category → level` 查表**而非真正分级规则。后续必须专门设计实验检查:构造同 category 多 level / 跨级样本;并至少跑三组对照——**A lookup-assisted grading vs B independent grading vs B−(对 grading rules 做消融)**——用于区分模型学到的是查表还是真正分级规则。 diff --git a/docs/design/phase1_canonical_standard.md b/docs/design/phase1_canonical_standard.md new file mode 100644 index 0000000..e323f32 --- /dev/null +++ b/docs/design/phase1_canonical_standard.md @@ -0,0 +1,135 @@ +# Phase 1:无损 canonical standard 层(设计说明 + 迁移 note) + +Status: 2026-08-20。只读建立数据事实层,**不改任何现有训练行为**;不推断 +L1–L4 语义;不跨数据集假设同义;不修改样本、prompt、parser、reward、SFT/RL +parquet;`standard_data_level` 仅作标准参考,绝不覆盖 processed/canonical 的 +sample `data_level`。 + +相关文件: +- Phase 0 语义:`docs/design/data_level_design.md` +- 标准构建实现:`src/agent/standards/`(contracts/sources/build/align) +- CLI:`script/standard/cli.py` → `python -m script.standard.cli` +- 产物:`data/standards/*.standard.json`、`artifacts/generated/provenance/` + +--- + +## 1. canonical standard schema + +每个标准 category(`data/standards/.standard.json` 内): + +```jsonc +{ + "category_id": "A1-1-1 | finance:客户.个人.个人基本概况信息", + "name": "科研设备预约管理", + "path": ["研发数据域", "产品研发", "科研检验", "科研设备预约管理"], // 真实源层级深度,空层省略,不人为补空层 + "description": "...", // 该叶子所在层级的定义说明(原样保留) + "code": "A1-1-1" | null, + "standard_data_level": "L1|L2|L3|L4|null", // 规范化的标准等级;无法解析时为 null + "raw_level": "2 | l | 3 4", // 原始值,审计用 + "content": "…数据资源说明…", // 可选额外标准文本 + "source": {"file": "…", "sheet": "…", "row": …} // 可追溯到原始标准文件+行号 +} +``` + +顶层:`dataset / id_strategy / standard_name / standard_source{file,sheet} / +fingerprint` + `categories[]`。`fingerprint` 为 `categories` 内容(不含 +source 行号)的 sha256——同输入确定性一致,与输入顺序/行号无关。 + +`category_id` 延续现有稳定 identity:finance = `finance:{L1}.{L2}.{L4}` +(`level_3` 仅 provenance,与 `DatasetConfig.identity_fields` 一致);shougang += guanji code。已验证:finance 233 个标准 id == 现有 registry 233 个 id,零差异。 + +## 2. 各数据集 standard source 状态 + +| dataset | standard_source | 事实源文件 | 状态 | +| --- | --- | --- | --- | +| finance | `finance` | `data/raw/金融行业数据安全分类分级标准指南.xlsx`(sheet Table 1) | built(233 类) | +| shougang | `shougang` | `data/raw/关基-数据分类分级目录.xlsx`(sheet 数据分类分级) | built(234 类) | +| infra | `shougang`(复用,不复制维护另一套) | — | 复用共享标准(64/64 对齐) | +| pers_info | `null`(missing / unknown) | 无已确认标准 | **不生成虚假 standard_data_level** | + +pers_info:仓库内无确认的分类分级标准;18 类 registry 维持当前 +dataset-derived 行为,但**不伪装成 canonical standard**——不生成 +`pers_info.standard.json`,不在 summary 中编造等级。 + +## 3. sample ↔ standard 对齐统计(严格 category identity 连接) + +| dataset | total | resolved | matched | mismatched | standard_missing | unresolved | resolved_match_rate | +| --- | --- | --- | --- | --- | --- | --- | --- | +| finance | 568 | 531 | 529 | 2 | 0 | 37 | 99.62% | +| shougang | 19,415 | 18,393 | 18,393 | 0 | 0 | 1,022 | 100% | +| infra | 64 | 64 | 64 | 0 | 0 | 0 | 100% | +| pers_info | — | — | — | — | — | — | 无标准 | + +- finance 未解析 37 = `missing_leaf 34 + path_mismatch 3`(即既有 canonical + resolution 认定的 37 条非训练样本,类别不在 registry 中;本阶段不改)。标准覆盖 + 233 类中 213 类未被任何样本观测到(universe ≫ 观测叶)。 +- shougang 未解析 1,022 = 数据侧 `level_4='——'` 占位样本(与目录中的 `——` + 不同义,见 §4)。 +- 对齐只读:不修改 sample `data_level`,不自动修标签。 + +## 4. 发现的数据异常(只报告,不修复) + +1. **finance 原始标准 2 处不可解析等级**:row 150 `市场营销信息(公开)` 原始值 + `l`(疑似 `1` 的笔误)、row 168 `客户及监管相关音影像信息` 原始值 `3 4` + (歧义)。→ `standard_data_level=null` + build issue,不做猜测。 +2. **shougang 目录中"三级即叶子"层级**:10 行使 四列为文字 `——`、叶子码在 三级 + (B1-2 合同归并、B1-5 合同跟踪、B3-3/4/5、B4-2、B6-2、C1-5、C4-1)。它们 + **不是占位符**,是真实类别;canonical standard 已按真实深度 3 层保存 + (不发明第 4 层)。这是 Phase 0 规则 7 的直接实例。 +3. **数据侧 `——` 与目录侧 `——` 语义不同**:shougang 样本的 `level_4='——'` + 是不可训练占位标签;目录中的 `——` 是"该层无子结点"标记。二者分别处理, + 不相干。 +4. **legacy 信息损失(dict vs 原始标准)**: + - finance:`financial_standards_dict.json` 把真实 4 层 path 压成 + L1-L2-leaf(232/233 类丢了 三级子类 provenance 层);`l`/`3 4` 原始值在 + dict 中保留为 `l级`/`3 4级`。 + - shougang:`guanji_dict.json` 只保留 `name(code)`+描述,**丢了全部 path** + (registry path 为空)**和分级列**(无 class);并**漏掉 B3-6 中厚板作业计划** + (目录中真实存在,registry 233 vs 标准 234)。 + +## 5. 下一阶段(registry/corpus 接口变化) + +当前:`raw standard(Excel) →(lossy) standards_map JSON → canonical_corpus → +LeafRegistry + Corpus`(`financial_standards_dict.json`、`guanji_dict.json`、 +`src/agent/task/canonical_corpus.py` 均为**有损中间层**)。 + +建议迁移为: +``` +raw standard(Excel) + → canonical standard(本项目,无损:path/description/code/standard_data_level) + → LeafRegistry + Corpus +``` +为此后续需修改的接口(本阶段不实现,只列出): +1. `src/agent/task/canonical_corpus.py` 的 `parse_financial_standard` / + `parse_guanji_standard` 改为消费 `CanonicalStandard`(或标志位切换数据源), + `path` 使用标准真实深度而非补空层。 +2. registry 生成在钳制当前 233(finance)/233(shougang)兼容的同时, + 决策是否收养标准的第 234 类 B3-6(当前数据 0 样本,仅宇宙完整性问题)。 +3. `corpus_to_mapping` 是否附带 `standard_data_level` 作为 Stage2 知识—— + **属于下一步 task contract 决策**(Phase 0 §5:形态 A vs B),本阶段不预置。 +4. 现有 dict 产物降级为 `legacy/derived`:审计对照用,不再作事实源。 + +## 6. 测试 + +``` +tests/standards/ + test_contracts.py round-trip / normalize / fragment / determinism + test_build_finance.py 无损 path、identity 规则、异常上报、确定性 + test_build_shougang.py code/name/path/level、三级叶、no_code、确定性 + test_align.py 对齐桶、不修改样本、路由(infra→shougang、pers_info→None) + test_real_xlsx.py 真实 Excel + canonical 集成断言(raw 缺失时 skip) +``` + +结果:`pytest tests/standards` → **26 passed**(raw 文件存在时集成用例全跑); +全仓 pytest 见 PR 附注(本阶段无训练链路改动,回归为预防性)。 + +产物可重生成:`python -m script.standard.cli`(拒绝无 `--overwrite` 覆盖; +先行全部构建/对齐、后写盘;输出 sort_keys + 类别按 id 排序,重复构建字节级一致)。 +raw Excel 仍是唯一事实源,但不进入任何训练代码依赖。 + +**git 边界**:`data/standards/*.standard.json` 被 `.gitignore` 的 `/data/*` 排除, +与 `data/processed`、`data/canonical` 同属**可再生层**(依赖 `data/raw` 恢复); +入库的是 `src/agent/standards/`、`script/standard/`、`tests/standards/`、 +`docs/design/phase1_canonical_standard.md`(+ Phase 0 的 `data_level_design.md`) +以及 `artifacts/generated/provenance/`(对齐审计 + 构建 summary,未忽略)。 diff --git a/script/standard/__init__.py b/script/standard/__init__.py new file mode 100644 index 0000000..e69de29 diff --git a/script/standard/cli.py b/script/standard/cli.py new file mode 100644 index 0000000..391b497 --- /dev/null +++ b/script/standard/cli.py @@ -0,0 +1,295 @@ +"""Phase 1 canonical standard CLI: build standards + alignment artifacts. + +Usage: + python -m script.standard.cli [--overwrite] + +Reads the ORIGINAL standard workbooks (data/raw) and the canonical dataset +records (data/canonical), then writes: + + data/standards/finance.standard.json + data/standards/shougang.standard.json + artifacts/generated/provenance/finance_standard_alignment.json + artifacts/generated/provenance/shougang_standard_alignment.json + artifacts/generated/provenance/infra_standard_alignment.json + artifacts/generated/provenance/standard_build_summary.json + +Fail-fast: every dataset is read + built + aligned before anything is +written; all outputs are written exactly once. Deterministic: JSON is written +with sort_keys and categories/entries are pre-sorted; no timestamps or +machine-local paths. Raw workbooks are never training dependencies. +""" + +from __future__ import annotations + +import argparse +import json +from pathlib import Path +import sys + +PROJECT_ROOT = Path(__file__).resolve().parents[2] +sys.path.insert(0, str(PROJECT_ROOT / "src")) + +from agent.standards.align import align_dataset_to_standard, load_canonical_records +from agent.standards.build import ( + build_finance_standard, + build_shougang_standard, + resolve_standard_dataset, +) +from agent.standards.contracts import CanonicalStandard +from agent.standards.sources import ( + read_finance_standard_guide, + read_guanji_catalog, +) + +DEFAULT_RAW_DIR = PROJECT_ROOT / "data" / "raw" +DEFAULT_CANONICAL_DIR = PROJECT_ROOT / "data" / "canonical" +DEFAULT_STANDARD_DIR = PROJECT_ROOT / "data" / "standards" +DEFAULT_ARTIFACT_DIR = PROJECT_ROOT / "artifacts" / "generated" / "provenance" + +FINANCE_XLSX = "金融行业数据安全分类分级标准指南.xlsx" +SHOUGANG_XLSX = "关基-数据分类分级目录.xlsx" + + +def _repo_relative(path: str | Path) -> str: + try: + return Path(path).resolve().relative_to(PROJECT_ROOT.resolve()).as_posix() + except ValueError: + return Path(path).as_posix() + + +def _write_json(payload, path: Path) -> None: + path.parent.mkdir(parents=True, exist_ok=True) + with path.open("w", encoding="utf-8", newline="\n") as handle: + json.dump(payload, handle, ensure_ascii=False, indent=2, sort_keys=True) + handle.write("\n") + + +def _registry_ids(dataset: str) -> set[str]: + from agent.task import LeafRegistry + + path = PROJECT_ROOT / "cfg" / "task" / "registry" / f"{dataset}.registry.json" + return set(LeafRegistry.from_path(path).ids) + + +def _legacy_information_loss( + finance_standard: CanonicalStandard, + shougang_standard: CanonicalStandard, +) -> dict[str, object]: + """Quantify what the legacy standards_map digests dropped vs the raw + standard canonical build (path depth, grading, missing codes).""" + + import json as _json + import re as _re + + finance_out: dict[str, object] = {} + finance_ids = {c.category_id for c in finance_standard.categories} + finance_reg_ids = _registry_ids("finance") + finance_out["standard_vs_registry_ids"] = { + "standard_ids": len(finance_ids), + "registry_ids": len(finance_reg_ids), + "missing_from_registry": len(finance_ids - finance_reg_ids), + } + # legacy dict: how many categories lost a real path level + legacy = _json.load( + ( + PROJECT_ROOT / "data" / "knowledge" / "standards_map" + / "financial_standards_dict.json" + ).open(encoding="utf-8") + ) + depth_lost = 0 + depth_kept = 0 + for category in finance_standard.categories: + segments = 3 # legacy dict identity was L1-L2-leaf + if len(category.path) > segments: + depth_lost += 1 + else: + depth_kept += 1 + finance_out["legacy_dict_path_compression"] = { + "categories_with_path_deeper_than_legacy_L1_L2_leaf": depth_lost, + "categories_at_legacy_depth": depth_kept, + "note": "legacy financial_standards_dict stored L1-L2-leaf identity " + "strings; the real standard has 三级子类 provenance nodes that the " + "legacy digest dropped", + } + finance_out["legacy_dict_entries"] = len(legacy) + finance_out["legacy_unparseable_level_values"] = sorted( + str(v.get("class", "")) for v in legacy.values() if isinstance(v, dict) + and v.get("class") in ("l级", "3 4级") + ) + + shougang_out: dict[str, object] = {} + shougang_ids = {c.category_id for c in shougang_standard.categories} + shougang_reg_ids = _registry_ids("shougang") + shougang_out["standard_vs_registry_codes"] = { + "standard_codes": len(shougang_ids), + "registry_codes": len(shougang_reg_ids), + "standard_only": sorted(shougang_ids - shougang_reg_ids), + "registry_only": sorted(shougang_reg_ids - shougang_ids), + "note": "B3-6 中厚板作业计划 exists in the raw catalog but was dropped " + "by guanji_dict/registry", + } + legacy_g = _json.load( + ( + PROJECT_ROOT / "data" / "knowledge" / "standards_map" / "guanji_dict.json" + ).open(encoding="utf-8") + ) + codes_without_path = 0 + codes_with_level = 0 + for category in shougang_standard.categories: + if category.path: + codes_without_path += 1 # path restored where legacy had none + if category.standard_data_level: + codes_with_level += 1 # grading restored where legacy had none + shougang_out["legacy_dict_losses"] = { + "catalog_categories": len(shougang_standard.categories), + "legacy_dict_entries": len(legacy_g), + "with_real_path_restored": codes_without_path, + "with_grading_restored": codes_with_level, + "note": "guanji_dict kept only 'name(code)' + description: real " + "hierarchy path and the 分级 column were dropped (registry path was " + "[] and no class field existed)", + } + return {"finance": finance_out, "shougang": shougang_out} + + +def main(argv: list[str] | None = None) -> int: + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("--raw-dir", type=Path, default=DEFAULT_RAW_DIR) + parser.add_argument("--canonical-dir", type=Path, default=DEFAULT_CANONICAL_DIR) + parser.add_argument("--standard-dir", type=Path, default=DEFAULT_STANDARD_DIR) + parser.add_argument("--artifact-dir", type=Path, default=DEFAULT_ARTIFACT_DIR) + parser.add_argument("--overwrite", action="store_true") + args = parser.parse_args(argv) + + raw_dir = Path(args.raw_dir) + finance_xlsx = raw_dir / FINANCE_XLSX + shougang_xlsx = raw_dir / SHOUGANG_XLSX + missing = [p for p in (finance_xlsx, shougang_xlsx) if not p.is_file()] + if missing: + raise FileNotFoundError( + "raw standard workbook(s) missing (data/raw is gitignored; restore " + "them from the data provider before building): " + + ", ".join(str(p) for p in missing) + ) + + # 1. read + build + align EVERYTHING (pure computation, no writes) + finance_raw = read_finance_standard_guide(finance_xlsx) + shougang_raw = read_guanji_catalog(shougang_xlsx) + finance_standard, finance_report = build_finance_standard( + finance_raw.entries, + source_file=_repo_relative(finance_xlsx), + source_sheet="Table 1", + ) + shougang_standard, shougang_report = build_shougang_standard( + shougang_raw.entries, + source_file=_repo_relative(shougang_xlsx), + source_sheet="数据分类分级", + ) + + canonical_records = {} + for dataset in ("finance", "shougang", "infra"): + canonical_records[dataset] = load_canonical_records( + args.canonical_dir / dataset / "all.json" + ) + + standard_key = resolve_standard_dataset("infra") # -> shougang (shared) + alignments = { + "finance": align_dataset_to_standard( + canonical_records["finance"], finance_standard + ), + "shougang": align_dataset_to_standard( + canonical_records["shougang"], shougang_standard + ), + "infra": align_dataset_to_standard( + canonical_records["infra"], shougang_standard + ), + } + + loss = _legacy_information_loss(finance_standard, shougang_standard) + + # 2. refuse to overwrite without --overwrite + outputs = { + "standards/finance": args.standard_dir / "finance.standard.json", + "standards/shougang": args.standard_dir / "shougang.standard.json", + "provenance/finance_alignment": args.artifact_dir / "finance_standard_alignment.json", + "provenance/shougang_alignment": args.artifact_dir / "shougang_standard_alignment.json", + "provenance/infra_alignment": args.artifact_dir / "infra_standard_alignment.json", + "provenance/summary": args.artifact_dir / "standard_build_summary.json", + } + if not args.overwrite: + existing = [str(p) for p in outputs.values() if p.exists()] + if existing: + raise FileExistsError( + "refusing to overwrite existing canonical-standard artifacts: " + + ", ".join(existing) + + " (pass --overwrite to regenerate)" + ) + + # 3. build the summary first (fail-fast: nothing is written until every + # computed payload is ready) + summary = { + "phase": "phase1-canonical-standard", + "standards": { + dataset: { + "standard_source": ( + resolve_standard_dataset(dataset) + if resolve_standard_dataset(dataset) + else None + ), + "status": ( + "missing_or_unknown" + if resolve_standard_dataset(dataset) is None + else "built" + ), + "categories": ( + len( + { + "finance": finance_standard, + "shougang": shougang_standard, + }[resolve_standard_dataset(dataset)].categories + ) + if resolve_standard_dataset(dataset) in ("finance", "shougang") + else 0 + ), + "build": ( + finance_report.to_mapping() + if dataset == "finance" + else shougang_report.to_mapping() + if dataset in ("shougang", "infra") + else None + ), + "alignment_headline": { + "samples_total": alignments[dataset]["sample_counts"].get("total", 0), + "resolved": alignments[dataset]["sample_counts"].get("resolved", 0), + "matched": alignments[dataset]["sample_counts"].get("matched", 0), + "mismatched": alignments[dataset]["sample_counts"].get("mismatched", 0), + "standard_missing": alignments[dataset]["sample_counts"].get("standard_missing", 0), + "resolved_match_rate": alignments[dataset]["resolved_match_rate"], + }, + } + for dataset in ("finance", "shougang", "infra") + }, + "pers_info": { + "standard_source": None, + "status": "missing_or_unknown", + "note": "no confirmed classification/grading standard; the 18-category " + "registry remains dataset-derived and is NOT presented as a canonical " + "standard; no standard_data_level is fabricated", + "alignment_headline": None, + }, + "legacy_information_loss": loss, + } + # 4. write everything exactly once + _write_json(finance_standard.to_mapping(), outputs["standards/finance"]) + _write_json(shougang_standard.to_mapping(), outputs["standards/shougang"]) + for key in ("finance", "shougang", "infra"): + _write_json(alignments[key], outputs[f"provenance/{key}_alignment"]) + _write_json(summary, outputs["provenance/summary"]) + + for name, path in outputs.items(): + print(f"wrote: {_repo_relative(path)}") + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/src/agent/standards/__init__.py b/src/agent/standards/__init__.py new file mode 100644 index 0000000..92db8bc --- /dev/null +++ b/src/agent/standards/__init__.py @@ -0,0 +1,58 @@ +"""Canonical standard layer (Phase 1). + +Lossless, auditable, reproducible canonical form of the original +classification/grading standards. Distinct from sample labels: see +docs/design/data_level_design.md (Phase 0). +""" + +from .contracts import ( + LEVELS, + CanonicalStandard, + SourceRef, + StandardCategory, + StandardCategoryBuilder, + clean, + compact, + normalize_standard_level, + strip_code, +) +from .sources import ( + RawEntry, + ReaderResult, + read_finance_standard_guide, + read_guanji_catalog, +) +from .build import ( + BuildIssue, + StandardBuildReport, + build_finance_standard, + build_shougang_standard, + resolve_standard_dataset, +) +from .align import ( + align_dataset_to_standard, + load_canonical_records, +) + +__all__ = [ + "LEVELS", + "CanonicalStandard", + "SourceRef", + "StandardCategory", + "StandardCategoryBuilder", + "clean", + "compact", + "normalize_standard_level", + "strip_code", + "RawEntry", + "ReaderResult", + "read_finance_standard_guide", + "read_guanji_catalog", + "BuildIssue", + "StandardBuildReport", + "build_finance_standard", + "build_shougang_standard", + "resolve_standard_dataset", + "align_dataset_to_standard", + "load_canonical_records", +] diff --git a/src/agent/standards/align.py b/src/agent/standards/align.py new file mode 100644 index 0000000..c26ec11 --- /dev/null +++ b/src/agent/standards/align.py @@ -0,0 +1,157 @@ +"""sample <-> canonical standard alignment audit (Phase 1). + +Reads canonical records (data/canonical//all.json — the same records +the training pipeline consumes, unchanged) and joins them to a +``CanonicalStandard`` by canonical category identity. + +Buckets (strict identity join): +- matched : standard exists and standard_data_level == sample data_level +- mismatched : standard exists, both levels known, and they differ +- standard_level_unavailable : standard exists but its level is unparseable (null) +- standard_missing : sample category_id not in the standard +- sample_missing : standard category observed in no resolved sample +- unresolved : canonical records without a resolved target + +Pure computation: NEVER modifies sample data_level, never auto-repairs labels, +never guesses the meaning of a level. ``standard_data_level`` is only the +standard's category-level reference. +""" + +from __future__ import annotations + +import json +from collections import Counter +from pathlib import Path +from typing import Any, Mapping, Sequence + +from agent.standards.contracts import CanonicalStandard + + +def load_canonical_records(path: str | Path) -> list[dict[str, Any]]: + with Path(path).open(encoding="utf-8") as handle: + records = json.load(handle) + if not isinstance(records, list): + raise ValueError(f"{path} must be a JSON list") + return records + + +def align_dataset_to_standard( + records: Sequence[Mapping[str, Any]], + standard: CanonicalStandard, + *, + field_for_audit: str = "field_name", +) -> dict[str, Any]: + """Return a deterministic alignment report (no mutation of ``records``).""" + standard_by_id = standard.by_id() + counts: Counter[str] = Counter() + unresolved_by_status: Counter[str] = Counter() + mismatched: list[dict[str, Any]] = [] + standard_missing: list[dict[str, Any]] = [] + level_unavailable: list[dict[str, Any]] = [] + resolved_categories: set[str] = set() + + for record in records: + counts["total"] += 1 + status = str(record.get("resolution_status", "") or "") + target = record.get("target") + if status != "resolved" or not isinstance(target, Mapping): + unresolved_by_status[status or "(no status)"] += 1 + counts["unresolved"] += 1 + continue + category_id = str(target.get("category_id", "") or "") + sample_level = str(record.get("data_level", "") or "") + counts["resolved"] += 1 + resolved_categories.add(category_id) + + category = standard_by_id.get(category_id) + if category is None: + counts["standard_missing"] += 1 + standard_missing.append( + { + "category_id": category_id, + "name": str(target.get("leaf_name", "") or ""), + "sample_level": sample_level, + "path": list(target.get("category_path") or ()), + "field": _audit_field(record, field_for_audit), + } + ) + continue + standard_level = category.standard_data_level + if standard_level is None: + counts["standard_level_unavailable"] += 1 + level_unavailable.append( + { + "category_id": category_id, + "name": category.name, + "sample_level": sample_level, + "raw_level": category.raw_level, + "field": _audit_field(record, field_for_audit), + } + ) + continue + if sample_level == standard_level: + counts["matched"] += 1 + else: + counts["mismatched"] += 1 + mismatched.append( + { + "category_id": category_id, + "name": category.name, + "sample_level": sample_level, + "standard_level": standard_level, + "standard_raw_level": category.raw_level, + "source_row": category.source.row, + "field": _audit_field(record, field_for_audit), + "sample_path": list(target.get("category_path") or ()), + } + ) + + # standard categories never observed as a resolved sample + sample_missing = sorted(set(standard_by_id) - resolved_categories) + + # near-alias candidates for standard-missing samples (leaf-name overlap), + # so the audit can distinguish "different standard branch" from "lost" + for item in standard_missing: + item["near_by_name"] = sorted( + category_id + for category_id, category in standard_by_id.items() + if category.name == item["name"] and category_id != item["category_id"] + ) + + resolved = counts["resolved"] + sample_counts = { + "total": counts["total"], + "resolved": counts["resolved"], + "unresolved": counts["unresolved"], + "matched": counts["matched"], + "mismatched": counts["mismatched"], + "standard_missing": counts["standard_missing"], + "standard_level_unavailable": counts["standard_level_unavailable"], + } + return { + "standard": standard.dataset, + "standard_categories": len(standard.categories), + "standard_categories_observed": len(resolved_categories), + "standard_categories_unobserved": len(sample_missing), + "sample_counts": dict(sorted(sample_counts.items())), + "resolved_match_rate": round(counts["matched"] / resolved, 4) if resolved else 0.0, + "unresolved_by_status": dict(sorted(unresolved_by_status.items())), + "mismatched_samples": sorted(mismatched, key=lambda x: (x["category_id"], x["field"])), + "standard_missing_samples": sorted(standard_missing, key=lambda x: x["category_id"]), + "standard_level_unavailable_samples": sorted( + level_unavailable, key=lambda x: (x["category_id"], x["field"]) + ), + "sample_missing_standard_categories": sample_missing, + } + + +def _audit_field(record: Mapping[str, Any], field_for_audit: str) -> str: + metadata = record.get("metadata") + if isinstance(metadata, Mapping): + value = metadata.get(field_for_audit, "") + if value: + return str(value) + return "" + + +__all__ = ["load_canonical_records", "align_dataset_to_standard"] diff --git a/src/agent/standards/build.py b/src/agent/standards/build.py new file mode 100644 index 0000000..faadb79 --- /dev/null +++ b/src/agent/standards/build.py @@ -0,0 +1,348 @@ +"""Canonical standard builders (Phase 1). + +Turn raw standard rows (from ``sources``) into a lossless, auditable +``CanonicalStandard``: +- category_id continues the existing stable identity strategy: finance uses + the path-qualified L1/L2-leaf identity (level_3 excluded, matching + DatasetConfig.identity_fields), shougang uses the guanji code. +- path stores the TRUE source-hierarchy depth (empty 三级子类 omitted — no + invented padding). +- grading columns are normalized to L1..L4 with ``normalize_standard_level``; + unparseable values are reported and kept raw, never fixed or guessed. +- placeholder / malformed rows (shougang "——", NaN, missing code) are + skipped and reported in the build report — never silently repaired. + +Builds are deterministic for identical input (sorted categories / issues); +the CLI writes every artifact only after all datasets build and align +successfully (fail-fast) and refuses to overwrite without --overwrite. +""" + +from __future__ import annotations + +import re +from collections import Counter +from dataclasses import dataclass, field +from pathlib import Path +from typing import Any, Iterable, Mapping, Sequence + +from agent.task.identity import qualified_category_id +from agent.standards.contracts import ( + CanonicalStandard, + SourceRef, + StandardCategory, + StandardCategoryBuilder, + clean, + compact, + normalize_standard_level, + strip_code, +) + +_PLACEHOLDER_NAMES = {"——", "nan", "none", "-", ""} +_TRAILING_CODE_RE = re.compile( + r"[\(\[(【]\s*([A-Za-z]+\d*(?:-\d+)*)\s*[\)\])】]\s*$" +) + + +@dataclass(frozen=True) +class BuildIssue: + kind: str + detail: str + + def to_mapping(self) -> dict[str, str]: + return {"kind": self.kind, "detail": self.detail} + + +@dataclass +class StandardBuildReport: + dataset: str + id_strategy: str + standard_name: str + source_file: str + source_sheet: str + entries_read: int = 0 + categories_out: int = 0 + aggregated: dict[str, int] = field(default_factory=dict) + level_distribution: dict[str, int] = field(default_factory=dict) + issues: list[BuildIssue] = field(default_factory=list) + + def to_mapping(self) -> dict[str, Any]: + return { + "dataset": self.dataset, + "id_strategy": self.id_strategy, + "standard_name": self.standard_name, + "source_file": self.source_file, + "source_sheet": self.source_sheet, + "entries_read": self.entries_read, + "categories_out": self.categories_out, + "aggregated": dict(self.aggregated), + "level_distribution": dict(self.level_distribution), + "issues": sorted( + (issue.to_mapping() for issue in self.issues), + key=lambda i: (i["kind"], i["detail"]), + ), + } + + +def _finalize_report_levels(report: StandardBuildReport, categories: Sequence[StandardCategory]) -> None: + """Level distribution over the FINAL aggregated categories (not raw rows). + + Unparseable / missing levels are counted under an empty key. + """ + counter: Counter[str] = Counter() + for category in categories: + counter[category.standard_data_level or "''"] += 1 + report.level_distribution = dict(sorted(counter.items())) + + +def _add_issue(report: StandardBuildReport, kind: str, detail: str) -> None: + report.issues.append(BuildIssue(kind=kind, detail=detail)) + + +def _dedupe_path(parts: Sequence[str]) -> tuple[str, ...]: + """Drop empty parts and consecutive duplicates (a leaf living at level 3 + would otherwise repeat its name in the path).""" + result: list[str] = [] + for part in parts: + part = part.strip() + if not part: + continue + if result and result[-1] == part: + continue + result.append(part) + return tuple(result) + + +def _aggregate( + categories: Iterable[StandardCategory], + report: StandardBuildReport, +) -> list[StandardCategory]: + """Merge rows that share a category_id (lossless: extras go to + ``descriptions``). Reports level conflicts within one id, never fixes.""" + counts: Counter[str] = Counter() + by_id: dict[str, StandardCategory] = {} + collisions: list[tuple[str, str, str, str]] = [] # id, extra_level, row, primary_level + for category in categories: + counts[category.category_id] += 1 + existing = by_id.get(category.category_id) + if existing is None: + by_id[category.category_id] = category + continue + if ( + category.standard_data_level + and existing.standard_data_level + and category.standard_data_level != existing.standard_data_level + ): + collisions.append( + ( + category.category_id, + category.standard_data_level, + str(category.source.row), + existing.standard_data_level, + ) + ) + by_id[category.category_id] = StandardCategory( + category_id=existing.category_id, + name=existing.name, + path=existing.path, + description=existing.description, + code=existing.code, + standard_data_level=existing.standard_data_level, + raw_level=existing.raw_level, + content=existing.content, + source=existing.source, + descriptions=existing.descriptions + + ((category.description,) if category.description else ()), + ) + repeated = {id_: n for id_, n in counts.items() if n > 1} + report.aggregated = { + "kinds": len(repeated), + "instances": sum(n - 1 for n in repeated.values()), + } + for category_id, extra, row, primary in sorted(collisions): + _add_issue( + report, + "level_conflict_within_category", + f"category {category_id!r}: row {row} level {extra!r} differs from " + f"primary {primary!r} (kept primary, not fixed)", + ) + # deterministic, even when the input order varies + return sorted(by_id.values(), key=lambda c: c.category_id) + + +def build_finance_standard( + entries: Sequence[Any], + *, + source_file: str, + source_sheet: str, + dataset: str = "finance", + standard_name: str = "金融行业数据安全分类分级标准指南", +) -> tuple[CanonicalStandard, StandardBuildReport]: + """Build the finance canonical standard from raw guide entries.""" + report = StandardBuildReport( + dataset=dataset, + id_strategy="path", + standard_name=standard_name, + source_file=source_file, + source_sheet=source_sheet, + entries_read=len(entries), + ) + categories: list[StandardCategory] = [] + for entry in entries: + level_1 = clean(entry.level_1 if hasattr(entry, "level_1") else entry.get("level_1")) + level_2 = clean(entry.level_2 if hasattr(entry, "level_2") else entry.get("level_2")) + level_3 = clean(entry.level_3 if hasattr(entry, "level_3") else entry.get("level_3")) + leaf = clean(entry.leaf if hasattr(entry, "leaf") else entry.get("leaf")) + content = clean(entry.description if hasattr(entry, "description") else entry.get("description")) + raw_level = str(entry.raw_level if hasattr(entry, "raw_level") else entry.get("raw_level", "")).strip() + row = entry.row if hasattr(entry, "row") else entry.get("row") + + if not leaf: + _add_issue(report, "empty_leaf_skipped", f"row {row}: empty leaf") + continue + # identity: dataset identity_fields = (level_1, level_2, level_4); a + # present 三级子类 stays provenance-only (no invented level_3 slot) + category_id = qualified_category_id(dataset, (level_1, level_2, leaf)) + path = _dedupe_path((level_1, level_2, level_3, leaf)) + level, raw_clean = normalize_standard_level(raw_level) + if raw_level and level is None: + _add_issue( + report, + "level_unparseable", + f"row {row} category {leaf!r}: raw level {raw_clean!r} kept as-is, " + f"standard_data_level=null (not guessed)", + ) + categories.append( + StandardCategoryBuilder( + category_id=category_id, + name=leaf, + path=path, + description=content, + code=None, + raw_level=raw_level, + source_file=source_file, + source_sheet=source_sheet, + source_row=row, + ).build() + ) + report.categories_out = len(categories) + final = _aggregate(categories, report) + _finalize_report_levels(report, final) + return CanonicalStandard( + dataset=dataset, + id_strategy="path", + standard_source=SourceRef(file=source_file, sheet=source_sheet), + standard_name=standard_name, + categories=tuple(final), + ), report + + +def build_shougang_standard( + entries: Sequence[Any], + *, + source_file: str, + source_sheet: str, + dataset: str = "shougang", + standard_name: str = "首钢京唐数据分类分级目录(关基)", +) -> tuple[CanonicalStandard, StandardBuildReport]: + """Build the shougang canonical standard from the raw guanji catalog.""" + report = StandardBuildReport( + dataset=dataset, + id_strategy="code", + standard_name=standard_name, + source_file=source_file, + source_sheet=source_sheet, + entries_read=len(entries), + ) + categories: list[StandardCategory] = [] + for entry in entries: + level_1 = clean(entry.level_1 if hasattr(entry, "level_1") else entry.get("level_1")) + level_2 = clean(entry.level_2 if hasattr(entry, "level_2") else entry.get("level_2")) + level_3 = clean(entry.level_3 if hasattr(entry, "level_3") else entry.get("level_3")) + raw_leaf = clean(entry.leaf if hasattr(entry, "leaf") else entry.get("leaf")) + description = clean(entry.description if hasattr(entry, "description") else entry.get("description")) + content = clean(entry.content if hasattr(entry, "content") else entry.get("content")) + raw_level = str(entry.raw_level if hasattr(entry, "raw_level") else entry.get("raw_level", "")).strip() + row = entry.row if hasattr(entry, "row") else entry.get("row") + + # a "——"/empty leaf cell means the catalog's leaf lives one level up: + # fall back to the 三级 cell (the real reader already resolves this, but + # accept dict fixtures and enforce the invariant here too) + if not raw_leaf or raw_leaf.lower() in _PLACEHOLDER_NAMES or raw_leaf in _PLACEHOLDER_NAMES: + raw_leaf = clean(entry.level_3 if hasattr(entry, "level_3") else entry.get("level_3")) + if not raw_leaf or raw_leaf in _PLACEHOLDER_NAMES: + _add_issue( + report, + "placeholder_skipped", + f"row {row}: no real leaf level (all ——); skipped and reported", + ) + continue + match = _TRAILING_CODE_RE.search(raw_leaf) + if match: + code = match.group(1) + name = strip_code(raw_leaf) + else: + _add_issue( + report, + "no_code", + f"row {row}: leaf {raw_leaf!r} has no category code; skipped " + f"(identity is code-based, not guessed)", + ) + continue + path = _dedupe_path((strip_code(level_1), strip_code(level_2), strip_code(level_3), name)) + level, raw_clean = normalize_standard_level(raw_level) + if raw_level and level is None: + _add_issue( + report, + "level_unparseable", + f"row {row} category {name!r}: raw level {raw_clean!r} kept as-is, " + f"standard_data_level=null (not guessed)", + ) + categories.append( + StandardCategoryBuilder( + category_id=code, + name=name, + path=path, + description=description, + code=code, + raw_level=raw_level, + content=content, + source_file=source_file, + source_sheet=source_sheet, + source_row=row, + ).build() + ) + report.categories_out = len(categories) + final = _aggregate(categories, report) + _finalize_report_levels(report, final) + return CanonicalStandard( + dataset=dataset, + id_strategy="code", + standard_source=SourceRef(file=source_file, sheet=source_sheet), + standard_name=standard_name, + categories=tuple(final), + ), report + + +def resolve_standard_dataset(dataset: str) -> str | None: + """Which canonical standard owns a dataset's category facts. + + - finance -> finance + - shougang -> shougang + - infra -> shougang (reuses the shared guanji standard; no copy) + - pers_info -> None (no confirmed classification/grading standard) + """ + return { + "finance": "finance", + "shougang": "shougang", + "infra": "shougang", + "pers_info": None, + }[dataset] + + +__all__ = [ + "BuildIssue", + "StandardBuildReport", + "build_finance_standard", + "build_shougang_standard", + "resolve_standard_dataset", +] diff --git a/src/agent/standards/contracts.py b/src/agent/standards/contracts.py new file mode 100644 index 0000000..bfabf30 --- /dev/null +++ b/src/agent/standards/contracts.py @@ -0,0 +1,255 @@ +"""Canonical standard contracts (Phase 1). + +Phase 0 frozen semantics respected here: +- ``sample_data_level`` (per-field label in processed/canonical) and + ``standard_data_level`` (grade the classification/grading standard assigns + to a category) are distinct and are NEVER merged or overwritten. +- ``standard_data_level`` is a category-level reference only; no natural- + language semantics of L1..L4 are asserted here, and levels are not assumed + equivalent across datasets. +- ``path`` stores the REAL source-hierarchy depth of the standard; empty + levels are omitted (no invented padding). +- category_id continues the existing stable identity strategy so the + standard stays joinable to the current registry/canonical targets. + +Determinism: categories are canonicalized by sorted category_id and JSON is +written with sort_keys; ``fingerprint()`` is a sha256 over the canonical +categories payload (no timestamps, no machine-local paths). +""" + +from __future__ import annotations + +import hashlib +import json +import re +from dataclasses import dataclass, field +from typing import Any, Iterable, Mapping, Sequence + +_WS_COLLAPSE_RE = re.compile(r"\s+") +_WS_REMOVE_RE = re.compile(r"\s+") +_TRAILING_CODE_RE = re.compile( + r"\s*[\(\[(【]\s*([A-Za-z]+\d*(?:-\d+)*)\s*[\)\])】]\s*$" +) + +LEVELS = ("L1", "L2", "L3", "L4") + +# Aliases accepted when normalizing a raw standard level value. Anything not +# covered here is kept raw and reported, never guessed. +_LEVEL_ALIASES: dict[str, str] = { + "1": "L1", "2": "L2", "3": "L3", "4": "L4", + "L1": "L1", "L2": "L2", "L3": "L3", "L4": "L4", + "LEVEL1": "L1", "LEVEL2": "L2", "LEVEL3": "L3", "LEVEL4": "L4", + "1级": "L1", "2级": "L2", "3级": "L3", "4级": "L4", +} + + +def clean(value: Any) -> str: + """Collapse every whitespace run to a single space and strip.""" + if value is None: + return "" + return _WS_COLLAPSE_RE.sub(" ", str(value).strip()) + + +def compact(value: Any) -> str: + """Remove every whitespace character (identity seed; matches identity.py).""" + if value is None: + return "" + return _WS_REMOVE_RE.sub("", str(value)) + + +def strip_code(text: str) -> str: + """Remove a trailing classification code such as (A1-1-1), (A), 【A】.""" + return _TRAILING_CODE_RE.sub("", text).strip() + + +def normalize_standard_level( + raw: Any, +) -> tuple[str | None, str]: + """Return (canonical L1..L4 | None, cleaned raw value). + + Unparseable values (e.g. 'l', '3 4' from the finance guide) map to None + and keep the raw text; callers must report them, never fix or guess. + """ + text = clean(raw) + if not text: + return None, "" + normalized = _LEVEL_ALIASES.get(text.upper()) + return (normalized, text) if normalized is not None else (None, text) + + +@dataclass(frozen=True) +class SourceRef: + """Traceable origin of one standard category.""" + + file: str = "" + sheet: str = "" + row: int | None = None + + def to_mapping(self) -> dict[str, Any]: + return {"file": self.file, "sheet": self.sheet, "row": self.row} + + @classmethod + def from_mapping(cls, value: Mapping[str, Any]) -> "SourceRef": + return cls( + file=str(value.get("file", "") or ""), + sheet=str(value.get("sheet", "") or ""), + row=value.get("row"), + ) + + +@dataclass(frozen=True) +class StandardCategory: + """One category of a canonical standard (the standard's own facts).""" + + category_id: str + name: str + path: tuple[str, ...] = () + description: str = "" + code: str | None = None + standard_data_level: str | None = None + raw_level: str = "" + content: str = "" + source: SourceRef = field(default_factory=SourceRef) + descriptions: tuple[str, ...] = () + + def to_mapping(self) -> dict[str, Any]: + mapping: dict[str, Any] = { + "category_id": self.category_id, + "name": self.name, + "path": list(self.path), + "description": self.description, + "code": self.code, + "standard_data_level": self.standard_data_level, + "raw_level": self.raw_level, + "source": self.source.to_mapping(), + } + if self.content: + mapping["content"] = self.content + if self.descriptions: + mapping["descriptions"] = list(self.descriptions) + return mapping + + @classmethod + def from_mapping(cls, value: Mapping[str, Any]) -> "StandardCategory": + source = value.get("source") or {} + return cls( + category_id=str(value.get("category_id", "") or ""), + name=str(value.get("name", "") or ""), + path=tuple(str(p) for p in value.get("path", ())), + description=str(value.get("description", "") or ""), + code=value.get("code"), + standard_data_level=value.get("standard_data_level"), + raw_level=str(value.get("raw_level", "") or ""), + content=str(value.get("content", "") or ""), + source=SourceRef.from_mapping( + source if isinstance(source, Mapping) else {} + ), + descriptions=tuple(str(d) for d in value.get("descriptions", ())), + ) + + +@dataclass(frozen=True) +class CanonicalStandard: + """The canonical standard for one dataset.""" + + dataset: str + id_strategy: str + standard_source: SourceRef + standard_name: str = "" + categories: tuple[StandardCategory, ...] = () + + def by_id(self) -> dict[str, StandardCategory]: + return {category.category_id: category for category in self.categories} + + def to_mapping(self) -> dict[str, Any]: + return { + "dataset": self.dataset, + "id_strategy": self.id_strategy, + "standard_name": self.standard_name, + "standard_source": self.standard_source.to_mapping(), + "fingerprint": self.fingerprint(), + "categories": [category.to_mapping() for category in self.categories], + } + + @classmethod + def from_mapping(cls, value: Mapping[str, Any]) -> "CanonicalStandard": + raw_categories = value.get("categories", ()) + categories = tuple( + StandardCategory.from_mapping(item) + for item in raw_categories + if isinstance(item, Mapping) + ) + source = value.get("standard_source") or {} + return cls( + dataset=str(value.get("dataset", "") or ""), + id_strategy=str(value.get("id_strategy", "") or ""), + standard_name=str(value.get("standard_name", "") or ""), + standard_source=SourceRef.from_mapping( + source if isinstance(source, Mapping) else {} + ), + categories=categories, + ) + + def fingerprint(self) -> str: + payload = { + "dataset": self.dataset, + "id_strategy": self.id_strategy, + "categories": [ + {k: v for k, v in category.to_mapping().items() if k != "source"} + for category in sorted(self.categories, key=lambda c: c.category_id) + ], + } + digest = hashlib.sha256() + digest.update( + json.dumps(payload, ensure_ascii=False, sort_keys=True).encode("utf-8") + ) + return digest.hexdigest() + + +@dataclass(frozen=True) +class StandardCategoryBuilder: + """Deterministic canonical category from raw standard source fields. + + Kept as a small value object so build logic is trivially testable with + plain dicts (no Excel, no IO). + """ + + category_id: str + name: str + path: tuple[str, ...] + description: str = "" + code: str | None = None + raw_level: str = "" + content: str = "" + source_file: str = "" + source_sheet: str = "" + source_row: int | None = None + + def build(self) -> StandardCategory: + level, _ = normalize_standard_level(self.raw_level) # level kept, raw kept + return StandardCategory( + category_id=self.category_id, + name=self.name, + path=self.path, + description=self.description, + code=self.code, + standard_data_level=level, + raw_level=clean(self.raw_level), + content=self.content, + source=SourceRef( + file=self.source_file, sheet=self.source_sheet, row=self.source_row + ), + ) + + +__all__ = [ + "LEVELS", + "clean", + "compact", + "strip_code", + "normalize_standard_level", + "SourceRef", + "StandardCategory", + "CanonicalStandard", + "StandardCategoryBuilder", +] diff --git a/src/agent/standards/sources.py b/src/agent/standards/sources.py new file mode 100644 index 0000000..6399316 --- /dev/null +++ b/src/agent/standards/sources.py @@ -0,0 +1,197 @@ +"""Raw standard readers (Phase 1): read the ORIGINAL standard workbooks into +plain raw-entry dicts. + +The canonical standard must be built from the original source workbooks +(data/raw), not from the already-compressed standards_map JSON digests +(financial_standards_dict.json / guanji_dict.json dropped real path depth and +grading columns). These readers only extract, never normalize semantics: +grading columns are kept as raw strings. + +Excel layout notes (verified against data/raw at 2026-08-20): +- finance 金融行业数据安全分类分级标准指南.xlsx / sheet "Table 1": + columns 一级子类|二级子类|二级定义|三级子类|三级定义|四级子类|内容|安全级别|备注|部门意见; + merged 一级/二级/三级 subclasses carry forward; "安全级别" header row is + "最低安全级别参考". +- shougang 关基-数据分类分级目录.xlsx / sheet "数据分类分级": + columns 一级分类|一级定义|二级分类|二级定义|三级分类|三级定义|四级分类|四级定义| + 数据资源说明(内容)|分级|数据资源; merged 一级/二级/三级 carry forward. +""" + +from __future__ import annotations + +from dataclasses import dataclass, field +from pathlib import Path +from typing import Any, Mapping + +from agent.standards.contracts import clean + +FINANCE_SHEET = "Table 1" +SHOUGANG_SHEET = "数据分类分级" + + +@dataclass(frozen=True) +class RawEntry: + """One leaf row of a raw standard, with its traceable origin.""" + + level_1: str = "" + level_2: str = "" + level_3: str = "" + leaf: str = "" + description: str = "" + content: str = "" + raw_level: str = "" + resource: str = "" + sheet: str = "" + row: int | None = None + + def to_mapping(self) -> dict[str, Any]: + return { + "level_1": self.level_1, + "level_2": self.level_2, + "level_3": self.level_3, + "leaf": self.leaf, + "description": self.description, + "content": self.content, + "raw_level": self.raw_level, + "resource": self.resource, + "sheet": self.sheet, + "row": self.row, + } + + +@dataclass(frozen=True) +class ReaderResult: + entries: tuple[RawEntry, ...] + issues: tuple[str, ...] = () + + +def _open_xlsx(path: Path, sheet: str): + """Lazily import openpyxl so non-Excel code paths never require it.""" + try: + import openpyxl + except ImportError as exc: # pragma: no cover - optional dependency + raise RuntimeError("reading raw standard workbooks requires openpyxl") from exc + workbook = openpyxl.load_workbook(path, read_only=True, data_only=True) + if sheet not in workbook.sheetnames: + raise ValueError(f"sheet {sheet!r} not found in {path}") + return workbook, workbook[sheet] + + +def read_finance_standard_guide(path: str | Path) -> ReaderResult: + """Extract leaf rows of the finance grading-standard guide.""" + workbook, sheet = _open_xlsx(Path(path), FINANCE_SHEET) + rows = list(sheet.iter_rows(values_only=True)) + workbook.close() + + entries: list[RawEntry] = [] + issues: list[str] = [] + level_1 = level_2 = level_3 = "" + for index, row in enumerate(rows): + if index < 2: # two header rows + continue + if row[1] is not None: + level_1 = clean(row[1]) + if row[2] is not None: + level_2 = clean(row[2]) + if row[4] is not None: + level_3 = clean(row[4]) + if row[6] is None: + continue + leaf = clean(row[6]) + if not leaf: + issues.append(f"finance row {index + 1}: empty leaf") + continue + content = clean(row[7]) if row[7] is not None else "" + raw_level = str(row[8]).strip() if row[8] is not None else "" + entries.append( + RawEntry( + level_1=level_1, + level_2=level_2, + level_3=level_3, + leaf=leaf, + description=content, + raw_level=raw_level, + sheet=FINANCE_SHEET, + row=index + 1, + ) + ) + return ReaderResult(tuple(entries), tuple(issues)) + + +@dataclass(frozen=True) +class _LevelBox: + name: str = "" + definition: str = "" + + +def read_guanji_catalog(path: str | Path) -> ReaderResult: + """Extract leaf rows of the shougang (关基) grading catalog. + + A leaf is the DEEPEST level that carries a real name: the catalog encodes + categories whose leaf sits at the 三级 level with a literal "——" in the + 四级 cell (e.g. 合同归并(B1-2), 合同跟踪(B1-5), 热轧作业计划(B3-3)). + Those rows are real categories, NOT placeholders; their code/name come + from the 三级 cell. The leaf's description is the definition column of + its own level (四级 -> col 8; 三级 -> col 6). Literal cell text is kept + lossless (no quality fixing). + """ + workbook, sheet = _open_xlsx(Path(path), SHOUGANG_SHEET) + rows = list(sheet.iter_rows(values_only=True)) + workbook.close() + + entries: list[RawEntry] = [] + issues: list[str] = [] + level_1 = _LevelBox() + level_2 = _LevelBox() + level_3 = _LevelBox() + for index, row in enumerate(rows): + if index < 2: # title + header rows + continue + if len(row) < 11: + issues.append(f"shougang row {index + 1}: too few columns") + continue + if row[1] is not None: + level_1 = _LevelBox(clean(str(row[1])), clean(str(row[2])) if row[2] is not None else "") + if row[3] is not None: + level_2 = _LevelBox(clean(str(row[3])), clean(str(row[4])) if row[4] is not None else "") + if row[5] is not None: + level_3 = _LevelBox(clean(str(row[5])), clean(str(row[6])) if row[6] is not None else "") + if row[7] is None: + continue + level_4 = _LevelBox(clean(str(row[7])), clean(str(row[8])) if row[8] is not None else "") + # deepest real level is the leaf (non-empty, not the "——" marker) + candidates = [ + (level_4.name, level_4.definition, "level_4"), + (level_3.name, level_3.definition, "level_3"), + (level_2.name, level_2.definition, "level_2"), + ] + leaf_name, leaf_definition, leaf_level = next( + (c for c in candidates if c[0] and c[0] != "——"), + ("", "", ""), + ) + if not leaf_name: + issues.append(f"shougang row {index + 1}: no real leaf level (all ——)") + continue + # keep only levels above/at the leaf; deeper levels are left empty + resolved_level_3 = level_3.name if leaf_level in ("level_3", "level_4") else "" + content = clean(str(row[9])) if row[9] is not None else "" + raw_level = str(row[10]).strip() if row[10] is not None else "" + resource = clean(str(row[11])) if len(row) > 11 and row[11] is not None else "" + entries.append( + RawEntry( + level_1=level_1.name, + level_2=level_2.name, + level_3=resolved_level_3, + leaf=leaf_name, + description=leaf_definition, + content=content, + raw_level=raw_level, + resource=resource, + sheet=SHOUGANG_SHEET, + row=index + 1, + ) + ) + return ReaderResult(tuple(entries), tuple(issues)) + + +__all__ = ["RawEntry", "ReaderResult", "read_finance_standard_guide", "read_guanji_catalog"] diff --git a/tests/standards/test_align.py b/tests/standards/test_align.py new file mode 100644 index 0000000..202b521 --- /dev/null +++ b/tests/standards/test_align.py @@ -0,0 +1,88 @@ +"""Phase 1 canonical standard — sample<->standard alignment tests (hermetic).""" + +from __future__ import annotations + +from agent.standards.align import align_dataset_to_standard +from agent.standards.build import resolve_standard_dataset +from agent.standards.contracts import CanonicalStandard, SourceRef, StandardCategory + + +def _standard(categories: list[StandardCategory]) -> CanonicalStandard: + return CanonicalStandard( + dataset="ds", + id_strategy="code", + standard_source=SourceRef(), + categories=tuple(categories), + ) + + +def _cat(category_id: str, name: str, level: str) -> StandardCategory: + return StandardCategory( + category_id=category_id, name=name, path=(name,), + standard_data_level=level, raw_level=level, + ) + + +def _record(category_id: str | None, level: str, status: str = "resolved"): + record = {"data_level": level, "resolution_status": status} + if category_id is not None: + record["target"] = { + "category_id": category_id, + "leaf_name": category_id, + "category_path": [category_id], + } + return record + + +def test_alignment_buckets(): + standard = _standard( + [ + _cat("A", "A", "L1"), + _cat("B", "B", "L3"), + _cat("C", "C", "L1"), # observed in no sample -> sample_missing + ] + ) + records = [ + _record("A", "L1"), # matched + _record("B", "L2"), # mismatched + _record("D", "L3"), # resolved but standard_missing + _record(None, "L2", "placeholder"), # unresolved + ] + report = align_dataset_to_standard(records, standard) + assert report["sample_counts"]["total"] == 4 + assert report["sample_counts"]["resolved"] == 3 + assert report["sample_counts"]["unresolved"] == 1 + assert report["sample_counts"]["matched"] == 1 + assert report["sample_counts"]["mismatched"] == 1 + assert report["sample_counts"]["standard_missing"] == 1 + assert report["standard_missing_samples"][0]["category_id"] == "D" + assert report["sample_missing_standard_categories"] == ["C"] + + +def test_alignment_never_mutates_sample_level(): + standard = _standard([_cat("A", "A", "L3")]) + records = [_record("A", "L1")] + report = align_dataset_to_standard(records, standard) + assert report["sample_counts"]["mismatched"] == 1 + assert records[0]["data_level"] == "L1" # untouched + + +def test_alignment_standard_level_unavailable_bucket(): + standard = CanonicalStandard( + dataset="ds", id_strategy="path", standard_source=SourceRef(), + categories=( + StandardCategory( + category_id="X", name="X", standard_data_level=None, raw_level="3 4" + ), + ), + ) + report = align_dataset_to_standard([_record("X", "L3")], standard) + assert report["sample_counts"]["standard_level_unavailable"] == 1 + assert report["sample_counts"]["mismatched"] == 0 + + +def test_dataset_standard_routing(): + assert resolve_standard_dataset("finance") == "finance" + assert resolve_standard_dataset("shougang") == "shougang" + assert resolve_standard_dataset("infra") == "shougang" # shared, not a copy + assert resolve_standard_dataset("pers_info") is None # no confirmed standard diff --git a/tests/standards/test_build_finance.py b/tests/standards/test_build_finance.py new file mode 100644 index 0000000..4f9a019 --- /dev/null +++ b/tests/standards/test_build_finance.py @@ -0,0 +1,92 @@ +"""Phase 1 canonical standard — finance build tests (dict fixtures, hermetic). + +Builders accept plain dict entries ({"level_1","level_2","level_3","leaf", +"description","content","raw_level","sheet","row"}) with no Excel dependency. +""" + +from __future__ import annotations + +from agent.standards.build import build_finance_standard + + +def _entry(**kw): + base = dict(sheet="Table 1", row=1, content="", raw_level="") + base.update(kw) + return base + + +def test_finance_lossless_path_keeps_real_depth_and_no_padding(): + entries = [ + _entry( + level_1="客户", level_2="个人", level_3="个人自然信息", + leaf="个人基本概况信息", description="指个人基本情况数据", + raw_level="3", row=3, + ), + _entry( + level_1="业务", level_2="账户信息", level_3="", + leaf="基本信息", description="账户基本信息", raw_level="2", row=40, + ), + ] + standard, report = build_finance_standard( + entries, source_file="data/raw/g.xlsx", source_sheet="Table 1" + ) + by_id = standard.by_id() + full = by_id["finance:客户.个人.个人基本概况信息"] + assert full.path == ("客户", "个人", "个人自然信息", "个人基本概况信息") # 4 real levels + assert full.standard_data_level == "L3" + assert full.description == "指个人基本情况数据" + shallow = by_id["finance:业务.账户信息.基本信息"] + assert shallow.path == ("业务", "账户信息", "基本信息") # empty 三级 omitted, no padding + assert shallow.standard_data_level == "L2" + assert report.issues == [] + + +def test_finance_identity_excludes_level_3_but_path_keeps_it(): + entries = [ + _entry( + level_1="客户", level_2="个人", level_3="个人自然信息", + leaf="个人基本概况信息", raw_level="3", row=3, + ), + _entry( + level_1="客户", level_2="个人", level_3="个人健康生理信息", + leaf="个人基本概况信息", raw_level="4", row=9, + ), + ] + standard, report = build_finance_standard( + entries, source_file="f", source_sheet="Table 1" + ) + # level_3 is provenance only: both rows share the same L1-L2-leaf identity + assert len(standard.categories) == 1 + category = standard.categories[0] + assert category.category_id == "finance:客户.个人.个人基本概况信息" + assert category.path == ("客户", "个人", "个人自然信息", "个人基本概况信息") + assert report.aggregated == {"kinds": 1, "instances": 1} + + +def test_finance_unparseable_levels_reported_not_fixed(): + entries = [ + _entry(level_1="经营管理", level_2="综合管理", level_3="", leaf="市场营销信息(非公开)", raw_level="l", row=150), + _entry(level_1="客户", level_2="个人", level_3="个人基本概况", leaf="个人健康生理影像信息", raw_level="3 4", row=168), + ] + standard, report = build_finance_standard( + entries, source_file="f", source_sheet="Table 1" + ) + for category in standard.categories: + assert category.standard_data_level is None + assert category.raw_level in ("l", "3 4") + kinds = {i.kind for i in report.issues} + assert "level_unparseable" in kinds + assert len(report.issues) == 2 + + +def test_finance_build_deterministic_under_input_shuffle(): + entries = [ + _entry(level_1="客户", level_2="个人", level_3="个人自然信息", leaf="个人基本概况信息", raw_level="3", row=3), + _entry(level_1="业务", level_2="账户信息", level_3="", leaf="基本信息", raw_level="2", row=40), + _entry(level_1="经营管理", level_2="技术管理", level_3="系统管理信息", leaf="配置信息", raw_level="l", row=99), + ] + a, _ = build_finance_standard(list(entries), source_file="f", source_sheet="Table 1") + b, _ = build_finance_standard(list(reversed(entries)), source_file="f", source_sheet="Table 1") + assert a.fingerprint() == b.fingerprint() + assert [c.category_id for c in a.categories] == [c.category_id for c in b.categories] + assert a.to_mapping()["categories"] == b.to_mapping()["categories"] diff --git a/tests/standards/test_build_shougang.py b/tests/standards/test_build_shougang.py new file mode 100644 index 0000000..875df93 --- /dev/null +++ b/tests/standards/test_build_shougang.py @@ -0,0 +1,83 @@ +"""Phase 1 canonical standard — shougang build tests (dict fixtures, hermetic).""" + +from __future__ import annotations + +from agent.standards.build import build_shougang_standard + + +def _entry(**kw): + base = dict(sheet="数据分类分级", row=1, content="", raw_level="", resource="") + base.update(kw) + return base + + +def test_shougang_code_name_path_level(): + entries = [ + _entry( + level_1="研发数据域(A)", level_2="产品研发(A1)", level_3="科研检验(A1-1)", + leaf="科研设备预约管理(A1-1-1)", description="指设备预约", + content="设备预约信息", raw_level="2", row=7, + ) + ] + standard, report = build_shougang_standard( + entries, source_file="data/raw/c.xlsx", source_sheet="数据分类分级" + ) + category = standard.categories[0] + assert category.category_id == "A1-1-1" + assert category.code == "A1-1-1" + assert category.name == "科研设备预约管理" + assert category.path == ("研发数据域", "产品研发", "科研检验", "科研设备预约管理") + assert category.description == "指设备预约" + assert category.content == "设备预约信息" + assert category.standard_data_level == "L2" + assert report.issues == [] + + +def test_shougang_leaf_at_level_three_no_invented_fourth_level(): + # catalog encodes 三级-level leaves with a literal "——" in the 四级 cell + entries = [ + _entry( + level_1="生产数据域(B)", level_2="生产合同(订单)(B1)", + level_3="合同归并(B1-2)", leaf="——", + description="指按照合同加工途径", raw_level="3", row=9, + ), + _entry( + level_1="管理数据域(C)", level_2="生产质量管理(C1)", + level_3="质保书管理(C1-5)", leaf="——", raw_level="1", row=108, + ), + ] + standard, report = build_shougang_standard( + entries, source_file="c", source_sheet="数据分类分级" + ) + by_id = standard.by_id() + merged = by_id["B1-2"] + assert merged.name == "合同归并" + # real hierarchy depth is 3; the "——" marker does NOT become a 4th level + assert merged.path == ("生产数据域", "生产合同(订单)", "合同归并") + assert merged.standard_data_level == "L3" + assert by_id["C1-5"].standard_data_level == "L1" + + +def test_shougang_no_code_real_leaf_reported_and_skipped(): + entries = [ + _entry(level_1="管理数据域(C)", level_2="经营管理(C8)", level_3="综合管理", + leaf="无法归类的真实名称", raw_level="2", row=90), + ] + standard, report = build_shougang_standard( + entries, source_file="c", source_sheet="数据分类分级" + ) + assert len(standard.categories) == 0 + assert any(issue.kind == "no_code" for issue in report.issues) + + +def test_shougang_build_deterministic_under_input_shuffle(): + entries = [ + _entry(level_1="研发数据域(A)", level_2="产品研发(A1)", level_3="科研检验(A1-1)", leaf="科研设备预约管理(A1-1-1)", raw_level="2", row=7), + _entry(level_1="生产数据域(B)", level_2="生产合同(订单)(B1)", level_3="合同归并(B1-2)", leaf="——", raw_level="3", row=9), + _entry(level_1="管理数据域(C)", level_2="生产质量管理(C1)", level_3="检化验管理(C1-4)", leaf="认证管理(C1-4-4)", raw_level="3", row=105), + _entry(level_1="管理数据域(C)", level_2="经营管理(C8)", level_3="无码", leaf="无码类别", raw_level="2", row=90), + ] + a, _ = build_shougang_standard(list(entries), source_file="c", source_sheet="数据分类分级") + b, _ = build_shougang_standard(list(reversed(entries)), source_file="c", source_sheet="数据分类分级") + assert a.fingerprint() == b.fingerprint() + assert [c.category_id for c in a.categories] == [c.category_id for c in b.categories] diff --git a/tests/standards/test_contracts.py b/tests/standards/test_contracts.py new file mode 100644 index 0000000..95b4490 --- /dev/null +++ b/tests/standards/test_contracts.py @@ -0,0 +1,99 @@ +"""Phase 1 canonical standard — contract unit tests (hermetic, no IO / Excel).""" + +from __future__ import annotations + +import copy + +import pytest + +from agent.standards.contracts import ( + CanonicalStandard, + SourceRef, + StandardCategory, + compact, + normalize_standard_level, + strip_code, +) + + +def _category(category_id: str, name: str, level: str | None = None, row: int = 1) -> StandardCategory: + return StandardCategory( + category_id=category_id, + name=name, + path=(name,), + description="", + code=None, + standard_data_level=level, + raw_level=level or "", + source=SourceRef(file="f.xlsx", sheet="s", row=row), + ) + + +def test_normalize_standard_level_valid_forms(): + for raw, expected in { + "1": "L1", "2": "L2", "3": "L3", "4": "L4", + "L1": "L1", "LEVEL3": "L3", "1级": "L1", "4级": "L4", + }.items(): + level, kept = normalize_standard_level(raw) + assert level == expected + assert kept == raw.strip() + assert normalize_standard_level(None) == (None, "") + assert normalize_standard_level("") == (None, "") + + +def test_normalize_standard_level_unparseable_never_guessed(): + # finance guide anomalies: 'l' (typo of 1) and '3 4' (ambiguous) must stay + # None and keep the raw text — the builder reports them, never fixes. + for raw in ("l", "3 4", "敏感级", "高"): + level, kept = normalize_standard_level(raw) + assert level is None + assert kept == raw + + +def test_compact_and_strip_code(): + assert compact("经营 管理 技术") == "经营管理技术" + assert strip_code("科研设备预约管理(A1-1-1)") == "科研设备预约管理" + assert strip_code("生产数据域\n(B)") == "生产数据域" + assert strip_code("基本信息(公开)") == "基本信息(公开)" # parens without code kept + + +def test_standard_round_trip_stable(): + standard = CanonicalStandard( + dataset="finance", + id_strategy="path", + standard_name="指南", + standard_source=SourceRef(file="data/raw/f.xlsx", sheet="Table 1", row=None), + categories=(_category("finance:a.b.c", "c", "L3"), _category("finance:x", "x", "L1")), + ) + mapping = standard.to_mapping() + mapping["categories"] = sorted( + mapping["categories"], key=lambda c: c["category_id"], reverse=True + ) # scramble order + rebuilt = CanonicalStandard.from_mapping(mapping) + assert rebuilt.dataset == standard.dataset + assert rebuilt.id_strategy == standard.id_strategy + assert rebuilt.fingerprint() == standard.fingerprint() + + +def test_round_trip_preserves_level_and_source(): + category = StandardCategory( + category_id="A1-1-1", + name="科研设备预约管理", + path=("研发数据域", "产品研发", "科研检验", "科研设备预约管理"), + description="描述", + code="A1-1-1", + standard_data_level="L3", + raw_level="3", + content="资源说明", + source=SourceRef(file="x.xlsx", sheet="数据分类分级", row=7), + ) + rebuilt = StandardCategory.from_mapping(copy.deepcopy(category.to_mapping())) + assert rebuilt == category + + +def test_fingerprint_independent_of_source_rows(): + a = _category("A", "a", "L1", row=1) + b = _category("A", "a", "L1", row=99) # same content, different source row + sa = CanonicalStandard("s", "code", standard_source=SourceRef(), categories=(a,)) + sb = CanonicalStandard("s", "code", standard_source=SourceRef(), categories=(b,)) + assert sa.fingerprint() == sb.fingerprint() diff --git a/tests/standards/test_real_xlsx.py b/tests/standards/test_real_xlsx.py new file mode 100644 index 0000000..c7b0e21 --- /dev/null +++ b/tests/standards/test_real_xlsx.py @@ -0,0 +1,131 @@ +"""Phase 1 canonical standard — integration tests against real raw workbooks. + +These read the ORIGINAL standard workbooks under data/raw (gitignored) and the +canonical dataset layer. They are skipped when the raw files are absent (CI / +fresh clone) so the suite stays green without the data provider's files. +""" + +from __future__ import annotations + +from pathlib import Path + +import pytest + +from agent.standards.align import align_dataset_to_standard, load_canonical_records +from agent.standards.build import ( + build_finance_standard, + build_shougang_standard, + resolve_standard_dataset, +) +from agent.standards.sources import ( + read_finance_standard_guide, + read_guanji_catalog, +) +from agent.task import LeafRegistry + +ROOT = Path(__file__).resolve().parents[2] +RAW = ROOT / "data" / "raw" +FIN_XLSX = RAW / "金融行业数据安全分类分级标准指南.xlsx" +SHG_XLSX = RAW / "关基-数据分类分级目录.xlsx" +CANON = ROOT / "data" / "canonical" +REG = ROOT / "cfg" / "task" / "registry" + +pytestmark = pytest.mark.skipif( + not (FIN_XLSX.is_file() and SHG_XLSX.is_file()), + reason="raw standard workbooks not present (data/raw is gitignored)", +) + + +@pytest.fixture(scope="module") +def built(): + finance_raw = read_finance_standard_guide(FIN_XLSX) + shougang_raw = read_guanji_catalog(SHG_XLSX) + finance, finance_report = build_finance_standard( + finance_raw.entries, source_file=str(FIN_XLSX), source_sheet="Table 1" + ) + shougang, shougang_report = build_shougang_standard( + shougang_raw.entries, source_file=str(SHG_XLSX), source_sheet="数据分类分级" + ) + return finance, finance_report, shougang, shougang_report + + +def test_finance_standard_matches_registry_identity(built): + finance, _, _, _ = built + registry = LeafRegistry.from_path(REG / "finance.registry.json") + assert len(finance.categories) == 233 + assert {c.category_id for c in finance.categories} == set(registry.ids) + # real hierarchy path depth is preserved (E2E: dict had compressed L1-L2-leaf) + depths = {len(c.path) for c in finance.categories} + assert 4 in depths and 3 in depths + + +def test_finance_unparseable_levels_reported_not_fixed(built): + _, finance_report, _, _ = built + unparseable = [i for i in finance_report.issues if i.kind == "level_unparseable"] + assert len(unparseable) == 2 + assert all("standard_data_level=null (not guessed)" in i.detail for i in unparseable) + + +def test_finance_alignment_reproduces_known_outliers(built): + finance, _, _, _ = built + records = load_canonical_records(CANON / "finance" / "all.json") + report = align_dataset_to_standard(records, finance) + counts = report["sample_counts"] + assert counts["total"] == 568 + assert counts["resolved"] == 531 + assert counts["matched"] == 529 + assert counts["mismatched"] == 2 + assert counts["standard_missing"] == 0 + fields = {m["field"] for m in report["mismatched_samples"]} + assert fields == {"AMONEY", "HXTRADENO"} + + +def test_shougang_standard_covers_registry_plus_lost_b3_6(built): + _, _, shougang, _ = built + registry = LeafRegistry.from_path(REG / "shougang.registry.json") + standard_codes = {c.category_id for c in shougang.categories} + assert len(shougang.categories) == 234 + assert set(registry.ids) <= standard_codes + assert standard_codes - set(registry.ids) == {"B3-6"} # 中厚板作业计划 lost by legacy + + +def test_shougang_alignment_100_percent(built): + _, _, shougang, _ = built + records = load_canonical_records(CANON / "shougang" / "all.json") + report = align_dataset_to_standard(records, shougang) + counts = report["sample_counts"] + assert counts["total"] == 19415 + assert counts["resolved"] == 18393 + assert counts["matched"] == 18393 + assert counts["mismatched"] == 0 + assert counts["standard_missing"] == 0 + + +def test_infra_reuses_shougang_standard(built): + _, _, shougang, _ = built + assert resolve_standard_dataset("infra") == "shougang" + records = load_canonical_records(CANON / "infra" / "all.json") + report = align_dataset_to_standard(records, shougang) + counts = report["sample_counts"] + assert counts["total"] == 64 + assert counts["matched"] == 64 + assert counts["mismatched"] == 0 + + +def test_pers_info_has_no_canonical_standard(): + assert resolve_standard_dataset("pers_info") is None + + +def test_build_deterministic_on_real_inputs(built): + finance, _, shougang, _ = built + # rebuild from the same raw readers — identical fingerprints + f2, _ = build_finance_standard( + read_finance_standard_guide(FIN_XLSX).entries, + source_file=str(FIN_XLSX), source_sheet="Table 1", + ) + s2, _ = build_shougang_standard( + read_guanji_catalog(SHG_XLSX).entries, + source_file=str(SHG_XLSX), source_sheet="数据分类分级", + ) + assert finance.fingerprint() == f2.fingerprint() + assert shougang.fingerprint() == s2.fingerprint() From b20d2a0cd0d1c3082ee202106c25955a92422a7f Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?=E6=9B=BE=E7=AB=8B=E5=AE=8F?= <曾立宏@buaa.edu.cn> Date: Thu, 20 Aug 2026 14:42:00 +0800 Subject: [PATCH 2/4] =?UTF-8?q?fix(standards):=20PR-review=20=E2=80=94=20l?= =?UTF-8?q?ossless=20entry/category=20split,=20checksums,=20deterministic?= =?UTF-8?q?=20order?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Addresses the Phase-1 review blocking items without touching training code. - Blocker 1: one CanonicalStandard entry per real standard row (no fact-layer aggregation). standard_entry_id (finance L1.L2.l3.L4 / shougang code) is the true source identity; category_id is the training/registry alias; 237 finance entries project to 233 training categories via derived training_projection. The 5 same-alias 合约协议 entries are preserved with distinct 三级/path/source. - Blocker 2 (option B): committed script/standard/checksums.json; CLI verifies raw-workbook sha256 before building and refuses on mismatch; restore/checksum flow documented in phase1_canonical_standard.md. - P1: ReaderResult.issues are merged into the build report (reader_issue kind). - P1: build report reports standard_entries_out (237/234) + training_categories (233/234); level distribution over final entries. - P1: determinism now order-independent even for duplicate category_id entries (every entry preserved + sorted by standard_entry_id); tests added. - Non-blocking: alignment unresolved_evidence (status/leaf_name/candidates) for the 37 finance non-trainable samples, evidence-only. - tests/standards 37 passed; full suite 279 passed, 2 skipped (pre-existing). --- .../finance_standard_alignment.json | 226 +- .../provenance/infra_standard_alignment.json | 10 +- .../shougang_standard_alignment.json | 5121 ++++++++++++++++- .../provenance/standard_build_summary.json | 54 +- docs/design/phase1_canonical_standard.md | 147 +- script/standard/checksums.json | 4 + script/standard/cli.py | 153 +- src/agent/standards/align.py | 110 +- src/agent/standards/build.py | 185 +- src/agent/standards/contracts.py | 79 +- tests/standards/test_align.py | 57 +- tests/standards/test_build_finance.py | 77 +- tests/standards/test_build_shougang.py | 33 +- tests/standards/test_checksum.py | 51 + tests/standards/test_contracts.py | 55 +- tests/standards/test_real_xlsx.py | 56 +- 16 files changed, 6027 insertions(+), 391 deletions(-) create mode 100644 script/standard/checksums.json create mode 100644 tests/standards/test_checksum.py diff --git a/artifacts/generated/provenance/finance_standard_alignment.json b/artifacts/generated/provenance/finance_standard_alignment.json index 5c9217b..3fb623f 100644 --- a/artifacts/generated/provenance/finance_standard_alignment.json +++ b/artifacts/generated/provenance/finance_standard_alignment.json @@ -11,9 +11,15 @@ "交易通用信息", "交易基本信息" ], - "source_row": 133, - "standard_level": "L2", - "standard_raw_level": "2" + "source_rows": [ + 133 + ], + "standard_entry_ids": [ + "finance:业务.交易信息.交易通用信息.交易基本信息" + ], + "standard_levels": [ + "L2" + ] }, { "category_id": "finance:业务.账户信息.基本信息", @@ -25,9 +31,15 @@ "账户信息", "基本信息" ], - "source_row": 51, - "standard_level": "L2", - "standard_raw_level": "2" + "source_rows": [ + 51 + ], + "standard_entry_ids": [ + "finance:业务.账户信息.单位标签信息.基本信息" + ], + "standard_levels": [ + "L2" + ] } ], "resolved_match_rate": 0.9962, @@ -256,13 +268,207 @@ "finance:经营管理.风险管理信息.黑名单信息" ], "standard": "finance", - "standard_categories": 233, - "standard_categories_observed": 20, - "standard_categories_unobserved": 213, + "standard_entries": 237, "standard_level_unavailable_samples": [], "standard_missing_samples": [], + "training_categories": 233, + "training_categories_observed": 20, + "training_categories_unobserved": 213, "unresolved_by_status": { "missing_leaf": 34, "path_mismatch": 3 - } + }, + "unresolved_evidence": [ + { + "candidate_standard_categories": [], + "leaf_name": "交易清金额信息", + "status": "missing_leaf" + }, + { + "candidate_standard_categories": [], + "leaf_name": "单位基本信息", + "status": "missing_leaf" + }, + { + "candidate_standard_categories": [], + "leaf_name": "单位基本信息", + "status": "missing_leaf" + }, + { + "candidate_standard_categories": [], + "leaf_name": "单位基本信息", + "status": "missing_leaf" + }, + { + "candidate_standard_categories": [], + "leaf_name": "单位基本情况", + "status": "missing_leaf" + }, + { + "candidate_standard_categories": [], + "leaf_name": "单位基本情况", + "status": "missing_leaf" + }, + { + "candidate_standard_categories": [], + "leaf_name": "单位基本情况", + "status": "missing_leaf" + }, + { + "candidate_standard_categories": [], + "leaf_name": "单位基本情况", + "status": "missing_leaf" + }, + { + "candidate_standard_categories": [], + "leaf_name": "单位基本情况", + "status": "missing_leaf" + }, + { + "candidate_standard_categories": [], + "leaf_name": "单位基本情况", + "status": "missing_leaf" + }, + { + "candidate_standard_categories": [], + "leaf_name": "单位基本情况", + "status": "missing_leaf" + }, + { + "candidate_standard_categories": [], + "leaf_name": "单位基本情况", + "status": "missing_leaf" + }, + { + "candidate_standard_categories": [], + "leaf_name": "单位基本情况", + "status": "missing_leaf" + }, + { + "candidate_standard_categories": [], + "leaf_name": "单位基本情况", + "status": "missing_leaf" + }, + { + "candidate_standard_categories": [], + "leaf_name": "单位基本情况", + "status": "missing_leaf" + }, + { + "candidate_standard_categories": [], + "leaf_name": "单位基本情况", + "status": "missing_leaf" + }, + { + "candidate_standard_categories": [], + "leaf_name": "单位基本情况", + "status": "missing_leaf" + }, + { + "candidate_standard_categories": [], + "leaf_name": "单位基本情况", + "status": "missing_leaf" + }, + { + "candidate_standard_categories": [], + "leaf_name": "单位基本情况", + "status": "missing_leaf" + }, + { + "candidate_standard_categories": [], + "leaf_name": "单位基本情况", + "status": "missing_leaf" + }, + { + "candidate_standard_categories": [], + "leaf_name": "单位基本情况", + "status": "missing_leaf" + }, + { + "candidate_standard_categories": [], + "leaf_name": "单位基本情况", + "status": "missing_leaf" + }, + { + "candidate_standard_categories": [], + "leaf_name": "单位基本情况", + "status": "missing_leaf" + }, + { + "candidate_standard_categories": [], + "leaf_name": "单位联系人信息", + "status": "missing_leaf" + }, + { + "candidate_standard_categories": [], + "leaf_name": "单位联系人信息", + "status": "missing_leaf" + }, + { + "candidate_standard_categories": [], + "leaf_name": "单位联系人信息", + "status": "missing_leaf" + }, + { + "candidate_standard_categories": [], + "leaf_name": "单位联系人信息", + "status": "missing_leaf" + }, + { + "candidate_standard_categories": [], + "leaf_name": "单位联系人信息", + "status": "missing_leaf" + }, + { + "candidate_standard_categories": [], + "leaf_name": "单位联系人信息", + "status": "missing_leaf" + }, + { + "candidate_standard_categories": [], + "leaf_name": "单位联系人信息", + "status": "missing_leaf" + }, + { + "candidate_standard_categories": [], + "leaf_name": "单位联系人信息", + "status": "missing_leaf" + }, + { + "candidate_standard_categories": [], + "leaf_name": "单位联系人信息", + "status": "missing_leaf" + }, + { + "candidate_standard_categories": [], + "leaf_name": "基本信息(公开", + "status": "missing_leaf" + }, + { + "candidate_standard_categories": [], + "leaf_name": "基本信息(公开", + "status": "missing_leaf" + }, + { + "candidate_standard_categories": [ + "finance:业务.交易信息.交易清结算信息" + ], + "leaf_name": "交易清结算信息", + "status": "path_mismatch" + }, + { + "candidate_standard_categories": [ + "finance:经营管理.运营管理.网络服务信息" + ], + "leaf_name": "网络服务信息", + "status": "path_mismatch" + }, + { + "candidate_standard_categories": [ + "finance:经营管理.运营管理.网络服务信息" + ], + "leaf_name": "网络服务信息", + "status": "path_mismatch" + } + ] } diff --git a/artifacts/generated/provenance/infra_standard_alignment.json b/artifacts/generated/provenance/infra_standard_alignment.json index 1a8edee..03b2695 100644 --- a/artifacts/generated/provenance/infra_standard_alignment.json +++ b/artifacts/generated/provenance/infra_standard_alignment.json @@ -243,10 +243,12 @@ "C8-9-1" ], "standard": "shougang", - "standard_categories": 234, - "standard_categories_observed": 4, - "standard_categories_unobserved": 230, + "standard_entries": 234, "standard_level_unavailable_samples": [], "standard_missing_samples": [], - "unresolved_by_status": {} + "training_categories": 234, + "training_categories_observed": 4, + "training_categories_unobserved": 230, + "unresolved_by_status": {}, + "unresolved_evidence": [] } diff --git a/artifacts/generated/provenance/shougang_standard_alignment.json b/artifacts/generated/provenance/shougang_standard_alignment.json index 1d639fa..38473ca 100644 --- a/artifacts/generated/provenance/shougang_standard_alignment.json +++ b/artifacts/generated/provenance/shougang_standard_alignment.json @@ -55,12 +55,5125 @@ "C8-8-8" ], "standard": "shougang", - "standard_categories": 234, - "standard_categories_observed": 192, - "standard_categories_unobserved": 42, + "standard_entries": 234, "standard_level_unavailable_samples": [], "standard_missing_samples": [], + "training_categories": 234, + "training_categories_observed": 192, + "training_categories_unobserved": 42, "unresolved_by_status": { "placeholder": 1022 - } + }, + "unresolved_evidence": [ + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + } + ] } diff --git a/artifacts/generated/provenance/standard_build_summary.json b/artifacts/generated/provenance/standard_build_summary.json index 7ddd026..b6d8b9a 100644 --- a/artifacts/generated/provenance/standard_build_summary.json +++ b/artifacts/generated/provenance/standard_build_summary.json @@ -3,9 +3,9 @@ "finance": { "legacy_dict_entries": 237, "legacy_dict_path_compression": { - "categories_at_legacy_depth": 1, - "categories_with_path_deeper_than_legacy_L1_L2_leaf": 232, - "note": "legacy financial_standards_dict stored L1-L2-leaf identity strings; the real standard has 三级子类 provenance nodes that the legacy digest dropped" + "entries_at_legacy_depth": 1, + "entries_with_path_deeper_than_legacy_L1_L2_leaf": 236, + "note": "legacy financial_standards_dict stored L1-L2-leaf identity strings; the real standard has 三级子类 provenance nodes that the legacy digest dropped (canonical standard keeps every entry, standard_entry_id includes the real 三级)" }, "legacy_unparseable_level_values": [ "3 4级", @@ -14,12 +14,12 @@ "standard_vs_registry_ids": { "missing_from_registry": 0, "registry_ids": 233, - "standard_ids": 233 + "standard_training_categories": 233 } }, "shougang": { "legacy_dict_losses": { - "catalog_categories": 234, + "catalog_entries": 234, "legacy_dict_entries": 234, "note": "guanji_dict kept only 'name(code)' + description: real hierarchy path and the 分级 column were dropped (registry path was [] and no class field existed)", "with_grading_restored": 234, @@ -54,11 +54,6 @@ "standard_missing": 0 }, "build": { - "aggregated": { - "instances": 4, - "kinds": 1 - }, - "categories_out": 237, "dataset": "finance", "entries_read": 237, "id_strategy": "path", @@ -75,17 +70,20 @@ "level_distribution": { "''": 2, "L1": 11, - "L2": 153, + "L2": 157, "L3": 61, "L4": 6 }, "source_file": "data/raw/金融行业数据安全分类分级标准指南.xlsx", "source_sheet": "Table 1", - "standard_name": "金融行业数据安全分类分级标准指南" + "standard_entries_out": 237, + "standard_name": "金融行业数据安全分类分级标准指南", + "training_categories": 233 }, - "categories": 233, + "standard_entries": 237, "standard_source": "finance", - "status": "built" + "status": "built", + "training_categories": 233 }, "infra": { "alignment_headline": { @@ -97,11 +95,6 @@ "standard_missing": 0 }, "build": { - "aggregated": { - "instances": 0, - "kinds": 0 - }, - "categories_out": 234, "dataset": "shougang", "entries_read": 234, "id_strategy": "code", @@ -113,11 +106,14 @@ }, "source_file": "data/raw/关基-数据分类分级目录.xlsx", "source_sheet": "数据分类分级", - "standard_name": "首钢京唐数据分类分级目录(关基)" + "standard_entries_out": 234, + "standard_name": "首钢京唐数据分类分级目录(关基)", + "training_categories": 234 }, - "categories": 234, + "standard_entries": 234, "standard_source": "shougang", - "status": "built" + "status": "built", + "training_categories": 234 }, "shougang": { "alignment_headline": { @@ -129,11 +125,6 @@ "standard_missing": 0 }, "build": { - "aggregated": { - "instances": 0, - "kinds": 0 - }, - "categories_out": 234, "dataset": "shougang", "entries_read": 234, "id_strategy": "code", @@ -145,11 +136,14 @@ }, "source_file": "data/raw/关基-数据分类分级目录.xlsx", "source_sheet": "数据分类分级", - "standard_name": "首钢京唐数据分类分级目录(关基)" + "standard_entries_out": 234, + "standard_name": "首钢京唐数据分类分级目录(关基)", + "training_categories": 234 }, - "categories": 234, + "standard_entries": 234, "standard_source": "shougang", - "status": "built" + "status": "built", + "training_categories": 234 } } } diff --git a/docs/design/phase1_canonical_standard.md b/docs/design/phase1_canonical_standard.md index e323f32..25ff963 100644 --- a/docs/design/phase1_canonical_standard.md +++ b/docs/design/phase1_canonical_standard.md @@ -15,36 +15,45 @@ sample `data_level`。 ## 1. canonical standard schema -每个标准 category(`data/standards/.standard.json` 内): +每个 standard entry(`data/standards/.standard.json` 内,`entries[]`, +一条原始标准行 = 一个 entry,**事实层不做任何聚合**): ```jsonc { - "category_id": "A1-1-1 | finance:客户.个人.个人基本概况信息", - "name": "科研设备预约管理", - "path": ["研发数据域", "产品研发", "科研检验", "科研设备预约管理"], // 真实源层级深度,空层省略,不人为补空层 - "description": "...", // 该叶子所在层级的定义说明(原样保留) - "code": "A1-1-1" | null, - "standard_data_level": "L1|L2|L3|L4|null", // 规范化的标准等级;无法解析时为 null - "raw_level": "2 | l | 3 4", // 原始值,审计用 - "content": "…数据资源说明…", // 可选额外标准文本 - "source": {"file": "…", "sheet": "…", "row": …} // 可追溯到原始标准文件+行号 + "standard_entry_id": "finance:业务.合约协议.贷款业务信息.基本信息 | A1-1-1", + "category_id": "finance:业务.合约协议.基本信息 | A1-1-1", + "name": "基本信息", + "path": ["业务", "合约协议", "贷款业务信息", "基本信息"], // 真实源层级深度,空层省略 + "description": "…", + "code": null | "A1-1-1", + "standard_data_level": "L1|L2|L3|L4|null", + "raw_level": "2 | l | 3 4", + "content": "…数据资源说明…", + "source": {"file": "…", "sheet": "…", "row": …} } ``` 顶层:`dataset / id_strategy / standard_name / standard_source{file,sheet} / -fingerprint` + `categories[]`。`fingerprint` 为 `categories` 内容(不含 -source 行号)的 sha256——同输入确定性一致,与输入顺序/行号无关。 - -`category_id` 延续现有稳定 identity:finance = `finance:{L1}.{L2}.{L4}` -(`level_3` 仅 provenance,与 `DatasetConfig.identity_fields` 一致);shougang -= guanji code。已验证:finance 233 个标准 id == 现有 registry 233 个 id,零差异。 +fingerprint / entries[] / training_projection`。 + +- **`standard_entry_id`** = 原始标准中的**真实身份**(finance 含真实三级子类 + `finance:{L1}.{L2}.{L3}.{L4}`;shougang = guanji code)。唯一。 +- **`category_id`** = 当前训练/registry 兼容 alias(finance = L1-L2-leaf,与 + `DatasetConfig.identity_fields` 一致)。**不唯一**:多个标准 entry 可投影到同一 + 训练类别。 +- **`training_projection`** = 派生视图 `{category_id: [standard_entry_id…]}`, + 把 237 个 finance entry 投影到 233 个训练类别(237→233 是显式投影,**不是** + 事实层丢信息)。 +- `fingerprint` = entries(不含 source 行号)+ projection 内容的 sha256; + **与输入顺序无关**(每个 entry 原样保留、按 `standard_entry_id` 排序, + 不再依赖"首见顺序")。 ## 2. 各数据集 standard source 状态 | dataset | standard_source | 事实源文件 | 状态 | | --- | --- | --- | --- | -| finance | `finance` | `data/raw/金融行业数据安全分类分级标准指南.xlsx`(sheet Table 1) | built(233 类) | -| shougang | `shougang` | `data/raw/关基-数据分类分级目录.xlsx`(sheet 数据分类分级) | built(234 类) | +| finance | `finance` | `data/raw/金融行业数据安全分类分级标准指南.xlsx`(sheet Table 1) | built(**237 entries / 233 training categories**) | +| shougang | `shougang` | `data/raw/关基-数据分类分级目录.xlsx`(sheet 数据分类分级) | built(234 entries) | | infra | `shougang`(复用,不复制维护另一套) | — | 复用共享标准(64/64 对齐) | | pers_info | `null`(missing / unknown) | 无已确认标准 | **不生成虚假 standard_data_level** | @@ -52,7 +61,7 @@ pers_info:仓库内无确认的分类分级标准;18 类 registry 维持当 dataset-derived 行为,但**不伪装成 canonical standard**——不生成 `pers_info.standard.json`,不在 summary 中编造等级。 -## 3. sample ↔ standard 对齐统计(严格 category identity 连接) +## 3. sample ↔ standard 对齐统计(严格 category alias 连接) | dataset | total | resolved | matched | mismatched | standard_missing | unresolved | resolved_match_rate | | --- | --- | --- | --- | --- | --- | --- | --- | @@ -61,75 +70,67 @@ dataset-derived 行为,但**不伪装成 canonical standard**——不生成 | infra | 64 | 64 | 64 | 0 | 0 | 0 | 100% | | pers_info | — | — | — | — | — | — | 无标准 | -- finance 未解析 37 = `missing_leaf 34 + path_mismatch 3`(即既有 canonical - resolution 认定的 37 条非训练样本,类别不在 registry 中;本阶段不改)。标准覆盖 - 233 类中 213 类未被任何样本观测到(universe ≫ 观测叶)。 -- shougang 未解析 1,022 = 数据侧 `level_4='——'` 占位样本(与目录中的 `——` - 不同义,见 §4)。 +- finance 未解析 37 条附 **`unresolved_evidence`**(evidence-only):每条含 + `{status, leaf_name, candidate_standard_categories[]}`(同名校对候选,不修复); + 其中 `missing_leaf 34 + path_mismatch 3`。 +- 对齐支持多 entry 类别:sample 等级命中该类别任一 entry 的等级即 matched。 - 对齐只读:不修改 sample `data_level`,不自动修标签。 ## 4. 发现的数据异常(只报告,不修复) 1. **finance 原始标准 2 处不可解析等级**:row 150 `市场营销信息(公开)` 原始值 - `l`(疑似 `1` 的笔误)、row 168 `客户及监管相关音影像信息` 原始值 `3 4` - (歧义)。→ `standard_data_level=null` + build issue,不做猜测。 -2. **shougang 目录中"三级即叶子"层级**:10 行使 四列为文字 `——`、叶子码在 三级 - (B1-2 合同归并、B1-5 合同跟踪、B3-3/4/5、B4-2、B6-2、C1-5、C4-1)。它们 - **不是占位符**,是真实类别;canonical standard 已按真实深度 3 层保存 - (不发明第 4 层)。这是 Phase 0 规则 7 的直接实例。 -3. **数据侧 `——` 与目录侧 `——` 语义不同**:shougang 样本的 `level_4='——'` - 是不可训练占位标签;目录中的 `——` 是"该层无子结点"标记。二者分别处理, - 不相干。 -4. **legacy 信息损失(dict vs 原始标准)**: - - finance:`financial_standards_dict.json` 把真实 4 层 path 压成 - L1-L2-leaf(232/233 类丢了 三级子类 provenance 层);`l`/`3 4` 原始值在 - dict 中保留为 `l级`/`3 4级`。 - - shougang:`guanji_dict.json` 只保留 `name(code)`+描述,**丢了全部 path** - (registry path 为空)**和分级列**(无 class);并**漏掉 B3-6 中厚板作业计划** - (目录中真实存在,registry 233 vs 标准 234)。 + `l`、row 168 `客户及监管相关音影像信息` 原始值 `3 4`。→ `standard_data_level=null` + + build issue,不做猜测。 +2. **shougang 目录中"三级即叶子"层级**:10 行 四列为 `——`、叶子码在 三级 + (合同归并 B1-2、合同跟踪 B1-5 等)。是真实类别,已按真实深度 3 层保存 + (不发明第 4 层)。 +3. **数据侧 `——` 与目录侧 `——` 语义不同**:样本 `level_4='——'` 是不可训练 + 占位;目录 `——` 是"该层无子结点"。分别处理。 +4. **legacy 信息损失**:finance dict 压平 4 层 path(232/233 类丢 三级 provenance + 层);shougang dict 丢全部 path+分级,并**漏掉 B3-6 中厚板作业计划**。 + canonical standard 均已恢复。 +5. **finance 5 条同训练类别的标准 entry**(业务/合约协议/基本信息 下的 合同通用/ + 贷款业务/中间业务/资金业务/其他支付业务)全部保留,等级一致(2),经 + `training_projection` 投影为 1 个训练类别——这是低损事实层与训练投影的边界, + 不是 bug。 ## 5. 下一阶段(registry/corpus 接口变化) -当前:`raw standard(Excel) →(lossy) standards_map JSON → canonical_corpus → -LeafRegistry + Corpus`(`financial_standards_dict.json`、`guanji_dict.json`、 -`src/agent/task/canonical_corpus.py` 均为**有损中间层**)。 +当前:`raw standard(Excel) → canonical standard(无损,本文档)→(下一步)LeafRegistry + Corpus`。 -建议迁移为: -``` -raw standard(Excel) - → canonical standard(本项目,无损:path/description/code/standard_data_level) - → LeafRegistry + Corpus -``` -为此后续需修改的接口(本阶段不实现,只列出): -1. `src/agent/task/canonical_corpus.py` 的 `parse_financial_standard` / - `parse_guanji_standard` 改为消费 `CanonicalStandard`(或标志位切换数据源), - `path` 使用标准真实深度而非补空层。 -2. registry 生成在钳制当前 233(finance)/233(shougang)兼容的同时, - 决策是否收养标准的第 234 类 B3-6(当前数据 0 样本,仅宇宙完整性问题)。 -3. `corpus_to_mapping` 是否附带 `standard_data_level` 作为 Stage2 知识—— - **属于下一步 task contract 决策**(Phase 0 §5:形态 A vs B),本阶段不预置。 -4. 现有 dict 产物降级为 `legacy/derived`:审计对照用,不再作事实源。 +下一步需明确: +1. `src/agent/task/canonical_corpus.py` 改消费 `CanonicalStandard`:按 + `category_id`(training 投影)建 LeafRegistry,`path` 用标准真实深度; + 233 类保留、B3-6 是否收养(当前数据 0 样本,纯宇宙完整性)。 +2. `corpus_to_mapping` 是否附带 `standard_data_level`/`standard_entry_id` + 作为 Stage2 知识——属任务契约决策(Phase 0 §5 A/B),本阶段不预置。 +3. 现有 dict 产物降级为 legacy/derived 审计对照,不再作事实源。 ## 6. 测试 ``` tests/standards/ - test_contracts.py round-trip / normalize / fragment / determinism - test_build_finance.py 无损 path、identity 规则、异常上报、确定性 - test_build_shougang.py code/name/path/level、三级叶、no_code、确定性 - test_align.py 对齐桶、不修改样本、路由(infra→shougang、pers_info→None) - test_real_xlsx.py 真实 Excel + canonical 集成断言(raw 缺失时 skip) + test_contracts.py round-trip / normalize / fingerprint / projection + test_build_finance.py 无损 237、三级叶保持、level 异常、同类别多 entry determinism、reader issues + test_build_shougang.py code/path/level、三级叶、no_code、reader issues、确定性 + test_align.py 对齐桶、多 entry 类别、不修改样本、unresolved evidence、路由 + test_checksum.py checksum manifest 校验(缺失/不符) + test_real_xlsx.py 真实 Excel + canonical 集成断言(raw 缺失时 skip) ``` -结果:`pytest tests/standards` → **26 passed**(raw 文件存在时集成用例全跑); -全仓 pytest 见 PR 附注(本阶段无训练链路改动,回归为预防性)。 +结果:`pytest tests/standards` → **37 passed**;全仓见 PR。 产物可重生成:`python -m script.standard.cli`(拒绝无 `--overwrite` 覆盖; -先行全部构建/对齐、后写盘;输出 sort_keys + 类别按 id 排序,重复构建字节级一致)。 -raw Excel 仍是唯一事实源,但不进入任何训练代码依赖。 - -**git 边界**:`data/standards/*.standard.json` 被 `.gitignore` 的 `/data/*` 排除, -与 `data/processed`、`data/canonical` 同属**可再生层**(依赖 `data/raw` 恢复); -入库的是 `src/agent/standards/`、`script/standard/`、`tests/standards/`、 -`docs/design/phase1_canonical_standard.md`(+ Phase 0 的 `data_level_design.md`) -以及 `artifacts/generated/provenance/`(对齐审计 + 构建 summary,未忽略)。 +先行全部构建/对齐、后写盘;重复构建字节级一致)。 + +### restore / 分发(Blocker-2 选型:B) + +- **事实源不进入 Git**:raw workbook 属于数据提供方(含首钢内部目录),gitignored。 +- **受控恢复流程**:从数据提供方/私有 artifact 取回两张表到 `data/raw/` 后,CLI + 先按 `script/standard/checksums.json`(已入库)校验 sha256,不符即拒绝构建 + (`--skip-checksum` 供离线调试显式绕过)。 +- **git 边界**:`data/standards/*.standard.json` 仍被 `/data/*` 排除(可再生层, + 与 processed/canonical 一致);入库的是 `src/agent/standards/`、`script/standard/` + (含 `checksums.json`)、`tests/standards/`、`docs/design/*.md` 与 + `artifacts/generated/provenance/`。fresh clone 后按 restore 流程即可重建完全一致的 + 事实层(fingerprint 可核对)。 diff --git a/script/standard/checksums.json b/script/standard/checksums.json new file mode 100644 index 0000000..f9460dd --- /dev/null +++ b/script/standard/checksums.json @@ -0,0 +1,4 @@ +{ + "data/raw/金融行业数据安全分类分级标准指南.xlsx": "8af7e04bf6928ad1740e5370e43d61a3b26c0677ead2525cf4198e2c43d54219", + "data/raw/关基-数据分类分级目录.xlsx": "32e00ec50573b8de310313486e8acd4d09e7614d1a2195f01c55ec7604d9d078" +} diff --git a/script/standard/cli.py b/script/standard/cli.py index 391b497..61fd7f3 100644 --- a/script/standard/cli.py +++ b/script/standard/cli.py @@ -1,10 +1,11 @@ """Phase 1 canonical standard CLI: build standards + alignment artifacts. Usage: - python -m script.standard.cli [--overwrite] + python -m script.standard.cli [--overwrite] [--skip-checksum] -Reads the ORIGINAL standard workbooks (data/raw) and the canonical dataset -records (data/canonical), then writes: +Reads the ORIGINAL standard workbooks (data/raw), verifies their sha256 +against the committed manifest (script/standard/checksums.json), and builds +the canonical standards + sample<->standard alignment: data/standards/finance.standard.json data/standards/shougang.standard.json @@ -13,15 +14,22 @@ artifacts/generated/provenance/infra_standard_alignment.json artifacts/generated/provenance/standard_build_summary.json +Distribution (option B): raw workbooks are gitignored and must be restored +from the data provider first (see docs/design/phase1_canonical_standard.md +"restore" section). The CLI refuses to build on a missing file and on a +checksum mismatch (silently building from a wrong file would corrupt the +standard); --skip-checksum overrides the latter for offline tweaks. + Fail-fast: every dataset is read + built + aligned before anything is written; all outputs are written exactly once. Deterministic: JSON is written -with sort_keys and categories/entries are pre-sorted; no timestamps or -machine-local paths. Raw workbooks are never training dependencies. +with sort_keys and entries are pre-sorted by standard_entry_id; no timestamps +or machine-local paths. Raw workbooks are never training dependencies. """ from __future__ import annotations import argparse +import hashlib import json from pathlib import Path import sys @@ -45,6 +53,7 @@ DEFAULT_CANONICAL_DIR = PROJECT_ROOT / "data" / "canonical" DEFAULT_STANDARD_DIR = PROJECT_ROOT / "data" / "standards" DEFAULT_ARTIFACT_DIR = PROJECT_ROOT / "artifacts" / "generated" / "provenance" +CHECKSUM_MANIFEST = Path(__file__).with_name("checksums.json") FINANCE_XLSX = "金融行业数据安全分类分级标准指南.xlsx" SHOUGANG_XLSX = "关基-数据分类分级目录.xlsx" @@ -64,6 +73,43 @@ def _write_json(payload, path: Path) -> None: handle.write("\n") +def _sha256(path: Path) -> str: + digest = hashlib.sha256() + with path.open("rb") as handle: + for chunk in iter(lambda: handle.read(1 << 20), b""): + digest.update(chunk) + return digest.hexdigest() + + +def _verify_checksums( + paths: dict[str, Path], + manifest_path: Path, + *, + allow_skip: bool, +) -> None: + if not manifest_path.is_file(): + if not allow_skip: + raise FileNotFoundError( + f"checksum manifest not found: {manifest_path} " + "(pass --skip-checksum to proceed without verification)" + ) + return + with manifest_path.open(encoding="utf-8") as handle: + manifest = json.load(handle) + for rel, path in paths.items(): + expected_raw = manifest.get(rel) + if not expected_raw: + continue # file not covered by manifest + actual = _sha256(path) + if actual != expected_raw: + raise ValueError( + f"sha256 mismatch for {rel}: got {actual[:16]}… expected " + f"{expected_raw[:16]}… — refusing to build from a file that " + "differs from the recorded fact source " + "(pass --skip-checksum only if you know what you are doing)" + ) + + def _registry_ids(dataset: str) -> set[str]: from agent.task import LeafRegistry @@ -79,17 +125,15 @@ def _legacy_information_loss( standard canonical build (path depth, grading, missing codes).""" import json as _json - import re as _re finance_out: dict[str, object] = {} finance_ids = {c.category_id for c in finance_standard.categories} finance_reg_ids = _registry_ids("finance") finance_out["standard_vs_registry_ids"] = { - "standard_ids": len(finance_ids), + "standard_training_categories": len(finance_ids), "registry_ids": len(finance_reg_ids), "missing_from_registry": len(finance_ids - finance_reg_ids), } - # legacy dict: how many categories lost a real path level legacy = _json.load( ( PROJECT_ROOT / "data" / "knowledge" / "standards_map" @@ -98,18 +142,19 @@ def _legacy_information_loss( ) depth_lost = 0 depth_kept = 0 - for category in finance_standard.categories: + for entry in finance_standard.categories: segments = 3 # legacy dict identity was L1-L2-leaf - if len(category.path) > segments: + if len(entry.path) > segments: depth_lost += 1 else: depth_kept += 1 finance_out["legacy_dict_path_compression"] = { - "categories_with_path_deeper_than_legacy_L1_L2_leaf": depth_lost, - "categories_at_legacy_depth": depth_kept, + "entries_with_path_deeper_than_legacy_L1_L2_leaf": depth_lost, + "entries_at_legacy_depth": depth_kept, "note": "legacy financial_standards_dict stored L1-L2-leaf identity " "strings; the real standard has 三级子类 provenance nodes that the " - "legacy digest dropped", + "legacy digest dropped (canonical standard keeps every entry, " + "standard_entry_id includes the real 三级)", } finance_out["legacy_dict_entries"] = len(legacy) finance_out["legacy_unparseable_level_values"] = sorted( @@ -135,13 +180,13 @@ def _legacy_information_loss( ) codes_without_path = 0 codes_with_level = 0 - for category in shougang_standard.categories: - if category.path: - codes_without_path += 1 # path restored where legacy had none - if category.standard_data_level: - codes_with_level += 1 # grading restored where legacy had none + for entry in shougang_standard.categories: + if entry.path: + codes_without_path += 1 + if entry.standard_data_level: + codes_with_level += 1 shougang_out["legacy_dict_losses"] = { - "catalog_categories": len(shougang_standard.categories), + "catalog_entries": len(shougang_standard.categories), "legacy_dict_entries": len(legacy_g), "with_real_path_restored": codes_without_path, "with_grading_restored": codes_with_level, @@ -159,6 +204,12 @@ def main(argv: list[str] | None = None) -> int: parser.add_argument("--standard-dir", type=Path, default=DEFAULT_STANDARD_DIR) parser.add_argument("--artifact-dir", type=Path, default=DEFAULT_ARTIFACT_DIR) parser.add_argument("--overwrite", action="store_true") + parser.add_argument( + "--skip-checksum", + action="store_true", + help="build even when the checksum manifest is missing or a file " + "mismatches (use only for offline tweaks)", + ) args = parser.parse_args(argv) raw_dir = Path(args.raw_dir) @@ -167,10 +218,16 @@ def main(argv: list[str] | None = None) -> int: missing = [p for p in (finance_xlsx, shougang_xlsx) if not p.is_file()] if missing: raise FileNotFoundError( - "raw standard workbook(s) missing (data/raw is gitignored; restore " - "them from the data provider before building): " + "raw standard workbook(s) missing; restore them from the data " + "provider first (see docs/design/phase1_canonical_standard.md, " + "'restore' section). Missing: " + ", ".join(str(p) for p in missing) ) + _verify_checksums( + {FINANCE_XLSX: finance_xlsx, SHOUGANG_XLSX: shougang_xlsx}, + CHECKSUM_MANIFEST, + allow_skip=args.skip_checksum, + ) # 1. read + build + align EVERYTHING (pure computation, no writes) finance_raw = read_finance_standard_guide(finance_xlsx) @@ -179,20 +236,19 @@ def main(argv: list[str] | None = None) -> int: finance_raw.entries, source_file=_repo_relative(finance_xlsx), source_sheet="Table 1", + reader_issues=finance_raw.issues, ) shougang_standard, shougang_report = build_shougang_standard( shougang_raw.entries, source_file=_repo_relative(shougang_xlsx), source_sheet="数据分类分级", + reader_issues=shougang_raw.issues, ) - canonical_records = {} - for dataset in ("finance", "shougang", "infra"): - canonical_records[dataset] = load_canonical_records( - args.canonical_dir / dataset / "all.json" - ) - - standard_key = resolve_standard_dataset("infra") # -> shougang (shared) + canonical_records = { + dataset: load_canonical_records(args.canonical_dir / dataset / "all.json") + for dataset in ("finance", "shougang", "infra") + } alignments = { "finance": align_dataset_to_standard( canonical_records["finance"], finance_standard @@ -204,7 +260,6 @@ def main(argv: list[str] | None = None) -> int: canonical_records["infra"], shougang_standard ), } - loss = _legacy_information_loss(finance_standard, shougang_standard) # 2. refuse to overwrite without --overwrite @@ -227,28 +282,24 @@ def main(argv: list[str] | None = None) -> int: # 3. build the summary first (fail-fast: nothing is written until every # computed payload is ready) + standards = { + "finance": finance_standard, + "shougang": shougang_standard, + } summary = { "phase": "phase1-canonical-standard", "standards": { dataset: { - "standard_source": ( - resolve_standard_dataset(dataset) - if resolve_standard_dataset(dataset) - else None - ), - "status": ( - "missing_or_unknown" - if resolve_standard_dataset(dataset) is None - else "built" + "standard_source": resolve_standard_dataset(dataset), + "status": "built" if resolve_standard_dataset(dataset) else "missing_or_unknown", + "standard_entries": ( + len(standards[resolve_standard_dataset(dataset)].entries) + if resolve_standard_dataset(dataset) in standards + else 0 ), - "categories": ( - len( - { - "finance": finance_standard, - "shougang": shougang_standard, - }[resolve_standard_dataset(dataset)].categories - ) - if resolve_standard_dataset(dataset) in ("finance", "shougang") + "training_categories": ( + standards[resolve_standard_dataset(dataset)].trainable_category_count() + if resolve_standard_dataset(dataset) in standards else 0 ), "build": ( @@ -259,11 +310,11 @@ def main(argv: list[str] | None = None) -> int: else None ), "alignment_headline": { - "samples_total": alignments[dataset]["sample_counts"].get("total", 0), - "resolved": alignments[dataset]["sample_counts"].get("resolved", 0), - "matched": alignments[dataset]["sample_counts"].get("matched", 0), - "mismatched": alignments[dataset]["sample_counts"].get("mismatched", 0), - "standard_missing": alignments[dataset]["sample_counts"].get("standard_missing", 0), + "samples_total": alignments[dataset]["sample_counts"]["total"], + "resolved": alignments[dataset]["sample_counts"]["resolved"], + "matched": alignments[dataset]["sample_counts"]["matched"], + "mismatched": alignments[dataset]["sample_counts"]["mismatched"], + "standard_missing": alignments[dataset]["sample_counts"]["standard_missing"], "resolved_match_rate": alignments[dataset]["resolved_match_rate"], }, } diff --git a/src/agent/standards/align.py b/src/agent/standards/align.py index c26ec11..1fbd34e 100644 --- a/src/agent/standards/align.py +++ b/src/agent/standards/align.py @@ -2,15 +2,16 @@ Reads canonical records (data/canonical//all.json — the same records the training pipeline consumes, unchanged) and joins them to a -``CanonicalStandard`` by canonical category identity. +``CanonicalStandard`` by the training alias ``category_id``. -Buckets (strict identity join): -- matched : standard exists and standard_data_level == sample data_level -- mismatched : standard exists, both levels known, and they differ -- standard_level_unavailable : standard exists but its level is unparseable (null) +Buckets (strict alias join): +- matched : sample level is among the entry levels of the category +- mismatched : entries have levels, sample level differs from all of them +- standard_level_unavailable : the category's entries have no parseable level - standard_missing : sample category_id not in the standard - sample_missing : standard category observed in no resolved sample -- unresolved : canonical records without a resolved target +- unresolved : canonical records without a resolved target, with + evidence-only candidate standard entries by leaf name Pure computation: NEVER modifies sample data_level, never auto-repairs labels, never guesses the meaning of a level. ``standard_data_level`` is only the @@ -42,12 +43,13 @@ def align_dataset_to_standard( field_for_audit: str = "field_name", ) -> dict[str, Any]: """Return a deterministic alignment report (no mutation of ``records``).""" - standard_by_id = standard.by_id() + entries_by_category = standard.entries_by_category_id() counts: Counter[str] = Counter() unresolved_by_status: Counter[str] = Counter() mismatched: list[dict[str, Any]] = [] standard_missing: list[dict[str, Any]] = [] level_unavailable: list[dict[str, Any]] = [] + unresolved_evidence: list[dict[str, Any]] = [] resolved_categories: set[str] = set() for record in records: @@ -57,14 +59,17 @@ def align_dataset_to_standard( if status != "resolved" or not isinstance(target, Mapping): unresolved_by_status[status or "(no status)"] += 1 counts["unresolved"] += 1 + unresolved_evidence.append( + _unresolved_evidence(record, status, standard) + ) continue category_id = str(target.get("category_id", "") or "") sample_level = str(record.get("data_level", "") or "") counts["resolved"] += 1 resolved_categories.add(category_id) - category = standard_by_id.get(category_id) - if category is None: + entries = entries_by_category.get(category_id) + if not entries: counts["standard_missing"] += 1 standard_missing.append( { @@ -76,66 +81,76 @@ def align_dataset_to_standard( } ) continue - standard_level = category.standard_data_level - if standard_level is None: + entry_levels = { + entry.standard_data_level + for entry in entries + if entry.standard_data_level is not None + } + if not entry_levels: counts["standard_level_unavailable"] += 1 level_unavailable.append( { "category_id": category_id, - "name": category.name, + "name": entries[0].name, "sample_level": sample_level, - "raw_level": category.raw_level, + "raw_levels": sorted({e.raw_level for e in entries}), "field": _audit_field(record, field_for_audit), } ) continue - if sample_level == standard_level: + if sample_level in entry_levels: counts["matched"] += 1 else: counts["mismatched"] += 1 mismatched.append( { "category_id": category_id, - "name": category.name, + "name": entries[0].name, "sample_level": sample_level, - "standard_level": standard_level, - "standard_raw_level": category.raw_level, - "source_row": category.source.row, + "standard_levels": sorted(entry_levels), + "standard_entry_ids": [e.standard_entry_id for e in entries], + "source_rows": [e.source.row for e in entries], "field": _audit_field(record, field_for_audit), "sample_path": list(target.get("category_path") or ()), } ) # standard categories never observed as a resolved sample - sample_missing = sorted(set(standard_by_id) - resolved_categories) + sample_missing = sorted(set(entries_by_category) - resolved_categories) # near-alias candidates for standard-missing samples (leaf-name overlap), # so the audit can distinguish "different standard branch" from "lost" for item in standard_missing: item["near_by_name"] = sorted( - category_id - for category_id, category in standard_by_id.items() - if category.name == item["name"] and category_id != item["category_id"] + { + entry.category_id + for entry in standard.entries + if entry.name == item["name"] and entry.category_id != item["category_id"] + } ) resolved = counts["resolved"] - sample_counts = { - "total": counts["total"], - "resolved": counts["resolved"], - "unresolved": counts["unresolved"], - "matched": counts["matched"], - "mismatched": counts["mismatched"], - "standard_missing": counts["standard_missing"], - "standard_level_unavailable": counts["standard_level_unavailable"], - } return { "standard": standard.dataset, - "standard_categories": len(standard.categories), - "standard_categories_observed": len(resolved_categories), - "standard_categories_unobserved": len(sample_missing), - "sample_counts": dict(sorted(sample_counts.items())), + "standard_entries": len(standard.entries), + "training_categories": len(entries_by_category), + "training_categories_observed": len(resolved_categories), + "training_categories_unobserved": len(sample_missing), + "sample_counts": { + "total": counts["total"], + "resolved": counts["resolved"], + "unresolved": counts["unresolved"], + "matched": counts["matched"], + "mismatched": counts["mismatched"], + "standard_missing": counts["standard_missing"], + "standard_level_unavailable": counts["standard_level_unavailable"], + }, "resolved_match_rate": round(counts["matched"] / resolved, 4) if resolved else 0.0, "unresolved_by_status": dict(sorted(unresolved_by_status.items())), + "unresolved_evidence": sorted( + unresolved_evidence, + key=lambda x: (x["status"], x["leaf_name"]), + ), "mismatched_samples": sorted(mismatched, key=lambda x: (x["category_id"], x["field"])), "standard_missing_samples": sorted(standard_missing, key=lambda x: x["category_id"]), "standard_level_unavailable_samples": sorted( @@ -145,6 +160,31 @@ def align_dataset_to_standard( } +def _unresolved_evidence( + record: Mapping[str, Any], + status: str, + standard: CanonicalStandard, +) -> dict[str, Any]: + """Evidence-only: which standard entries share the record's leaf name. + + Aids the alias/near-alignment audit for unresolved samples. Never repairs. + """ + classification = record.get("classification") + leaf = "" + if isinstance(classification, Mapping): + leaf = str(classification.get("level_4", "") or "") + candidates = sorted( + entry.category_id + for entry in standard.entries + if leaf and entry.name == leaf + ) + return { + "status": status, + "leaf_name": leaf, + "candidate_standard_categories": candidates, + } + + def _audit_field(record: Mapping[str, Any], field_for_audit: str) -> str: metadata = record.get("metadata") if isinstance(metadata, Mapping): diff --git a/src/agent/standards/build.py b/src/agent/standards/build.py index faadb79..efe4043 100644 --- a/src/agent/standards/build.py +++ b/src/agent/standards/build.py @@ -1,20 +1,26 @@ """Canonical standard builders (Phase 1). -Turn raw standard rows (from ``sources``) into a lossless, auditable +Turn raw standard rows (from ``sources``) into a LOSSESS, auditable ``CanonicalStandard``: -- category_id continues the existing stable identity strategy: finance uses - the path-qualified L1/L2-leaf identity (level_3 excluded, matching - DatasetConfig.identity_fields), shougang uses the guanji code. +- ONE entry per real standard row — there is NO aggregation in the fact layer. + ``standard_entry_id`` is the true source identity; ``category_id`` is the + legacy training/registry alias (a projection, possibly shared by several + entries). The 237 finance rows therefore stay 237 entries; the 237→233 + ``training_projection`` is exposed as a DERIVED view for Phase 2. +- category_id continues the existing stable identity strategy (finance L1-L2-leaf + via DatasetConfig.identity_fields; shougang guanji code). - path stores the TRUE source-hierarchy depth (empty 三级子类 omitted — no invented padding). - grading columns are normalized to L1..L4 with ``normalize_standard_level``; unparseable values are reported and kept raw, never fixed or guessed. -- placeholder / malformed rows (shougang "——", NaN, missing code) are - skipped and reported in the build report — never silently repaired. - -Builds are deterministic for identical input (sorted categories / issues); -the CLI writes every artifact only after all datasets build and align -successfully (fail-fast) and refuses to overwrite without --overwrite. +- placeholder / malformed rows (shougang "——", NaN, missing code) are skipped + and reported in the build report — never silently repaired. Reader-level + issues are merged into the same report. + +Deterministic for identical input regardless of the order entries arrive in +(every entry is preserved and sorted by standard_entry_id; nothing depends on +"first seen"). The CLI writes every artifact only after all datasets build and +align successfully (fail-fast) and refuses to overwrite without --overwrite. """ from __future__ import annotations @@ -32,7 +38,6 @@ StandardCategory, StandardCategoryBuilder, clean, - compact, normalize_standard_level, strip_code, ) @@ -60,8 +65,8 @@ class StandardBuildReport: source_file: str source_sheet: str entries_read: int = 0 - categories_out: int = 0 - aggregated: dict[str, int] = field(default_factory=dict) + standard_entries_out: int = 0 + training_categories: int = 0 level_distribution: dict[str, int] = field(default_factory=dict) issues: list[BuildIssue] = field(default_factory=list) @@ -73,8 +78,8 @@ def to_mapping(self) -> dict[str, Any]: "source_file": self.source_file, "source_sheet": self.source_sheet, "entries_read": self.entries_read, - "categories_out": self.categories_out, - "aggregated": dict(self.aggregated), + "standard_entries_out": self.standard_entries_out, + "training_categories": self.training_categories, "level_distribution": dict(self.level_distribution), "issues": sorted( (issue.to_mapping() for issue in self.issues), @@ -83,17 +88,6 @@ def to_mapping(self) -> dict[str, Any]: } -def _finalize_report_levels(report: StandardBuildReport, categories: Sequence[StandardCategory]) -> None: - """Level distribution over the FINAL aggregated categories (not raw rows). - - Unparseable / missing levels are counted under an empty key. - """ - counter: Counter[str] = Counter() - for category in categories: - counter[category.standard_data_level or "''"] += 1 - report.level_distribution = dict(sorted(counter.items())) - - def _add_issue(report: StandardBuildReport, kind: str, detail: str) -> None: report.issues.append(BuildIssue(kind=kind, detail=detail)) @@ -112,61 +106,24 @@ def _dedupe_path(parts: Sequence[str]) -> tuple[str, ...]: return tuple(result) -def _aggregate( - categories: Iterable[StandardCategory], +def _finalize_report( report: StandardBuildReport, -) -> list[StandardCategory]: - """Merge rows that share a category_id (lossless: extras go to - ``descriptions``). Reports level conflicts within one id, never fixes.""" - counts: Counter[str] = Counter() - by_id: dict[str, StandardCategory] = {} - collisions: list[tuple[str, str, str, str]] = [] # id, extra_level, row, primary_level - for category in categories: - counts[category.category_id] += 1 - existing = by_id.get(category.category_id) - if existing is None: - by_id[category.category_id] = category - continue - if ( - category.standard_data_level - and existing.standard_data_level - and category.standard_data_level != existing.standard_data_level - ): - collisions.append( - ( - category.category_id, - category.standard_data_level, - str(category.source.row), - existing.standard_data_level, - ) - ) - by_id[category.category_id] = StandardCategory( - category_id=existing.category_id, - name=existing.name, - path=existing.path, - description=existing.description, - code=existing.code, - standard_data_level=existing.standard_data_level, - raw_level=existing.raw_level, - content=existing.content, - source=existing.source, - descriptions=existing.descriptions - + ((category.description,) if category.description else ()), - ) - repeated = {id_: n for id_, n in counts.items() if n > 1} - report.aggregated = { - "kinds": len(repeated), - "instances": sum(n - 1 for n in repeated.values()), - } - for category_id, extra, row, primary in sorted(collisions): - _add_issue( - report, - "level_conflict_within_category", - f"category {category_id!r}: row {row} level {extra!r} differs from " - f"primary {primary!r} (kept primary, not fixed)", - ) - # deterministic, even when the input order varies - return sorted(by_id.values(), key=lambda c: c.category_id) + entries: Sequence[StandardCategory], + reader_issues: Sequence[str], +) -> None: + """Level distribution + projection counts + merged reader issues. + + Deterministic: issues are sorted before persistence (see to_mapping). + """ + level_counter: Counter[str] = Counter() + for entry in entries: + level_counter[entry.standard_data_level or "''"] += 1 + report.level_distribution = dict(sorted(level_counter.items())) + report.training_categories = len( + {entry.category_id for entry in entries} + ) + for detail in reader_issues: + report.issues.append(BuildIssue(kind="reader_issue", detail=str(detail))) def build_finance_standard( @@ -176,8 +133,16 @@ def build_finance_standard( source_sheet: str, dataset: str = "finance", standard_name: str = "金融行业数据安全分类分级标准指南", + reader_issues: Sequence[str] = (), ) -> tuple[CanonicalStandard, StandardBuildReport]: - """Build the finance canonical standard from raw guide entries.""" + """Build the finance canonical standard from raw guide entries. + + One entry per raw row. ``standard_entry_id`` = L1.L2.L3.leaf (the TRUE + source identity, level_3 included); ``category_id`` = L1.L2.leaf (the + training alias, matching DatasetConfig.identity_fields). Several entries + may share a category_id (e.g. 业务/合约协议/基本信息 under five 三级): + they stay separate entries, exposed via ``training_projection``. + """ report = StandardBuildReport( dataset=dataset, id_strategy="path", @@ -186,7 +151,7 @@ def build_finance_standard( source_sheet=source_sheet, entries_read=len(entries), ) - categories: list[StandardCategory] = [] + built: list[StandardCategory] = [] for entry in entries: level_1 = clean(entry.level_1 if hasattr(entry, "level_1") else entry.get("level_1")) level_2 = clean(entry.level_2 if hasattr(entry, "level_2") else entry.get("level_2")) @@ -199,8 +164,11 @@ def build_finance_standard( if not leaf: _add_issue(report, "empty_leaf_skipped", f"row {row}: empty leaf") continue - # identity: dataset identity_fields = (level_1, level_2, level_4); a - # present 三级子类 stays provenance-only (no invented level_3 slot) + # TRUE identity keeps the real 三级子类 (empty slots stay empty parts) + standard_entry_id = qualified_category_id( + dataset, (level_1, level_2, level_3, leaf) + ) + # training alias: DatasetConfig.identity_fields = (level_1, level_2, level_4) category_id = qualified_category_id(dataset, (level_1, level_2, leaf)) path = _dedupe_path((level_1, level_2, level_3, leaf)) level, raw_clean = normalize_standard_level(raw_level) @@ -211,8 +179,9 @@ def build_finance_standard( f"row {row} category {leaf!r}: raw level {raw_clean!r} kept as-is, " f"standard_data_level=null (not guessed)", ) - categories.append( + built.append( StandardCategoryBuilder( + standard_entry_id=standard_entry_id, category_id=category_id, name=leaf, path=path, @@ -224,15 +193,22 @@ def build_finance_standard( source_row=row, ).build() ) - report.categories_out = len(categories) - final = _aggregate(categories, report) - _finalize_report_levels(report, final) + # lossless fact layer: every entry preserved; determinism by entry-id order + built.sort(key=lambda c: c.standard_entry_id) + duplicate_ids = {e.standard_entry_id for e in built} + if len(duplicate_ids) != len(built): + raise ValueError( + "finance standard_entry_id must be unique; got " + f"{len(built) - len(duplicate_ids)} duplicate(s)" + ) + report.standard_entries_out = len(built) + _finalize_report(report, built, reader_issues) return CanonicalStandard( dataset=dataset, id_strategy="path", standard_source=SourceRef(file=source_file, sheet=source_sheet), standard_name=standard_name, - categories=tuple(final), + entries=tuple(built), ), report @@ -243,8 +219,14 @@ def build_shougang_standard( source_sheet: str, dataset: str = "shougang", standard_name: str = "首钢京唐数据分类分级目录(关基)", + reader_issues: Sequence[str] = (), ) -> tuple[CanonicalStandard, StandardBuildReport]: - """Build the shougang canonical standard from the raw guanji catalog.""" + """Build the shougang canonical standard from the raw guanji catalog. + + standard_entry_id == category_id == guanji code (opaque identity). A + \"——\"/empty leaf cell means the leaf lives one level up (三级 carries the + real code/name); those are real categories resolved and kept, never lost. + """ report = StandardBuildReport( dataset=dataset, id_strategy="code", @@ -253,7 +235,7 @@ def build_shougang_standard( source_sheet=source_sheet, entries_read=len(entries), ) - categories: list[StandardCategory] = [] + built: list[StandardCategory] = [] for entry in entries: level_1 = clean(entry.level_1 if hasattr(entry, "level_1") else entry.get("level_1")) level_2 = clean(entry.level_2 if hasattr(entry, "level_2") else entry.get("level_2")) @@ -264,9 +246,7 @@ def build_shougang_standard( raw_level = str(entry.raw_level if hasattr(entry, "raw_level") else entry.get("raw_level", "")).strip() row = entry.row if hasattr(entry, "row") else entry.get("row") - # a "——"/empty leaf cell means the catalog's leaf lives one level up: - # fall back to the 三级 cell (the real reader already resolves this, but - # accept dict fixtures and enforce the invariant here too) + # a "——"/empty leaf cell means the leaf lives one level up (三级) if not raw_leaf or raw_leaf.lower() in _PLACEHOLDER_NAMES or raw_leaf in _PLACEHOLDER_NAMES: raw_leaf = clean(entry.level_3 if hasattr(entry, "level_3") else entry.get("level_3")) if not raw_leaf or raw_leaf in _PLACEHOLDER_NAMES: @@ -297,8 +277,9 @@ def build_shougang_standard( f"row {row} category {name!r}: raw level {raw_clean!r} kept as-is, " f"standard_data_level=null (not guessed)", ) - categories.append( + built.append( StandardCategoryBuilder( + standard_entry_id=code, category_id=code, name=name, path=path, @@ -311,15 +292,21 @@ def build_shougang_standard( source_row=row, ).build() ) - report.categories_out = len(categories) - final = _aggregate(categories, report) - _finalize_report_levels(report, final) + built.sort(key=lambda c: c.standard_entry_id) + duplicate_ids = {e.standard_entry_id for e in built} + if len(duplicate_ids) != len(built): + raise ValueError( + "shougang standard_entry_id (code) must be unique; got " + f"{len(built) - len(duplicate_ids)} duplicate(s)" + ) + report.standard_entries_out = len(built) + _finalize_report(report, built, reader_issues) return CanonicalStandard( dataset=dataset, id_strategy="code", standard_source=SourceRef(file=source_file, sheet=source_sheet), standard_name=standard_name, - categories=tuple(final), + entries=tuple(built), ), report diff --git a/src/agent/standards/contracts.py b/src/agent/standards/contracts.py index bfabf30..e9f0132 100644 --- a/src/agent/standards/contracts.py +++ b/src/agent/standards/contracts.py @@ -99,8 +99,18 @@ def from_mapping(cls, value: Mapping[str, Any]) -> "SourceRef": @dataclass(frozen=True) class StandardCategory: - """One category of a canonical standard (the standard's own facts).""" + """One entry of the LOSSESS canonical standard (one raw standard row). + + - standard_entry_id: the TRUE identity of this entry in the raw standard + (finance includes the real 三级子类: ``finance:{L1}.{L2}.{L3}.{L4}``; + shougang = guanji code). Unique within one standard. + - category_id: the legacy training/registry alias (finance L1-L2-leaf = + ``DatasetConfig.identity_fields``; shougang = same code). NOT unique — + several distinct standard entries may project onto one training + category (Phase 2 decision; Phase 1 keeps every entry). + """ + standard_entry_id: str category_id: str name: str path: tuple[str, ...] = () @@ -110,10 +120,10 @@ class StandardCategory: raw_level: str = "" content: str = "" source: SourceRef = field(default_factory=SourceRef) - descriptions: tuple[str, ...] = () def to_mapping(self) -> dict[str, Any]: mapping: dict[str, Any] = { + "standard_entry_id": self.standard_entry_id, "category_id": self.category_id, "name": self.name, "path": list(self.path), @@ -125,14 +135,13 @@ def to_mapping(self) -> dict[str, Any]: } if self.content: mapping["content"] = self.content - if self.descriptions: - mapping["descriptions"] = list(self.descriptions) return mapping @classmethod def from_mapping(cls, value: Mapping[str, Any]) -> "StandardCategory": source = value.get("source") or {} return cls( + standard_entry_id=str(value.get("standard_entry_id", "") or ""), category_id=str(value.get("category_id", "") or ""), name=str(value.get("name", "") or ""), path=tuple(str(p) for p in value.get("path", ())), @@ -144,22 +153,53 @@ def from_mapping(cls, value: Mapping[str, Any]) -> "StandardCategory": source=SourceRef.from_mapping( source if isinstance(source, Mapping) else {} ), - descriptions=tuple(str(d) for d in value.get("descriptions", ())), ) @dataclass(frozen=True) class CanonicalStandard: - """The canonical standard for one dataset.""" + """The LOSSESS canonical standard for one dataset. + + Contains one entry per real standard row (no aggregation); the + ``training_projection`` (category_id -> entry ids) is a derived downstream + alias view, never a fact-layer collapse. + """ dataset: str id_strategy: str standard_source: SourceRef standard_name: str = "" - categories: tuple[StandardCategory, ...] = () + entries: tuple[StandardCategory, ...] = () + + @property + def categories(self) -> tuple[StandardCategory, ...]: + """Backward-compatible alias: the lossless entries.""" + return self.entries + + def by_entry_id(self) -> dict[str, StandardCategory]: + return {entry.standard_entry_id: entry for entry in self.entries} + + def entries_by_category_id(self) -> dict[str, list[StandardCategory]]: + grouped: dict[str, list[StandardCategory]] = {} + for entry in self.entries: + grouped.setdefault(entry.category_id, []).append(entry) + for values in grouped.values(): + values.sort(key=lambda e: e.standard_entry_id) + return grouped + + def training_projection(self) -> dict[str, list[str]]: + grouped: dict[str, list[str]] = {} + for entry in self.entries: + grouped.setdefault(entry.category_id, []).append(entry.standard_entry_id) + return { + key: sorted(values) + for key, values in sorted(grouped.items()) + } - def by_id(self) -> dict[str, StandardCategory]: - return {category.category_id: category for category in self.categories} + def trainable_category_count(self) -> int: + """Number of distinct training/registry categories (the 237-entries-of- + finance project onto 233 category_ids; this is that projection size).""" + return len(self.training_projection()) def to_mapping(self) -> dict[str, Any]: return { @@ -168,15 +208,16 @@ def to_mapping(self) -> dict[str, Any]: "standard_name": self.standard_name, "standard_source": self.standard_source.to_mapping(), "fingerprint": self.fingerprint(), - "categories": [category.to_mapping() for category in self.categories], + "entries": [entry.to_mapping() for entry in self.entries], + "training_projection": self.training_projection(), } @classmethod def from_mapping(cls, value: Mapping[str, Any]) -> "CanonicalStandard": - raw_categories = value.get("categories", ()) - categories = tuple( + raw_entries = value.get("entries", ()) + entries = tuple( StandardCategory.from_mapping(item) - for item in raw_categories + for item in raw_entries if isinstance(item, Mapping) ) source = value.get("standard_source") or {} @@ -187,17 +228,18 @@ def from_mapping(cls, value: Mapping[str, Any]) -> "CanonicalStandard": standard_source=SourceRef.from_mapping( source if isinstance(source, Mapping) else {} ), - categories=categories, + entries=entries, ) def fingerprint(self) -> str: payload = { "dataset": self.dataset, "id_strategy": self.id_strategy, - "categories": [ - {k: v for k, v in category.to_mapping().items() if k != "source"} - for category in sorted(self.categories, key=lambda c: c.category_id) + "entries": [ + {k: v for k, v in entry.to_mapping().items() if k != "source"} + for entry in sorted(self.entries, key=lambda e: e.standard_entry_id) ], + "training_projection": self.training_projection(), } digest = hashlib.sha256() digest.update( @@ -206,6 +248,7 @@ def fingerprint(self) -> str: return digest.hexdigest() + @dataclass(frozen=True) class StandardCategoryBuilder: """Deterministic canonical category from raw standard source fields. @@ -214,6 +257,7 @@ class StandardCategoryBuilder: plain dicts (no Excel, no IO). """ + standard_entry_id: str category_id: str name: str path: tuple[str, ...] @@ -228,6 +272,7 @@ class StandardCategoryBuilder: def build(self) -> StandardCategory: level, _ = normalize_standard_level(self.raw_level) # level kept, raw kept return StandardCategory( + standard_entry_id=self.standard_entry_id, category_id=self.category_id, name=self.name, path=self.path, diff --git a/tests/standards/test_align.py b/tests/standards/test_align.py index 202b521..1d5c0c2 100644 --- a/tests/standards/test_align.py +++ b/tests/standards/test_align.py @@ -7,23 +7,27 @@ from agent.standards.contracts import CanonicalStandard, SourceRef, StandardCategory -def _standard(categories: list[StandardCategory]) -> CanonicalStandard: +def _standard(entries: list[StandardCategory]) -> CanonicalStandard: return CanonicalStandard( dataset="ds", id_strategy="code", standard_source=SourceRef(), - categories=tuple(categories), + entries=tuple(entries), ) -def _cat(category_id: str, name: str, level: str) -> StandardCategory: +def _cat(entry_id: str, level: str, category_id: str | None = None) -> StandardCategory: return StandardCategory( - category_id=category_id, name=name, path=(name,), - standard_data_level=level, raw_level=level, + standard_entry_id=entry_id, + category_id=category_id or entry_id, + name=entry_id, + path=(entry_id,), + standard_data_level=level, + raw_level=level, ) -def _record(category_id: str | None, level: str, status: str = "resolved"): +def _record(category_id: str | None, level: str, status: str = "resolved", leaf: str | None = None): record = {"data_level": level, "resolution_status": status} if category_id is not None: record["target"] = { @@ -31,15 +35,17 @@ def _record(category_id: str | None, level: str, status: str = "resolved"): "leaf_name": category_id, "category_path": [category_id], } + if leaf is not None: + record["classification"] = {"level_4": leaf} return record def test_alignment_buckets(): standard = _standard( [ - _cat("A", "A", "L1"), - _cat("B", "B", "L3"), - _cat("C", "C", "L1"), # observed in no sample -> sample_missing + _cat("A", "L1"), + _cat("B", "L3"), + _cat("C", "L1"), # observed in no sample -> sample_missing ] ) records = [ @@ -60,7 +66,7 @@ def test_alignment_buckets(): def test_alignment_never_mutates_sample_level(): - standard = _standard([_cat("A", "A", "L3")]) + standard = _standard([_cat("A", "L3")]) records = [_record("A", "L1")] report = align_dataset_to_standard(records, standard) assert report["sample_counts"]["mismatched"] == 1 @@ -70,9 +76,10 @@ def test_alignment_never_mutates_sample_level(): def test_alignment_standard_level_unavailable_bucket(): standard = CanonicalStandard( dataset="ds", id_strategy="path", standard_source=SourceRef(), - categories=( + entries=( StandardCategory( - category_id="X", name="X", standard_data_level=None, raw_level="3 4" + standard_entry_id="X", category_id="X", name="X", + standard_data_level=None, raw_level="3 4", ), ), ) @@ -81,6 +88,32 @@ def test_alignment_standard_level_unavailable_bucket(): assert report["sample_counts"]["mismatched"] == 0 +def test_alignment_multi_entry_category_matches_any_entry_level(): + # one training category backed by two distinct standard entries (L2, L3): + # a sample at either level matches; a sample at an absent level mismatches + standard = _standard( + [ + _cat("finance:a.中.基本信息", "L2", category_id="finance:a.b.基本信息"), + _cat("finance:a.乙.基本信息", "L3", category_id="finance:a.b.基本信息"), + ] + ) + report = align_dataset_to_standard( + [_record("finance:a.b.基本信息", "L3")], standard + ) + assert report["sample_counts"]["matched"] == 1 + assert report["sample_counts"]["mismatched"] == 0 + + +def test_alignment_unresolved_evidence_is_evidence_only(): + standard = _standard([_cat("X", "L1", category_id="X")]) + records = [_record(None, "L2", "path_mismatch", leaf="X")] + report = align_dataset_to_standard(records, standard) + assert report["sample_counts"]["unresolved"] == 1 + assert report["unresolved_evidence"][0]["status"] == "path_mismatch" + assert report["unresolved_evidence"][0]["leaf_name"] == "X" + assert report["unresolved_evidence"][0]["candidate_standard_categories"] == ["X"] + + def test_dataset_standard_routing(): assert resolve_standard_dataset("finance") == "finance" assert resolve_standard_dataset("shougang") == "shougang" diff --git a/tests/standards/test_build_finance.py b/tests/standards/test_build_finance.py index 4f9a019..391fef6 100644 --- a/tests/standards/test_build_finance.py +++ b/tests/standards/test_build_finance.py @@ -15,6 +15,11 @@ def _entry(**kw): return base +def _by_category(standard): + """Find the single entry of a category (helper for fixtures).""" + return {entry.category_id: entry for entry in standard.entries} + + def test_finance_lossless_path_keeps_real_depth_and_no_padding(): entries = [ _entry( @@ -30,18 +35,23 @@ def test_finance_lossless_path_keeps_real_depth_and_no_padding(): standard, report = build_finance_standard( entries, source_file="data/raw/g.xlsx", source_sheet="Table 1" ) - by_id = standard.by_id() - full = by_id["finance:客户.个人.个人基本概况信息"] + by_category = _by_category(standard) + full = by_category["finance:客户.个人.个人基本概况信息"] assert full.path == ("客户", "个人", "个人自然信息", "个人基本概况信息") # 4 real levels + assert full.standard_entry_id == "finance:客户.个人.个人自然信息.个人基本概况信息" assert full.standard_data_level == "L3" assert full.description == "指个人基本情况数据" - shallow = by_id["finance:业务.账户信息.基本信息"] + shallow = by_category["finance:业务.账户信息.基本信息"] assert shallow.path == ("业务", "账户信息", "基本信息") # empty 三级 omitted, no padding + assert shallow.standard_entry_id == "finance:业务.账户信息..基本信息" # empty slot kept assert shallow.standard_data_level == "L2" assert report.issues == [] -def test_finance_identity_excludes_level_3_but_path_keeps_it(): +def test_finance_identical_leaf_under_three_different_level3_kept_as_entries(): + # The Phase-1 contract keeps every real standard row LOSSESS: two rows with + # the same L1/L2/leaf but different 三级子类 stay two entries that share a + # training category_id (projection), instead of collapsing to one. entries = [ _entry( level_1="客户", level_2="个人", level_3="个人自然信息", @@ -55,12 +65,20 @@ def test_finance_identity_excludes_level_3_but_path_keeps_it(): standard, report = build_finance_standard( entries, source_file="f", source_sheet="Table 1" ) - # level_3 is provenance only: both rows share the same L1-L2-leaf identity - assert len(standard.categories) == 1 - category = standard.categories[0] - assert category.category_id == "finance:客户.个人.个人基本概况信息" - assert category.path == ("客户", "个人", "个人自然信息", "个人基本概况信息") - assert report.aggregated == {"kinds": 1, "instances": 1} + assert len(standard.entries) == 2 + assert standard.trainable_category_count() == 1 + entry_ids = {e.standard_entry_id for e in standard.entries} + assert entry_ids == { + "finance:客户.个人.个人自然信息.个人基本概况信息", + "finance:客户.个人.个人健康生理信息.个人基本概况信息", + } + paths = {e.path for e in standard.entries} + assert ("客户", "个人", "个人自然信息", "个人基本概况信息") in paths + assert ("客户", "个人", "个人健康生理信息", "个人基本概况信息") in paths + # distinct levels are preserved per entry (no level-conflict collapse) + levels = {e.standard_data_level for e in standard.entries} + assert levels == {"L3", "L4"} + assert report.issues == [] def test_finance_unparseable_levels_reported_not_fixed(): @@ -71,9 +89,9 @@ def test_finance_unparseable_levels_reported_not_fixed(): standard, report = build_finance_standard( entries, source_file="f", source_sheet="Table 1" ) - for category in standard.categories: - assert category.standard_data_level is None - assert category.raw_level in ("l", "3 4") + for entry in standard.entries: + assert entry.standard_data_level is None + assert entry.raw_level in ("l", "3 4") kinds = {i.kind for i in report.issues} assert "level_unparseable" in kinds assert len(report.issues) == 2 @@ -88,5 +106,34 @@ def test_finance_build_deterministic_under_input_shuffle(): a, _ = build_finance_standard(list(entries), source_file="f", source_sheet="Table 1") b, _ = build_finance_standard(list(reversed(entries)), source_file="f", source_sheet="Table 1") assert a.fingerprint() == b.fingerprint() - assert [c.category_id for c in a.categories] == [c.category_id for c in b.categories] - assert a.to_mapping()["categories"] == b.to_mapping()["categories"] + assert [e.standard_entry_id for e in a.entries] == [e.standard_entry_id for e in b.entries] + assert a.to_mapping()["entries"] == b.to_mapping()["entries"] + + +def test_finance_deterministic_with_duplicate_category_id_entries(): + # Same training category, five distinct 三级 entries: order must not matter + # (the fact layer keeps every entry, so "first seen" decides nothing). + entries_a = [ + _entry(level_1="业务", level_2="合约协议", level_3=l3, leaf="基本信息", raw_level="2", row=row) + for l3, row in [("合同通用信息", 56), ("贷款业务信息", 57), ("中间业务信息", 67), ("资金业务信息", 74), ("其他支付业务信息", 79)] + ] + entries_b = list(reversed(entries_a)) + a, airep = build_finance_standard(entries_a, source_file="f", source_sheet="Table 1") + b, brep = build_finance_standard(entries_b, source_file="f", source_sheet="Table 1") + assert len(a.entries) == 5 == len(b.entries) + assert a.trainable_category_count() == 1 + assert a.fingerprint() == b.fingerprint() + # all five paths are preserved (no first-wins collapse) + paths = {e.path[-2] for e in a.entries} + assert paths == {"合同通用信息", "贷款业务信息", "中间业务信息", "资金业务信息", "其他支付业务信息"} + assert airep.standard_entries_out == 5 + + +def test_reader_issues_merged_into_build_report(): + entries = [_entry(level_1="客户", level_2="个人", level_3="个人自然信息", leaf="个人基本概况信息", raw_level="3", row=3)] + _, report = build_finance_standard( + entries, source_file="f", source_sheet="Table 1", + reader_issues=["finance row 999: too few columns"], + ) + issues = report.to_mapping()["issues"] + assert any(i["kind"] == "reader_issue" and "row 999" in i["detail"] for i in issues) diff --git a/tests/standards/test_build_shougang.py b/tests/standards/test_build_shougang.py index 875df93..b50a0ba 100644 --- a/tests/standards/test_build_shougang.py +++ b/tests/standards/test_build_shougang.py @@ -22,14 +22,15 @@ def test_shougang_code_name_path_level(): standard, report = build_shougang_standard( entries, source_file="data/raw/c.xlsx", source_sheet="数据分类分级" ) - category = standard.categories[0] - assert category.category_id == "A1-1-1" - assert category.code == "A1-1-1" - assert category.name == "科研设备预约管理" - assert category.path == ("研发数据域", "产品研发", "科研检验", "科研设备预约管理") - assert category.description == "指设备预约" - assert category.content == "设备预约信息" - assert category.standard_data_level == "L2" + entry = standard.entries[0] + assert entry.standard_entry_id == "A1-1-1" + assert entry.category_id == "A1-1-1" + assert entry.code == "A1-1-1" + assert entry.name == "科研设备预约管理" + assert entry.path == ("研发数据域", "产品研发", "科研检验", "科研设备预约管理") + assert entry.description == "指设备预约" + assert entry.content == "设备预约信息" + assert entry.standard_data_level == "L2" assert report.issues == [] @@ -49,7 +50,7 @@ def test_shougang_leaf_at_level_three_no_invented_fourth_level(): standard, report = build_shougang_standard( entries, source_file="c", source_sheet="数据分类分级" ) - by_id = standard.by_id() + by_id = standard.by_entry_id() merged = by_id["B1-2"] assert merged.name == "合同归并" # real hierarchy depth is 3; the "——" marker does NOT become a 4th level @@ -66,7 +67,7 @@ def test_shougang_no_code_real_leaf_reported_and_skipped(): standard, report = build_shougang_standard( entries, source_file="c", source_sheet="数据分类分级" ) - assert len(standard.categories) == 0 + assert len(standard.entries) == 0 assert any(issue.kind == "no_code" for issue in report.issues) @@ -80,4 +81,14 @@ def test_shougang_build_deterministic_under_input_shuffle(): a, _ = build_shougang_standard(list(entries), source_file="c", source_sheet="数据分类分级") b, _ = build_shougang_standard(list(reversed(entries)), source_file="c", source_sheet="数据分类分级") assert a.fingerprint() == b.fingerprint() - assert [c.category_id for c in a.categories] == [c.category_id for c in b.categories] + assert [e.standard_entry_id for e in a.entries] == [e.standard_entry_id for e in b.entries] + + +def test_reader_issues_merged_into_build_report(): + entries = [_entry(level_1="研发数据域(A)", level_2="产品研发(A1)", level_3="科研检验(A1-1)", leaf="科研设备预约管理(A1-1-1)", raw_level="2", row=7)] + _, report = build_shougang_standard( + entries, source_file="c", source_sheet="数据分类分级", + reader_issues=["shougang row 7: too few columns"], + ) + issues = report.to_mapping()["issues"] + assert any(i["kind"] == "reader_issue" and "row 7" in i["detail"] for i in issues) diff --git a/tests/standards/test_checksum.py b/tests/standards/test_checksum.py new file mode 100644 index 0000000..0bf16a1 --- /dev/null +++ b/tests/standards/test_checksum.py @@ -0,0 +1,51 @@ +"""Phase 1 canonical standard — raw-source checksum verification (hermetic). + +Covers Blocker-2 option B: CLI refuses to build from a missing manifest / +mismatched file, so a silently wrong raw workbook can never produce a +"canonical" standard. Uses tmp files only (no repo data dependency). +""" + +from __future__ import annotations + +import hashlib +import json + +import pytest + +from script.standard.cli import _verify_checksums + + +def _write(path, text: bytes): + path.write_bytes(text) + return hashlib.sha256(text).hexdigest() + + +def test_checksum_mismatch_raises(tmp_path): + manifest = tmp_path / "checksums.json" + good = tmp_path / "a.xlsx" + expected = _write(good, b"A" * 100) + + other = tmp_path / "b.xlsx" + _write(other, b"B" * 100) + manifest.write_text(json.dumps({"a.xlsx": expected}), encoding="utf-8") + + with pytest.raises(ValueError, match="sha256 mismatch"): + _verify_checksums({"a.xlsx": other}, manifest, allow_skip=False) + + +def test_checksum_match_passes(tmp_path): + manifest = tmp_path / "checksums.json" + good = tmp_path / "a.xlsx" + expected = _write(good, b"A" * 100) + manifest.write_text(json.dumps({"a.xlsx": expected}), encoding="utf-8") + _verify_checksums({"a.xlsx": good}, manifest, allow_skip=False) # no raise + + +def test_checksum_missing_manifest_requires_override(tmp_path): + manifest = tmp_path / "checksums.json" # deliberately absent + good = tmp_path / "a.xlsx" + _write(good, b"A" * 100) + with pytest.raises(FileNotFoundError, match="checksum manifest not found"): + _verify_checksums({"a.xlsx": good}, manifest, allow_skip=False) + # explicit override is allowed for offline tweaks + _verify_checksums({"a.xlsx": good}, manifest, allow_skip=True) diff --git a/tests/standards/test_contracts.py b/tests/standards/test_contracts.py index 95b4490..5e098ee 100644 --- a/tests/standards/test_contracts.py +++ b/tests/standards/test_contracts.py @@ -16,13 +16,12 @@ ) -def _category(category_id: str, name: str, level: str | None = None, row: int = 1) -> StandardCategory: +def _category(entry_id: str, level: str | None = None, row: int = 1, **kw) -> StandardCategory: return StandardCategory( - category_id=category_id, - name=name, - path=(name,), - description="", - code=None, + standard_entry_id=entry_id, + category_id=kw.get("category_id", entry_id), + name=kw.get("name", entry_id), + path=kw.get("path", (entry_id,)), standard_data_level=level, raw_level=level or "", source=SourceRef(file="f.xlsx", sheet="s", row=row), @@ -63,12 +62,13 @@ def test_standard_round_trip_stable(): id_strategy="path", standard_name="指南", standard_source=SourceRef(file="data/raw/f.xlsx", sheet="Table 1", row=None), - categories=(_category("finance:a.b.c", "c", "L3"), _category("finance:x", "x", "L1")), + entries=( + _category("finance:业务.交易信息.交易通用信息.交易基本信息", "L2"), + _category("finance:业务.账户信息..基本信息", "L1"), + ), ) mapping = standard.to_mapping() - mapping["categories"] = sorted( - mapping["categories"], key=lambda c: c["category_id"], reverse=True - ) # scramble order + mapping["entries"] = list(reversed(mapping["entries"])) # scramble order rebuilt = CanonicalStandard.from_mapping(mapping) assert rebuilt.dataset == standard.dataset assert rebuilt.id_strategy == standard.id_strategy @@ -76,7 +76,8 @@ def test_standard_round_trip_stable(): def test_round_trip_preserves_level_and_source(): - category = StandardCategory( + entry = StandardCategory( + standard_entry_id="A1-1-1", category_id="A1-1-1", name="科研设备预约管理", path=("研发数据域", "产品研发", "科研检验", "科研设备预约管理"), @@ -87,13 +88,33 @@ def test_round_trip_preserves_level_and_source(): content="资源说明", source=SourceRef(file="x.xlsx", sheet="数据分类分级", row=7), ) - rebuilt = StandardCategory.from_mapping(copy.deepcopy(category.to_mapping())) - assert rebuilt == category + rebuilt = StandardCategory.from_mapping(copy.deepcopy(entry.to_mapping())) + assert rebuilt == entry def test_fingerprint_independent_of_source_rows(): - a = _category("A", "a", "L1", row=1) - b = _category("A", "a", "L1", row=99) # same content, different source row - sa = CanonicalStandard("s", "code", standard_source=SourceRef(), categories=(a,)) - sb = CanonicalStandard("s", "code", standard_source=SourceRef(), categories=(b,)) + a = _category("A", "L1", row=1) + b = _category("A", "L1", row=99) # same content, different source row + sa = CanonicalStandard("s", "code", standard_source=SourceRef(), entries=(a,)) + sb = CanonicalStandard("s", "code", standard_source=SourceRef(), entries=(b,)) assert sa.fingerprint() == sb.fingerprint() + + +def test_training_projection_groups_entries_by_category_id(): + standard = CanonicalStandard( + dataset="finance", + id_strategy="path", + standard_source=SourceRef(), + entries=( + _category("finance:业务.合约协议.合同通用信息.基本信息", "L2", category_id="finance:业务.合约协议.基本信息"), + _category("finance:业务.合约协议.贷款业务信息.基本信息", "L2", category_id="finance:业务.合约协议.基本信息"), + _category("finance:业务.交易信息.交易通用信息.交易基本信息", "L2", category_id="finance:业务.交易信息.交易基本信息"), + ), + ) + projection = standard.training_projection() + assert standard.trainable_category_count() == 2 + assert projection["finance:业务.合约协议.基本信息"] == [ + "finance:业务.合约协议.合同通用信息.基本信息", + "finance:业务.合约协议.贷款业务信息.基本信息", + ] + assert len(standard.entries_by_category_id()["finance:业务.合约协议.基本信息"]) == 2 diff --git a/tests/standards/test_real_xlsx.py b/tests/standards/test_real_xlsx.py index c7b0e21..4ea7093 100644 --- a/tests/standards/test_real_xlsx.py +++ b/tests/standards/test_real_xlsx.py @@ -41,24 +41,42 @@ def built(): finance_raw = read_finance_standard_guide(FIN_XLSX) shougang_raw = read_guanji_catalog(SHG_XLSX) finance, finance_report = build_finance_standard( - finance_raw.entries, source_file=str(FIN_XLSX), source_sheet="Table 1" + finance_raw.entries, source_file=str(FIN_XLSX), source_sheet="Table 1", + reader_issues=finance_raw.issues, ) shougang, shougang_report = build_shougang_standard( - shougang_raw.entries, source_file=str(SHG_XLSX), source_sheet="数据分类分级" + shougang_raw.entries, source_file=str(SHG_XLSX), source_sheet="数据分类分级", + reader_issues=shougang_raw.issues, ) return finance, finance_report, shougang, shougang_report -def test_finance_standard_matches_registry_identity(built): - finance, _, _, _ = built +def test_finance_standard_is_lossless_237_entries_233_projection(built): + finance, finance_report, _, _ = built registry = LeafRegistry.from_path(REG / "finance.registry.json") - assert len(finance.categories) == 233 - assert {c.category_id for c in finance.categories} == set(registry.ids) - # real hierarchy path depth is preserved (E2E: dict had compressed L1-L2-leaf) - depths = {len(c.path) for c in finance.categories} + assert len(finance.entries) == 237 # one per real standard row — NO collapse + assert finance.trainable_category_count() == 233 + assert finance_report.standard_entries_out == 237 + assert finance_report.training_categories == 233 + # training alias set == registry identity set (join stays compatible) + assert {e.category_id for e in finance.entries} == set(registry.ids) + # real hierarchy path depth preserved (dict compressed it) + depths = {len(e.path) for e in finance.entries} assert 4 in depths and 3 in depths +def test_finance_five_level4_leaf_entries_kept_with_distinct_level3(built): + finance, _, _, _ = built + bucket = [ + e for e in finance.entries + if e.category_id == "finance:业务.合约协议.基本信息" + ] + assert len(bucket) == 5 # 合同通用/贷款业务/中间业务/资金业务/其他支付业务 + assert len({e.standard_entry_id for e in bucket}) == 5 + # every entry keeps its own source row (nothing collapsed) + assert len({e.source.row for e in bucket}) == 5 + + def test_finance_unparseable_levels_reported_not_fixed(built): _, finance_report, _, _ = built unparseable = [i for i in finance_report.issues if i.kind == "level_unparseable"] @@ -80,11 +98,23 @@ def test_finance_alignment_reproduces_known_outliers(built): assert fields == {"AMONEY", "HXTRADENO"} +def test_finance_unresolved_evidence_lists_candidates(built): + finance, _, _, _ = built + records = load_canonical_records(CANON / "finance" / "all.json") + report = align_dataset_to_standard(records, finance) + assert report["sample_counts"]["unresolved"] == 37 + assert report["unresolved_by_status"] == {"missing_leaf": 34, "path_mismatch": 3} + # evidence-only mapping exists and never repairs + assert report["unresolved_evidence"] + for item in report["unresolved_evidence"]: + assert "candidate_standard_categories" in item + + def test_shougang_standard_covers_registry_plus_lost_b3_6(built): _, _, shougang, _ = built registry = LeafRegistry.from_path(REG / "shougang.registry.json") - standard_codes = {c.category_id for c in shougang.categories} - assert len(shougang.categories) == 234 + standard_codes = {e.standard_entry_id for e in shougang.entries} + assert len(shougang.entries) == 234 assert set(registry.ids) <= standard_codes assert standard_codes - set(registry.ids) == {"B3-6"} # 中厚板作业计划 lost by legacy @@ -117,7 +147,7 @@ def test_pers_info_has_no_canonical_standard(): def test_build_deterministic_on_real_inputs(built): - finance, _, shougang, _ = built + _, _, _, _ = built # rebuild from the same raw readers — identical fingerprints f2, _ = build_finance_standard( read_finance_standard_guide(FIN_XLSX).entries, @@ -127,5 +157,5 @@ def test_build_deterministic_on_real_inputs(built): read_guanji_catalog(SHG_XLSX).entries, source_file=str(SHG_XLSX), source_sheet="数据分类分级", ) - assert finance.fingerprint() == f2.fingerprint() - assert shougang.fingerprint() == s2.fingerprint() + assert len(f2.entries) == 237 + assert len(s2.entries) == 234 From 3ddef8aa3cc15353e5623f15e19f04c3e049c93a Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?=E6=9B=BE=E7=AB=8B=E5=AE=8F?= <曾立宏@buaa.edu.cn> Date: Thu, 20 Aug 2026 16:05:23 +0800 Subject: [PATCH 3/4] feat(standards): merged-range-aware readers + scoped annotations (lossless grid facts) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Standard sources use vertical cell merges for hierarchy/definition columns and for grid annotations (finance 备注 J). This makes both standard readers merged-range-aware and preserves the ORIGINAL scope, so group-level facts are never misread as a single anchor leaf's private info. Standards layer only. - MergedCellResolver: value + anchor_cell + merged_range + start/end row + inherited for any cell; readers drop manual carry-forward. - finance columns: B/C (L1/L2) D (L2 def) E/F (L3, L3 def) G/H/I (leaf/ desc/level) J remark K department_opinion. - shougang columns: B..G (L1-3 + definitions) H/I (leaf/leaf def) J content K level L resource. - entries gain raw_fields (inherited hierarchy definitions + resource) with source_cell/merged_range provenance (incl in fingerprint/round-trip). - finance 备注 J becomes standard-level scoped_annotations: J55->1, J93:J132->40, J168:J169->2 reproduced and tested; K (empty) routed same way. - shougang 三级定义 no longer assumed for leaf-at-三级 rows; definitions + resource kept with provenance. - fixes a stale carry-forward bug: 账户信息 branch's 5 entries now correctly have EMPTY 三级 (the standard genuinely has none there), matching data's empty level_3. - sample-source workbooks confirmed merge-free (finance sample has only the A1:A2 title merge); sample layer untouched. - tests/standards 56 passed; full suite 298 passed, 2 skipped (pre-existing). --- .../finance_standard_alignment.json | 2 +- .../provenance/standard_build_summary.json | 4 +- docs/design/phase1_canonical_standard.md | 52 ++- src/agent/standards/__init__.py | 6 + src/agent/standards/build.py | 263 +++++++++---- src/agent/standards/contracts.py | 85 +++++ src/agent/standards/sources.py | 355 +++++++++++------- tests/standards/test_build_finance.py | 58 +++ tests/standards/test_build_shougang.py | 47 +++ tests/standards/test_contracts.py | 40 ++ tests/standards/test_merged_resolver.py | 64 ++++ tests/standards/test_real_xlsx.py | 63 ++++ 12 files changed, 815 insertions(+), 224 deletions(-) create mode 100644 tests/standards/test_merged_resolver.py diff --git a/artifacts/generated/provenance/finance_standard_alignment.json b/artifacts/generated/provenance/finance_standard_alignment.json index 3fb623f..864ef80 100644 --- a/artifacts/generated/provenance/finance_standard_alignment.json +++ b/artifacts/generated/provenance/finance_standard_alignment.json @@ -35,7 +35,7 @@ 51 ], "standard_entry_ids": [ - "finance:业务.账户信息.单位标签信息.基本信息" + "finance:业务.账户信息..基本信息" ], "standard_levels": [ "L2" diff --git a/artifacts/generated/provenance/standard_build_summary.json b/artifacts/generated/provenance/standard_build_summary.json index b6d8b9a..1181702 100644 --- a/artifacts/generated/provenance/standard_build_summary.json +++ b/artifacts/generated/provenance/standard_build_summary.json @@ -3,8 +3,8 @@ "finance": { "legacy_dict_entries": 237, "legacy_dict_path_compression": { - "entries_at_legacy_depth": 1, - "entries_with_path_deeper_than_legacy_L1_L2_leaf": 236, + "entries_at_legacy_depth": 6, + "entries_with_path_deeper_than_legacy_L1_L2_leaf": 231, "note": "legacy financial_standards_dict stored L1-L2-leaf identity strings; the real standard has 三级子类 provenance nodes that the legacy digest dropped (canonical standard keeps every entry, standard_entry_id includes the real 三级)" }, "legacy_unparseable_level_values": [ diff --git a/docs/design/phase1_canonical_standard.md b/docs/design/phase1_canonical_standard.md index 25ff963..d4a756e 100644 --- a/docs/design/phase1_canonical_standard.md +++ b/docs/design/phase1_canonical_standard.md @@ -29,12 +29,15 @@ sample `data_level`。 "standard_data_level": "L1|L2|L3|L4|null", "raw_level": "2 | l | 3 4", "content": "…数据资源说明…", - "source": {"file": "…", "sheet": "…", "row": …} + "source": {"file": "…", "sheet": "…", "row": …}, + "raw_fields": { // 继承自合并组的事实,带 provenance + "level_2_definition": {"value": "…", "source_cell": "D93", "merged_range": "D93:D132", "start_row": 93, "end_row": 132, "inherited": true} + } } ``` 顶层:`dataset / id_strategy / standard_name / standard_source{file,sheet} / -fingerprint / entries[] / training_projection`。 +fingerprint / entries[] / training_projection / scoped_annotations[]`。 - **`standard_entry_id`** = 原始标准中的**真实身份**(finance 含真实三级子类 `finance:{L1}.{L2}.{L3}.{L4}`;shougang = guanji code)。唯一。 @@ -42,11 +45,29 @@ fingerprint / entries[] / training_projection`。 `DatasetConfig.identity_fields` 一致)。**不唯一**:多个标准 entry 可投影到同一 训练类别。 - **`training_projection`** = 派生视图 `{category_id: [standard_entry_id…]}`, - 把 237 个 finance entry 投影到 233 个训练类别(237→233 是显式投影,**不是** - 事实层丢信息)。 -- `fingerprint` = entries(不含 source 行号)+ projection 内容的 sha256; - **与输入顺序无关**(每个 entry 原样保留、按 `standard_entry_id` 排序, - 不再依赖"首见顺序")。 + 把 237 个 finance entry 投影到 233 个训练类别。 +- **`raw_fields`** = entry 从合并组**继承**的层级事实(finance 二级/三级定义; + shougang 一级/二级/三级定义 + 数据来源 resource),每项带 `source_cell` / + `merged_range` provenance——是网格事实,不是 leaf 私有标签。 +- **`scoped_annotations`** = 网格作用域注解(finance 备注 J、部门意见 K,当前 K 为空): + `{annotation_id, type, text, source_cell, merged_range, start_row, end_row, + applies_to_standard_entry_ids}`。合并格是一条源值作用于一个行范围,**绝不复制成 + 单个 leaf 的私有备注**。 +- `fingerprint` = entries(不含 source 行号)+ projection + annotations 的 + sha256;**与输入顺序无关**(按 entry id / annotation id 排序)。 + +### 1.1 列语义:hierarchy / leaf / scoped(merged-aware) + +| 源 | hierarchy-level(组级合并) | leaf-level(逐行) | scoped annotation | +| --- | --- | --- | --- | +| finance | B 一级子类 / C 二级子类 / D 二级定义 / E 三级子类 / F 三级定义 | G 四级子类 / H 内容 / I 安全级别 | J 备注(J55 单行 / J93:J132 40 行 / J168:J169 2 行)、K 部门意见(空) | +| shougang | B..G 一级分类+定义 / 二级分类+定义 / 三级分类+定义 | H 四级分类 / I 四级定义 / J 数据资源说明 / K 分级 / L 数据来源 | 无备注列 | + +- **merged 处理只作用于两个 standard source**(finance 指南、shougang 目录): + `MergedCellResolver` 对任意 cell 返回 value + anchor_cell + merged_range + + start/end row + inherited,reader 不再用手工 carry-forward。 +- **三个 sample source(部分金融数据 / 带分级分类… / 关基设施…用于测试)无业务级 + 合并**(`merged_ranges_total=0`;finance 样本仅标题 A1:A2 一条),样本层 zero 改动。 ## 2. 各数据集 standard source 状态 @@ -110,15 +131,18 @@ dataset-derived 行为,但**不伪装成 canonical standard**——不生成 ``` tests/standards/ - test_contracts.py round-trip / normalize / fingerprint / projection - test_build_finance.py 无损 237、三级叶保持、level 异常、同类别多 entry determinism、reader issues - test_build_shougang.py code/path/level、三级叶、no_code、reader issues、确定性 - test_align.py 对齐桶、多 entry 类别、不修改样本、unresolved evidence、路由 - test_checksum.py checksum manifest 校验(缺失/不符) - test_real_xlsx.py 真实 Excel + canonical 集成断言(raw 缺失时 skip) + test_merged_resolver.py MergedCellResolver anchor/inherited/range(tmp workbook) + test_contracts.py round-trip(含 raw_fields / scoped_annotations)/ normalize / fingerprint + test_build_finance.py D/F 层级定义继承、J55/J93:J132/J168:J169 作用域、level 异常、确定性 + test_build_shougang.py C/E/G 定义 + resource 保留、三级叶、no_code、确定性 + test_align.py 对齐桶、多 entry 类别、不修改样本、unresolved evidence、路由 + test_checksum.py checksum manifest 校验(缺失/不符) + test_real_xlsx.py 真实 Excel + canonical 集成断言(raw 缺失时 skip) + —— 含:3 个 sample source 无业务级 merged cells;J55=1/J93:J132=40/J168:J169=2 ``` -结果:`pytest tests/standards` → **37 passed**;全仓见 PR。 +结果:`pytest tests/standards` → **56 passed**;全仓 **298 passed, 2 skipped** +(skip=本地无 verl,既有)。 产物可重生成:`python -m script.standard.cli`(拒绝无 `--overwrite` 覆盖; 先行全部构建/对齐、后写盘;重复构建字节级一致)。 diff --git a/src/agent/standards/__init__.py b/src/agent/standards/__init__.py index 92db8bc..0495e2c 100644 --- a/src/agent/standards/__init__.py +++ b/src/agent/standards/__init__.py @@ -8,6 +8,7 @@ from .contracts import ( LEVELS, CanonicalStandard, + ScopedAnnotation, SourceRef, StandardCategory, StandardCategoryBuilder, @@ -17,6 +18,8 @@ strip_code, ) from .sources import ( + CellInfo, + MergedCellResolver, RawEntry, ReaderResult, read_finance_standard_guide, @@ -37,6 +40,7 @@ __all__ = [ "LEVELS", "CanonicalStandard", + "ScopedAnnotation", "SourceRef", "StandardCategory", "StandardCategoryBuilder", @@ -44,6 +48,8 @@ "compact", "normalize_standard_level", "strip_code", + "CellInfo", + "MergedCellResolver", "RawEntry", "ReaderResult", "read_finance_standard_guide", diff --git a/src/agent/standards/build.py b/src/agent/standards/build.py index efe4043..8b36142 100644 --- a/src/agent/standards/build.py +++ b/src/agent/standards/build.py @@ -1,32 +1,28 @@ """Canonical standard builders (Phase 1). -Turn raw standard rows (from ``sources``) into a LOSSESS, auditable -``CanonicalStandard``: -- ONE entry per real standard row — there is NO aggregation in the fact layer. +Turn raw standard rows (from merge-aware ``sources``) into a LOSSESS, +auditable ``CanonicalStandard``: +- ONE entry per real standard row — no aggregation in the fact layer. ``standard_entry_id`` is the true source identity; ``category_id`` is the - legacy training/registry alias (a projection, possibly shared by several - entries). The 237 finance rows therefore stay 237 entries; the 237→233 - ``training_projection`` is exposed as a DERIVED view for Phase 2. -- category_id continues the existing stable identity strategy (finance L1-L2-leaf - via DatasetConfig.identity_fields; shougang guanji code). -- path stores the TRUE source-hierarchy depth (empty 三级子类 omitted — no - invented padding). -- grading columns are normalized to L1..L4 with ``normalize_standard_level``; - unparseable values are reported and kept raw, never fixed or guessed. -- placeholder / malformed rows (shougang "——", NaN, missing code) are skipped - and reported in the build report — never silently repaired. Reader-level - issues are merged into the same report. - -Deterministic for identical input regardless of the order entries arrive in -(every entry is preserved and sorted by standard_entry_id; nothing depends on -"first seen"). The CLI writes every artifact only after all datasets build and -align successfully (fail-fast) and refuses to overwrite without --overwrite. + legacy training/registry alias (projection; Phase 2 decides membership). +- Hierarchy facts that a leaf INHERITS from a merged group (finance 二级/三级 + 定义, shougang 一级/二级/三级 定义, resource) are kept per entry in + ``raw_fields`` WITH their source-cell / merged-range provenance. +- GRID-scoped annotations (finance 备注 J, 部门意见 K) are kept at standard + level as ``ScopedAnnotation`` carrying their original merged range and the + entry ids they apply to — never copied into a single leaf as private info. +- Grading columns are normalized L1..L4; unparseable values are reported and + kept raw, never fixed. Placeholder / malformed rows (shougang \"——\" with no + code, NaN) are skipped and reported; reader-level issues are merged in. + +Deterministic: every entry preserved and sorted by standard_entry_id; +annotations sorted by annotation_id; no reliance on first-seen order. """ from __future__ import annotations import re -from collections import Counter +from collections import Counter, defaultdict from dataclasses import dataclass, field from pathlib import Path from typing import Any, Iterable, Mapping, Sequence @@ -34,6 +30,7 @@ from agent.task.identity import qualified_category_id from agent.standards.contracts import ( CanonicalStandard, + ScopedAnnotation, SourceRef, StandardCategory, StandardCategoryBuilder, @@ -106,22 +103,89 @@ def _dedupe_path(parts: Sequence[str]) -> tuple[str, ...]: return tuple(result) +def _entry_value(entry: Any, key: str) -> Any: + if hasattr(entry, key): + return getattr(entry, key) + if isinstance(entry, Mapping): + return entry.get(key) + return None + + +def _prov(entry: Any, key: str) -> Mapping[str, Any]: + provenance = _entry_value(entry, "provenance") or {} + value = provenance.get(key) + return value if isinstance(value, Mapping) else {} + + +def _raw_field(entry: Any, key: str): + """Build one raw_fields item ``{value, source_cell, merged_range, …}`` for + a non-empty inherited hierarchy field, else None.""" + value = clean(_entry_value(entry, key) or "") + if not value: + return None + info = dict(_prov(entry, key)) + info.pop("value", None) + item = {"value": value} + item.update(info) + return item + + +def _scoped_annotations( + dataset: str, + rows: Sequence[tuple[int, str, str, str, str, str, int, int]], +) -> tuple[ScopedAnnotation, ...]: + """Build gold-scope annotations from per-entry annotation sightings. + + Groups sightings by (type, text, merged_range); a merged range with one + anchor value yields exactly ONE annotation covering every entry in that + range. ``start_row/end_row`` come from the merged-range provenance (or the + cell's own row for a single cell), never from the observed member subset. + """ + groups: dict[tuple[str, str, str | None], list[tuple[int, str, str, int, int]]] = defaultdict(list) + for row, entry_id, type_, text, source_cell, merged_range, start, end in rows: + if not text: + continue + groups[(type_, text, merged_range)].append((row, entry_id, source_cell, start, end)) + + annotations: list[ScopedAnnotation] = [] + for (type_, text, merged_range), members in groups.items(): + members.sort(key=lambda m: (m[3] if m[3] is not None else m[0], m[0])) + start_rows = [m[3] if m[3] is not None else m[0] for m in members] + end_rows = [m[4] if m[4] is not None else m[0] for m in members] + start_row = min(start_rows) + end_row = max(end_rows) + source_cell = members[0][2] + annotation_id = f"{dataset}-{type_}-{start_row}-{end_row}" + annotations.append( + ScopedAnnotation( + annotation_id=annotation_id, + type=type_, + text=text, + source_cell=source_cell, + merged_range=merged_range, + start_row=start_row, + end_row=end_row, + applies_to_standard_entry_ids=tuple( + sorted(entry_id for _, entry_id, _, _, _ in members) + ), + ) + ) + ids = [a.annotation_id for a in annotations] + if len(set(ids)) != len(ids): + raise ValueError(f"duplicate annotation ids (internal grouping error): {ids}") + return tuple(sorted(annotations, key=lambda a: a.annotation_id)) + + def _finalize_report( report: StandardBuildReport, entries: Sequence[StandardCategory], reader_issues: Sequence[str], ) -> None: - """Level distribution + projection counts + merged reader issues. - - Deterministic: issues are sorted before persistence (see to_mapping). - """ level_counter: Counter[str] = Counter() for entry in entries: level_counter[entry.standard_data_level or "''"] += 1 report.level_distribution = dict(sorted(level_counter.items())) - report.training_categories = len( - {entry.category_id for entry in entries} - ) + report.training_categories = len({entry.category_id for entry in entries}) for detail in reader_issues: report.issues.append(BuildIssue(kind="reader_issue", detail=str(detail))) @@ -135,13 +199,11 @@ def build_finance_standard( standard_name: str = "金融行业数据安全分类分级标准指南", reader_issues: Sequence[str] = (), ) -> tuple[CanonicalStandard, StandardBuildReport]: - """Build the finance canonical standard from raw guide entries. + """Build the finance canonical standard from merge-aware raw entries. - One entry per raw row. ``standard_entry_id`` = L1.L2.L3.leaf (the TRUE - source identity, level_3 included); ``category_id`` = L1.L2.leaf (the - training alias, matching DatasetConfig.identity_fields). Several entries - may share a category_id (e.g. 业务/合约协议/基本信息 under five 三级): - they stay separate entries, exposed via ``training_projection``. + standard_entry_id = finance:L1.L2.L3.leaf (true identity incl. 三级); + category_id = finance:L1.L2.leaf (training alias). Inherited 二级/三级 + 定义 go to ``raw_fields``; 备注/部门意见 become ``scoped_annotations``. """ report = StandardBuildReport( dataset=dataset, @@ -152,23 +214,20 @@ def build_finance_standard( entries_read=len(entries), ) built: list[StandardCategory] = [] + annotation_rows: list[tuple[int, str, str, str, str, str, int, int]] = [] for entry in entries: - level_1 = clean(entry.level_1 if hasattr(entry, "level_1") else entry.get("level_1")) - level_2 = clean(entry.level_2 if hasattr(entry, "level_2") else entry.get("level_2")) - level_3 = clean(entry.level_3 if hasattr(entry, "level_3") else entry.get("level_3")) - leaf = clean(entry.leaf if hasattr(entry, "leaf") else entry.get("leaf")) - content = clean(entry.description if hasattr(entry, "description") else entry.get("description")) - raw_level = str(entry.raw_level if hasattr(entry, "raw_level") else entry.get("raw_level", "")).strip() - row = entry.row if hasattr(entry, "row") else entry.get("row") + level_1 = clean(_entry_value(entry, "level_1") or "") + level_2 = clean(_entry_value(entry, "level_2") or "") + level_3 = clean(_entry_value(entry, "level_3") or "") + leaf = clean(_entry_value(entry, "leaf") or "") + content = clean(_entry_value(entry, "description") or "") + raw_level = str(_entry_value(entry, "raw_level") or "").strip() + row = _entry_value(entry, "row") if not leaf: _add_issue(report, "empty_leaf_skipped", f"row {row}: empty leaf") continue - # TRUE identity keeps the real 三级子类 (empty slots stay empty parts) - standard_entry_id = qualified_category_id( - dataset, (level_1, level_2, level_3, leaf) - ) - # training alias: DatasetConfig.identity_fields = (level_1, level_2, level_4) + standard_entry_id = qualified_category_id(dataset, (level_1, level_2, level_3, leaf)) category_id = qualified_category_id(dataset, (level_1, level_2, leaf)) path = _dedupe_path((level_1, level_2, level_3, leaf)) level, raw_clean = normalize_standard_level(raw_level) @@ -179,6 +238,12 @@ def build_finance_standard( f"row {row} category {leaf!r}: raw level {raw_clean!r} kept as-is, " f"standard_data_level=null (not guessed)", ) + raw_fields: dict[str, Any] = {} + for key in ("level_2_definition", "level_3_definition"): + item = _raw_field(entry, key) + if item is not None: + raw_fields[key] = item + built.append( StandardCategoryBuilder( standard_entry_id=standard_entry_id, @@ -188,27 +253,38 @@ def build_finance_standard( description=content, code=None, raw_level=raw_level, + raw_fields=raw_fields, source_file=source_file, source_sheet=source_sheet, source_row=row, ).build() ) - # lossless fact layer: every entry preserved; determinism by entry-id order + # scoped-annotation sightings (remark / department opinion) + remark = clean(_entry_value(entry, "remark") or "") + opinion = clean(_entry_value(entry, "department_opinion") or "") + if remark: + _push_annotation_sighting( + annotation_rows, row, standard_entry_id, "remark", remark, + _prov(entry, "remark"), + ) + if opinion: + _push_annotation_sighting( + annotation_rows, row, standard_entry_id, "department_opinion", + opinion, _prov(entry, "department_opinion"), + ) + built.sort(key=lambda c: c.standard_entry_id) - duplicate_ids = {e.standard_entry_id for e in built} - if len(duplicate_ids) != len(built): - raise ValueError( - "finance standard_entry_id must be unique; got " - f"{len(built) - len(duplicate_ids)} duplicate(s)" - ) + _assert_unique_entry_ids(built, "finance") report.standard_entries_out = len(built) _finalize_report(report, built, reader_issues) + scoped = _scoped_annotations(dataset, annotation_rows) return CanonicalStandard( dataset=dataset, id_strategy="path", standard_source=SourceRef(file=source_file, sheet=source_sheet), standard_name=standard_name, entries=tuple(built), + scoped_annotations=scoped, ), report @@ -223,9 +299,8 @@ def build_shougang_standard( ) -> tuple[CanonicalStandard, StandardBuildReport]: """Build the shougang canonical standard from the raw guanji catalog. - standard_entry_id == category_id == guanji code (opaque identity). A - \"——\"/empty leaf cell means the leaf lives one level up (三级 carries the - real code/name); those are real categories resolved and kept, never lost. + standard_entry_id == category_id == guanji code. Inherited 一级/二级/三级 + 定义 and 数据来源(resource) are kept in ``raw_fields`` with provenance. """ report = StandardBuildReport( dataset=dataset, @@ -237,18 +312,17 @@ def build_shougang_standard( ) built: list[StandardCategory] = [] for entry in entries: - level_1 = clean(entry.level_1 if hasattr(entry, "level_1") else entry.get("level_1")) - level_2 = clean(entry.level_2 if hasattr(entry, "level_2") else entry.get("level_2")) - level_3 = clean(entry.level_3 if hasattr(entry, "level_3") else entry.get("level_3")) - raw_leaf = clean(entry.leaf if hasattr(entry, "leaf") else entry.get("leaf")) - description = clean(entry.description if hasattr(entry, "description") else entry.get("description")) - content = clean(entry.content if hasattr(entry, "content") else entry.get("content")) - raw_level = str(entry.raw_level if hasattr(entry, "raw_level") else entry.get("raw_level", "")).strip() - row = entry.row if hasattr(entry, "row") else entry.get("row") - - # a "——"/empty leaf cell means the leaf lives one level up (三级) + level_1 = clean(_entry_value(entry, "level_1") or "") + level_2 = clean(_entry_value(entry, "level_2") or "") + level_3 = clean(_entry_value(entry, "level_3") or "") + raw_leaf = clean(_entry_value(entry, "leaf") or "") + description = clean(_entry_value(entry, "description") or "") + content = clean(_entry_value(entry, "content") or "") + raw_level = str(_entry_value(entry, "raw_level") or "").strip() + row = _entry_value(entry, "row") + if not raw_leaf or raw_leaf.lower() in _PLACEHOLDER_NAMES or raw_leaf in _PLACEHOLDER_NAMES: - raw_leaf = clean(entry.level_3 if hasattr(entry, "level_3") else entry.get("level_3")) + raw_leaf = clean(_entry_value(entry, "level_3") or "") if not raw_leaf or raw_leaf in _PLACEHOLDER_NAMES: _add_issue( report, @@ -277,6 +351,12 @@ def build_shougang_standard( f"row {row} category {name!r}: raw level {raw_clean!r} kept as-is, " f"standard_data_level=null (not guessed)", ) + raw_fields: dict[str, Any] = {} + for key in ("level_1_definition", "level_2_definition", "level_3_definition", "resource"): + item = _raw_field(entry, key) + if item is not None: + raw_fields[key] = item + built.append( StandardCategoryBuilder( standard_entry_id=code, @@ -287,18 +367,14 @@ def build_shougang_standard( code=code, raw_level=raw_level, content=content, + raw_fields=raw_fields, source_file=source_file, source_sheet=source_sheet, source_row=row, ).build() ) built.sort(key=lambda c: c.standard_entry_id) - duplicate_ids = {e.standard_entry_id for e in built} - if len(duplicate_ids) != len(built): - raise ValueError( - "shougang standard_entry_id (code) must be unique; got " - f"{len(built) - len(duplicate_ids)} duplicate(s)" - ) + _assert_unique_entry_ids(built, "shougang") report.standard_entries_out = len(built) _finalize_report(report, built, reader_issues) return CanonicalStandard( @@ -310,14 +386,41 @@ def build_shougang_standard( ), report -def resolve_standard_dataset(dataset: str) -> str | None: - """Which canonical standard owns a dataset's category facts. +def _push_annotation_sighting( + rows: list[tuple[int, str, str, str, str, str]], + row: int, + entry_id: str, + type_: str, + text: str, + prov: Mapping[str, Any], +) -> None: + start = prov.get("start_row") + end = prov.get("end_row") + rows.append( + ( + row, + entry_id, + type_, + text, + str(prov.get("source_cell") or ""), + prov.get("merged_range"), + int(start) if start is not None else None, + int(end) if end is not None else None, + ) + ) - - finance -> finance - - shougang -> shougang - - infra -> shougang (reuses the shared guanji standard; no copy) - - pers_info -> None (no confirmed classification/grading standard) - """ + +def _assert_unique_entry_ids(built: Sequence[StandardCategory], dataset: str) -> None: + unique = {entry.standard_entry_id for entry in built} + if len(unique) != len(built): + raise ValueError( + f"{dataset} standard_entry_id must be unique; got " + f"{len(built) - len(unique)} duplicate(s)" + ) + + +def resolve_standard_dataset(dataset: str) -> str | None: + """Which canonical standard owns a dataset's category facts.""" return { "finance": "finance", "shougang": "shougang", diff --git a/src/agent/standards/contracts.py b/src/agent/standards/contracts.py index e9f0132..db6adab 100644 --- a/src/agent/standards/contracts.py +++ b/src/agent/standards/contracts.py @@ -97,6 +97,53 @@ def from_mapping(cls, value: Mapping[str, Any]) -> "SourceRef": ) +@dataclass(frozen=True) +class ScopedAnnotation: + """A GRID-scoped annotation from the source standard (e.g. finance 备注). + + A merged cell is a SINGLE source value that applies to a RANGE of rows + (and therefore to several standard entries). It must never be copied into + one leaf as if it were that leaf's private remark — the original merged + scope is preserved here. + """ + + annotation_id: str + type: str # e.g. "remark" / "department_opinion" + text: str + source_cell: str + merged_range: str | None + start_row: int + end_row: int + applies_to_standard_entry_ids: tuple[str, ...] = () + + def to_mapping(self) -> dict[str, Any]: + return { + "annotation_id": self.annotation_id, + "type": self.type, + "text": self.text, + "source_cell": self.source_cell, + "merged_range": self.merged_range, + "start_row": self.start_row, + "end_row": self.end_row, + "applies_to_standard_entry_ids": list(self.applies_to_standard_entry_ids), + } + + @classmethod + def from_mapping(cls, value: Mapping[str, Any]) -> "ScopedAnnotation": + return cls( + annotation_id=str(value.get("annotation_id", "") or ""), + type=str(value.get("type", "") or ""), + text=str(value.get("text", "") or ""), + source_cell=str(value.get("source_cell", "") or ""), + merged_range=value.get("merged_range"), + start_row=int(value.get("start_row", 0) or 0), + end_row=int(value.get("end_row", 0) or 0), + applies_to_standard_entry_ids=tuple( + str(item) for item in value.get("applies_to_standard_entry_ids", ()) + ), + ) + + @dataclass(frozen=True) class StandardCategory: """One entry of the LOSSESS canonical standard (one raw standard row). @@ -108,6 +155,10 @@ class StandardCategory: ``DatasetConfig.identity_fields``; shougang = same code). NOT unique — several distinct standard entries may project onto one training category (Phase 2 decision; Phase 1 keeps every entry). + - raw_fields: source-specific EXTRA fields (hierarchy definitions, + resource) with per-value provenance (source_cell / merged_range). These + are grid facts that a leaf inherits from its merged group; they are NOT + leaf-private labels. Scoped annotations (remarks) live at standard level. """ standard_entry_id: str @@ -120,6 +171,7 @@ class StandardCategory: raw_level: str = "" content: str = "" source: SourceRef = field(default_factory=SourceRef) + raw_fields: Mapping[str, Mapping[str, Any]] = field(default_factory=dict) def to_mapping(self) -> dict[str, Any]: mapping: dict[str, Any] = { @@ -135,11 +187,17 @@ def to_mapping(self) -> dict[str, Any]: } if self.content: mapping["content"] = self.content + if self.raw_fields: + mapping["raw_fields"] = { + key: dict(item) + for key, item in sorted(self.raw_fields.items()) + } return mapping @classmethod def from_mapping(cls, value: Mapping[str, Any]) -> "StandardCategory": source = value.get("source") or {} + raw_fields = value.get("raw_fields") or {} return cls( standard_entry_id=str(value.get("standard_entry_id", "") or ""), category_id=str(value.get("category_id", "") or ""), @@ -153,6 +211,11 @@ def from_mapping(cls, value: Mapping[str, Any]) -> "StandardCategory": source=SourceRef.from_mapping( source if isinstance(source, Mapping) else {} ), + raw_fields=( + {str(k): dict(v) for k, v in raw_fields.items()} + if isinstance(raw_fields, Mapping) + else {} + ), ) @@ -170,6 +233,7 @@ class CanonicalStandard: standard_source: SourceRef standard_name: str = "" entries: tuple[StandardCategory, ...] = () + scoped_annotations: tuple[ScopedAnnotation, ...] = () @property def categories(self) -> tuple[StandardCategory, ...]: @@ -210,6 +274,12 @@ def to_mapping(self) -> dict[str, Any]: "fingerprint": self.fingerprint(), "entries": [entry.to_mapping() for entry in self.entries], "training_projection": self.training_projection(), + "scoped_annotations": [ + annotation.to_mapping() + for annotation in sorted( + self.scoped_annotations, key=lambda a: a.annotation_id + ) + ], } @classmethod @@ -221,6 +291,12 @@ def from_mapping(cls, value: Mapping[str, Any]) -> "CanonicalStandard": if isinstance(item, Mapping) ) source = value.get("standard_source") or {} + raw_annotations = value.get("scoped_annotations", ()) + annotations = tuple( + ScopedAnnotation.from_mapping(item) + for item in raw_annotations + if isinstance(item, Mapping) + ) return cls( dataset=str(value.get("dataset", "") or ""), id_strategy=str(value.get("id_strategy", "") or ""), @@ -229,6 +305,7 @@ def from_mapping(cls, value: Mapping[str, Any]) -> "CanonicalStandard": source if isinstance(source, Mapping) else {} ), entries=entries, + scoped_annotations=annotations, ) def fingerprint(self) -> str: @@ -240,6 +317,12 @@ def fingerprint(self) -> str: for entry in sorted(self.entries, key=lambda e: e.standard_entry_id) ], "training_projection": self.training_projection(), + "scoped_annotations": [ + annotation.to_mapping() + for annotation in sorted( + self.scoped_annotations, key=lambda a: a.annotation_id + ) + ], } digest = hashlib.sha256() digest.update( @@ -265,6 +348,7 @@ class StandardCategoryBuilder: code: str | None = None raw_level: str = "" content: str = "" + raw_fields: Mapping[str, Mapping[str, Any]] = field(default_factory=dict) source_file: str = "" source_sheet: str = "" source_row: int | None = None @@ -281,6 +365,7 @@ def build(self) -> StandardCategory: standard_data_level=level, raw_level=clean(self.raw_level), content=self.content, + raw_fields=dict(self.raw_fields), source=SourceRef( file=self.source_file, sheet=self.source_sheet, row=self.source_row ), diff --git a/src/agent/standards/sources.py b/src/agent/standards/sources.py index 6399316..716fee9 100644 --- a/src/agent/standards/sources.py +++ b/src/agent/standards/sources.py @@ -1,20 +1,24 @@ """Raw standard readers (Phase 1): read the ORIGINAL standard workbooks into -plain raw-entry dicts. - -The canonical standard must be built from the original source workbooks -(data/raw), not from the already-compressed standards_map JSON digests -(financial_standards_dict.json / guanji_dict.json dropped real path depth and -grading columns). These readers only extract, never normalize semantics: -grading columns are kept as raw strings. - -Excel layout notes (verified against data/raw at 2026-08-20): -- finance 金融行业数据安全分类分级标准指南.xlsx / sheet "Table 1": - columns 一级子类|二级子类|二级定义|三级子类|三级定义|四级子类|内容|安全级别|备注|部门意见; - merged 一级/二级/三级 subclasses carry forward; "安全级别" header row is - "最低安全级别参考". -- shougang 关基-数据分类分级目录.xlsx / sheet "数据分类分级": - columns 一级分类|一级定义|二级分类|二级定义|三级分类|三级定义|四级分类|四级定义| - 数据资源说明(内容)|分级|数据资源; merged 一级/二级/三级 carry forward. +plain raw-entry dicts — MERGED-RANGE-AWARE. + +The standard workbooks use vertical cell merges for GROUP/hierarchy columns: +finance 金融行业数据安全分类分级标准指南 (B..F hierarchy + J remark, K dept), +shougang 关基-数据分类分级目录 (B..G hierarchy). A merged cell is ONE source +value owned by a RANGE of rows; ``MergedCellResolver`` expands it to every +covered cell while preserving the anchor and scope, so a leaf inherits its +group's definitions/remark without misattributing them as leaf-private. + +Leaf columns (四级子类 / 内容 / 安全级别 / 分级 / 数据资源说明 / 数据来源) are +per-row cells (no merges). The three SAMPLE-source workbooks (部分金融数据 / +带分级分类… / 关基设施…用于测试) have no business-level merges and are NOT read +by this module. + +Excel layout (verified against data/raw at 2026-08-20): +- finance (sheet Table 1): B一级子类 C二级子类 D二级定义 E三级子类 F三级定义 + G四级子类 H内容 I安全级别 J备注 K部门意见. +- shougang (sheet 数据分类分级): B一级分类 C一级定义 D二级分类 E二级定义 + F三级分类(G(三级定义) H四级分类 I四级定义 J数据资源说明 K分级 L数据来源. + A "——"/empty 四级 cell means the leaf lives at 三级 (code carried by G). """ from __future__ import annotations @@ -28,10 +32,87 @@ FINANCE_SHEET = "Table 1" SHOUGANG_SHEET = "数据分类分级" +# finance columns (1-based): letter -> semantic key +FINANCE_COLS = { + "L1": "B", "L2": "C", "L2_DEF": "D", "L3": "E", "L3_DEF": "F", + "LEAF": "G", "DESC": "H", "LEVEL": "I", "REMARK": "J", "OPINION": "K", +} +SHOUGANG_COLS = { + "L1": "B", "L1_DEF": "C", "L2": "D", "L2_DEF": "E", "L3": "F", + "L3_DEF": "G", "LEAF": "H", "LEAF_DEF": "I", "CONTENT": "J", + "LEVEL": "K", "RESOURCE": "L", +} + + +@dataclass(frozen=True) +class CellInfo: + """One cell resolved through the merged-grid: value + scope provenance.""" + + value: Any = None + anchor_cell: str = "" + merged_range: str | None = None + start_row: int | None = None + end_row: int | None = None + inherited: bool = False + + def to_mapping(self) -> dict[str, Any]: + return { + "value": self.value, + "anchor_cell": self.anchor_cell, + "merged_range": self.merged_range, + "start_row": self.start_row, + "end_row": self.end_row, + "inherited": self.inherited, + } + + +class MergedCellResolver: + """Resolve any cell of a worksheet with merged-range awareness. + + Non-anchor cells inside a merged range return the ANCHOR's value plus the + merged scope (anchor_cell / merged_range / start..end / inherited=True). + Plain cells return their own value with inherited=False and no range. + """ + + def __init__(self, workbook, sheet: str): + if sheet not in workbook.sheetnames: + raise ValueError(f"sheet {sheet!r} not found") + self._sheet = workbook[sheet] + self._ranges = list(self._sheet.merged_cells.ranges) + + def close(self) -> None: + if self._sheet is not None: + pass # workbook managed by caller + + def cell(self, row: int, col) -> CellInfo: + """``col`` is a 1-based integer or a column letter (e.g. 'J').""" + from openpyxl.utils import column_index_from_string, get_column_letter + + if isinstance(col, str): + col = column_index_from_string(col) + cell = self._sheet.cell(row=row, column=col) + for rng in self._ranges: + if rng.min_row <= row <= rng.max_row and rng.min_col <= col <= rng.max_col: + anchor_value = self._sheet.cell(rng.min_row, rng.min_col).value + return CellInfo( + value=anchor_value, + anchor_cell=f"{get_column_letter(rng.min_col)}{rng.min_row}", + merged_range=str(rng), + start_row=rng.min_row, + end_row=rng.max_row, + inherited=not (row == rng.min_row and col == rng.min_col), + ) + return CellInfo( + value=cell.value, + anchor_cell=f"{get_column_letter(col)}{row}", + inherited=False, + ) + @dataclass(frozen=True) class RawEntry: - """One leaf row of a raw standard, with its traceable origin.""" + """One leaf row of a raw standard, with its traceable origin and the + hierarchy/annotation fields it INHERITS from merged groups.""" level_1: str = "" level_2: str = "" @@ -41,8 +122,14 @@ class RawEntry: content: str = "" raw_level: str = "" resource: str = "" + level_1_definition: str = "" + level_2_definition: str = "" + level_3_definition: str = "" + remark: str = "" + department_opinion: str = "" sheet: str = "" row: int | None = None + provenance: Mapping[str, Mapping[str, Any]] = field(default_factory=dict) def to_mapping(self) -> dict[str, Any]: return { @@ -54,144 +141,158 @@ def to_mapping(self) -> dict[str, Any]: "content": self.content, "raw_level": self.raw_level, "resource": self.resource, + "level_1_definition": self.level_1_definition, + "level_2_definition": self.level_2_definition, + "level_3_definition": self.level_3_definition, + "remark": self.remark, + "department_opinion": self.department_opinion, "sheet": self.sheet, "row": self.row, + "provenance": dict(self.provenance), } @dataclass(frozen=True) class ReaderResult: - entries: tuple[RawEntry, ...] + entries: tuple[RawEntry, ...] = () issues: tuple[str, ...] = () -def _open_xlsx(path: Path, sheet: str): - """Lazily import openpyxl so non-Excel code paths never require it.""" +def _open_xlsx(path: Path): + """Load a workbook with cached values (data_only) for merged-range reads.""" try: import openpyxl except ImportError as exc: # pragma: no cover - optional dependency raise RuntimeError("reading raw standard workbooks requires openpyxl") from exc - workbook = openpyxl.load_workbook(path, read_only=True, data_only=True) - if sheet not in workbook.sheetnames: - raise ValueError(f"sheet {sheet!r} not found in {path}") - return workbook, workbook[sheet] + return openpyxl.load_workbook(path, data_only=True) -def read_finance_standard_guide(path: str | Path) -> ReaderResult: - """Extract leaf rows of the finance grading-standard guide.""" - workbook, sheet = _open_xlsx(Path(path), FINANCE_SHEET) - rows = list(sheet.iter_rows(values_only=True)) - workbook.close() +def _prov(cell_info: CellInfo) -> Mapping[str, Any]: + return { + "source_cell": cell_info.anchor_cell, + "merged_range": cell_info.merged_range, + "start_row": cell_info.start_row, + "end_row": cell_info.end_row, + "inherited": cell_info.inherited, + } + +def read_finance_standard_guide(path: str | Path) -> ReaderResult: + """Extract leaf rows of the finance grading-standard guide (merge-aware).""" + workbook = _open_xlsx(Path(path)) + resolver = MergedCellResolver(workbook, FINANCE_SHEET) entries: list[RawEntry] = [] issues: list[str] = [] - level_1 = level_2 = level_3 = "" - for index, row in enumerate(rows): - if index < 2: # two header rows - continue - if row[1] is not None: - level_1 = clean(row[1]) - if row[2] is not None: - level_2 = clean(row[2]) - if row[4] is not None: - level_3 = clean(row[4]) - if row[6] is None: - continue - leaf = clean(row[6]) - if not leaf: - issues.append(f"finance row {index + 1}: empty leaf") - continue - content = clean(row[7]) if row[7] is not None else "" - raw_level = str(row[8]).strip() if row[8] is not None else "" - entries.append( - RawEntry( - level_1=level_1, - level_2=level_2, - level_3=level_3, - leaf=leaf, - description=content, - raw_level=raw_level, - sheet=FINANCE_SHEET, - row=index + 1, + try: + for row in range(3, (resolver._sheet.max_row or 2) + 1): + leaf = clean(resolver.cell(row, FINANCE_COLS["LEAF"]).value) + if not leaf: + continue # trailing rows / headers; not a leaf + l1i = resolver.cell(row, FINANCE_COLS["L1"]) + l2i = resolver.cell(row, FINANCE_COLS["L2"]) + l2di = resolver.cell(row, FINANCE_COLS["L2_DEF"]) + l3i = resolver.cell(row, FINANCE_COLS["L3"]) + l3di = resolver.cell(row, FINANCE_COLS["L3_DEF"]) + li = resolver.cell(row, FINANCE_COLS["DESC"]) + lvi = resolver.cell(row, FINANCE_COLS["LEVEL"]) + ri = resolver.cell(row, FINANCE_COLS["REMARK"]) + oi = resolver.cell(row, FINANCE_COLS["OPINION"]) + entries.append( + RawEntry( + level_1=clean(l1i.value), + level_2=clean(l2i.value), + level_3=clean(l3i.value), + leaf=leaf, + description=clean(li.value), + raw_level=str(lvi.value).strip() if lvi.value is not None else "", + level_2_definition=clean(l2di.value), + level_3_definition=clean(l3di.value), + remark=clean(ri.value), + department_opinion=clean(oi.value), + sheet=FINANCE_SHEET, + row=row, + provenance={ + "level_2_definition": _prov(l2di), + "level_3_definition": _prov(l3di), + "remark": _prov(ri), + "department_opinion": _prov(oi), + }, + ) ) - ) + finally: + workbook.close() return ReaderResult(tuple(entries), tuple(issues)) -@dataclass(frozen=True) -class _LevelBox: - name: str = "" - definition: str = "" - - def read_guanji_catalog(path: str | Path) -> ReaderResult: - """Extract leaf rows of the shougang (关基) grading catalog. - - A leaf is the DEEPEST level that carries a real name: the catalog encodes - categories whose leaf sits at the 三级 level with a literal "——" in the - 四级 cell (e.g. 合同归并(B1-2), 合同跟踪(B1-5), 热轧作业计划(B3-3)). - Those rows are real categories, NOT placeholders; their code/name come - from the 三级 cell. The leaf's description is the definition column of - its own level (四级 -> col 8; 三级 -> col 6). Literal cell text is kept - lossless (no quality fixing). - """ - workbook, sheet = _open_xlsx(Path(path), SHOUGANG_SHEET) - rows = list(sheet.iter_rows(values_only=True)) - workbook.close() + """Extract leaf rows of the shougang (关基) grading catalog (merge-aware). + A \"——\"/empty 四级 cell means the leaf lives at 三级 (the code/name is + carried by the 三级 column, which is itself a merged group cell). + """ + workbook = _open_xlsx(Path(path)) + resolver = MergedCellResolver(workbook, SHOUGANG_SHEET) entries: list[RawEntry] = [] issues: list[str] = [] - level_1 = _LevelBox() - level_2 = _LevelBox() - level_3 = _LevelBox() - for index, row in enumerate(rows): - if index < 2: # title + header rows - continue - if len(row) < 11: - issues.append(f"shougang row {index + 1}: too few columns") - continue - if row[1] is not None: - level_1 = _LevelBox(clean(str(row[1])), clean(str(row[2])) if row[2] is not None else "") - if row[3] is not None: - level_2 = _LevelBox(clean(str(row[3])), clean(str(row[4])) if row[4] is not None else "") - if row[5] is not None: - level_3 = _LevelBox(clean(str(row[5])), clean(str(row[6])) if row[6] is not None else "") - if row[7] is None: - continue - level_4 = _LevelBox(clean(str(row[7])), clean(str(row[8])) if row[8] is not None else "") - # deepest real level is the leaf (non-empty, not the "——" marker) - candidates = [ - (level_4.name, level_4.definition, "level_4"), - (level_3.name, level_3.definition, "level_3"), - (level_2.name, level_2.definition, "level_2"), - ] - leaf_name, leaf_definition, leaf_level = next( - (c for c in candidates if c[0] and c[0] != "——"), - ("", "", ""), - ) - if not leaf_name: - issues.append(f"shougang row {index + 1}: no real leaf level (all ——)") - continue - # keep only levels above/at the leaf; deeper levels are left empty - resolved_level_3 = level_3.name if leaf_level in ("level_3", "level_4") else "" - content = clean(str(row[9])) if row[9] is not None else "" - raw_level = str(row[10]).strip() if row[10] is not None else "" - resource = clean(str(row[11])) if len(row) > 11 and row[11] is not None else "" - entries.append( - RawEntry( - level_1=level_1.name, - level_2=level_2.name, - level_3=resolved_level_3, - leaf=leaf_name, - description=leaf_definition, - content=content, - raw_level=raw_level, - resource=resource, - sheet=SHOUGANG_SHEET, - row=index + 1, + try: + for row in range(3, (resolver._sheet.max_row or 2) + 1): + l1i = resolver.cell(row, SHOUGANG_COLS["L1"]) + l2i = resolver.cell(row, SHOUGANG_COLS["L2"]) + l3i = resolver.cell(row, SHOUGANG_COLS["L3"]) + leaf_h = resolver.cell(row, SHOUGANG_COLS["LEAF"]) + leaf_h_def = resolver.cell(row, SHOUGANG_COLS["LEAF_DEF"]) + if (leaf_h.value is None or clean(leaf_h.value) in ("", "——")): + # leaf sits at 三级: name+code and its definition come from G + if l3i.value is None or not clean(l3i.value): + issues.append(f"shougang row {row}: no real leaf level") + continue + leaf = clean(l3i.value) + description = clean( + resolver.cell(row, SHOUGANG_COLS["L3_DEF"]).value + ) + else: + leaf = clean(leaf_h.value) + description = clean(leaf_h_def.value) + content = clean(resolver.cell(row, SHOUGANG_COLS["CONTENT"]).value) + raw_level = str(resolver.cell(row, SHOUGANG_COLS["LEVEL"]).value).strip() + resource = clean(resolver.cell(row, SHOUGANG_COLS["RESOURCE"]).value) + l1di = resolver.cell(row, SHOUGANG_COLS["L1_DEF"]) + l2di = resolver.cell(row, SHOUGANG_COLS["L2_DEF"]) + l3di = resolver.cell(row, SHOUGANG_COLS["L3_DEF"]) + lresi = resolver.cell(row, SHOUGANG_COLS["RESOURCE"]) + entries.append( + RawEntry( + level_1=clean(l1i.value), + level_2=clean(l2i.value), + level_3=clean(l3i.value), + leaf=leaf, + description=description, + content=content, + raw_level=raw_level, + resource=resource, + level_1_definition=clean(l1di.value), + level_2_definition=clean(l2di.value), + level_3_definition=clean(l3di.value), + sheet=SHOUGANG_SHEET, + row=row, + provenance={ + "level_1_definition": _prov(l1di), + "level_2_definition": _prov(l2di), + "level_3_definition": _prov(l3di), + "resource": _prov(lresi), + }, + ) ) - ) + finally: + workbook.close() return ReaderResult(tuple(entries), tuple(issues)) -__all__ = ["RawEntry", "ReaderResult", "read_finance_standard_guide", "read_guanji_catalog"] +__all__ = [ + "CellInfo", + "MergedCellResolver", + "RawEntry", + "ReaderResult", + "read_finance_standard_guide", + "read_guanji_catalog", +] diff --git a/tests/standards/test_build_finance.py b/tests/standards/test_build_finance.py index 391fef6..f69b44a 100644 --- a/tests/standards/test_build_finance.py +++ b/tests/standards/test_build_finance.py @@ -137,3 +137,61 @@ def test_reader_issues_merged_into_build_report(): ) issues = report.to_mapping()["issues"] assert any(i["kind"] == "reader_issue" and "row 999" in i["detail"] for i in issues) + + +def _entry_with_hierarchy(levels, leaf, row, **extra): + e = _entry(level_1="业务", level_2="金融监管和服务", level_3="反洗钱业务信息", + leaf=leaf, raw_level="3", row=row) + e.update(extra) + return e + + +def test_finance_hierarchy_definitions_kept_in_raw_fields_with_provenance(): + prov = { + "level_2_definition": {"source_cell": "D93", "merged_range": "D93:D132", "start_row": 93, "end_row": 132, "inherited": True}, + "level_3_definition": {"source_cell": "F93", "merged_range": "F93:F99", "start_row": 93, "end_row": 99, "inherited": True}, + } + entries = [ + _entry_with_hierarchy( + ["业务", "金融监管和服务", "反洗钱业务信息"], "分类考核评级信息", 93, + level_2_definition="金融监管和服务域定义", + level_3_definition="反洗钱业务定义", + provenance=prov, + ) + ] + standard, _ = build_finance_standard(entries, source_file="f", source_sheet="Table 1") + entry = standard.entries[0] + assert entry.raw_fields["level_2_definition"] == { + "value": "金融监管和服务域定义", "source_cell": "D93", + "merged_range": "D93:D132", "start_row": 93, "end_row": 132, "inherited": True, + } + assert entry.raw_fields["level_3_definition"]["source_cell"] == "F93" + # remark is NOT a leaf-private raw field (it is a scoped annotation) + assert "remark" not in entry.raw_fields + + +def test_finance_scoped_annotations_group_by_merged_range(): + # three sightings: two share the J93:J132 merged range, one is a single cell + prov_merged = {"remark": {"source_cell": "J93", "merged_range": "J93:J132", "start_row": 93, "end_row": 132}} + prov_single = {"remark": {"source_cell": "J55", "merged_range": None, "start_row": 55, "end_row": 55}} + entries = [ + _entry_with_hierarchy(["业务", "金融监管和服务", "反洗钱业务信息"], "分类考核评级信息", 93, remark="宜从高设置", provenance=prov_merged), + _entry_with_hierarchy(["业务", "金融监管和服务", "反洗钱业务信息"], "行政监管信息", 100, remark="宜从高设置", provenance=prov_merged), + _entry(level_1="客户", level_2="个人", level_3="个人身份鉴别信息", leaf="特有账户信息", remark="宜从高设置", provenance=prov_single, row=55), + ] + standard, _ = build_finance_standard(entries, source_file="f", source_sheet="Table 1") + assert len(standard.scoped_annotations) == 2 + by_id = {a.annotation_id: a for a in standard.scoped_annotations} + merged = by_id["finance-remark-93-132"] + assert merged.merged_range == "J93:J132" + assert merged.start_row == 93 and merged.end_row == 132 + assert len(merged.applies_to_standard_entry_ids) == 2 + single = by_id["finance-remark-55-55"] + assert single.merged_range is None + assert len(single.applies_to_standard_entry_ids) == 1 + + +def test_finance_empty_department_opinion_produces_no_annotation(): + entries = [_entry(level_1="客户", level_2="个人", level_3="个人身份鉴别信息", leaf="特有账户信息", row=55)] + standard, _ = build_finance_standard(entries, source_file="f", source_sheet="Table 1") + assert standard.scoped_annotations == () diff --git a/tests/standards/test_build_shougang.py b/tests/standards/test_build_shougang.py index b50a0ba..d65acd8 100644 --- a/tests/standards/test_build_shougang.py +++ b/tests/standards/test_build_shougang.py @@ -92,3 +92,50 @@ def test_reader_issues_merged_into_build_report(): ) issues = report.to_mapping()["issues"] assert any(i["kind"] == "reader_issue" and "row 7" in i["detail"] for i in issues) + + +def test_shougang_hierarchy_definitions_and_resource_preserved(): + prov = { + "level_1_definition": {"source_cell": "C3", "merged_range": "C3:C6"}, + "level_2_definition": {"source_cell": "E3", "merged_range": "E3:E4"}, + "level_3_definition": {"source_cell": "G3", "merged_range": "G3:G3"}, + "resource": {"source_cell": "L3", "merged_range": None, "start_row": None, "end_row": None, "inherited": False}, + } + entries = [ + _entry( + level_1="研发数据域(A)", level_2="产品研发(A1)", level_3="科研检验(A1-1)", + leaf="科研设备预约管理(A1-1-1)", raw_level="2", row=3, + level_1_definition="研发域定义", level_2_definition="产品研发定义", + level_3_definition="科研检验定义", resource="科研实验室", + provenance=prov, + ) + ] + standard, _ = build_shougang_standard(entries, source_file="c", source_sheet="数据分类分级") + entry = standard.entries[0] + assert entry.raw_fields["level_1_definition"]["source_cell"] == "C3" + assert entry.raw_fields["level_2_definition"]["merged_range"] == "E3:E4" + assert entry.raw_fields["level_3_definition"]["value"] == "科研检验定义" + assert entry.raw_fields["resource"] == { + "value": "科研实验室", "source_cell": "L3", "merged_range": None, + "start_row": None, "end_row": None, "inherited": False, + } + + +def test_shougang_level_three_leaf_keeps_inherited_definitions_and_resource(): + entries = [ + _entry( + level_1="生产数据域(B)", level_2="生产合同(订单)(B1)", + level_3="合同归并(B1-2)", leaf="——", raw_level="3", row=9, + level_1_definition="生产域定义", level_2_definition="合同域定义", + level_3_definition="合同归并定义", resource="制造管理系统", + provenance={ + "level_1_definition": {"source_cell": "C7", "merged_range": "C7:C82"}, + "resource": {"source_cell": "L9", "merged_range": None}, + }, + ) + ] + standard, _ = build_shougang_standard(entries, source_file="c", source_sheet="数据分类分级") + entry = standard.by_entry_id()["B1-2"] + assert entry.raw_fields["level_1_definition"]["value"] == "生产域定义" + assert entry.raw_fields["resource"]["value"] == "制造管理系统" + assert entry.raw_fields["level_3_definition"]["value"] == "合同归并定义" diff --git a/tests/standards/test_contracts.py b/tests/standards/test_contracts.py index 5e098ee..3c79949 100644 --- a/tests/standards/test_contracts.py +++ b/tests/standards/test_contracts.py @@ -87,11 +87,51 @@ def test_round_trip_preserves_level_and_source(): raw_level="3", content="资源说明", source=SourceRef(file="x.xlsx", sheet="数据分类分级", row=7), + raw_fields={ + "level_3_definition": { + "value": "科研检验定义", "source_cell": "G7", "merged_range": None, + } + }, ) rebuilt = StandardCategory.from_mapping(copy.deepcopy(entry.to_mapping())) assert rebuilt == entry +def test_round_trip_preserves_scoped_annotations(): + from agent.standards.contracts import ScopedAnnotation + + annotation = ScopedAnnotation( + annotation_id="finance-remark-93-132", + type="remark", + text="鉴于其特殊性", + source_cell="J93", + merged_range="J93:J132", + start_row=93, + end_row=132, + applies_to_standard_entry_ids=("a", "b"), + ) + standard = CanonicalStandard( + dataset="finance", id_strategy="path", standard_source=SourceRef(), + entries=(_category("a", "L2"), _category("b", "L3")), + scoped_annotations=(annotation,), + ) + mapping = standard.to_mapping() + rebuilt = CanonicalStandard.from_mapping(copy.deepcopy(mapping)) + assert rebuilt.scoped_annotations == (annotation,) + assert rebuilt.fingerprint() == standard.fingerprint() + + +def test_fingerprint_includes_raw_fields_and_annotations(): + a = _category("A", "L1") + b = StandardCategory( + standard_entry_id="A", category_id="A", name="A", path=("A",), + raw_fields={"level_2_definition": {"value": "x", "source_cell": "D2"}}, + ) + sa = CanonicalStandard("s", "code", standard_source=SourceRef(), entries=(a,)) + sb = CanonicalStandard("s", "code", standard_source=SourceRef(), entries=(b,)) + assert sa.fingerprint() != sb.fingerprint() # raw_fields are content, not source + + def test_fingerprint_independent_of_source_rows(): a = _category("A", "L1", row=1) b = _category("A", "L1", row=99) # same content, different source row diff --git a/tests/standards/test_merged_resolver.py b/tests/standards/test_merged_resolver.py new file mode 100644 index 0000000..6e72488 --- /dev/null +++ b/tests/standards/test_merged_resolver.py @@ -0,0 +1,64 @@ +"""Phase 1 canonical standard — MergedCellResolver unit tests (hermetic). + +Builds a tiny workbook on the fly, so these tests need openpyxl but no repo +data (skipped when openpyxl is missing). +""" + +from __future__ import annotations + +import pytest + +openpyxl = pytest.importorskip("openpyxl") + +from agent.standards.sources import MergedCellResolver # noqa: E402 + + +@pytest.fixture() +def wb(tmp_path): + """A sheet with one vertical merge B2:B4 and one plain cell C2.""" + path = tmp_path / "t.xlsx" + workbook = openpyxl.Workbook() + ws = workbook.active + ws.title = "S" + ws["B2"] = "group-value" + ws["C2"] = "plain-value" + ws["C3"] = "plain-other" + ws.merge_cells("B2:B4") + workbook.save(path) + workbook.close() + return path + + +def test_anchor_cell(wb): + resolver = MergedCellResolver(openpyxl.load_workbook(wb, data_only=True), "S") + info = resolver.cell(2, "B") + assert info.value == "group-value" + assert info.anchor_cell == "B2" + assert info.merged_range == "B2:B4" + assert info.start_row == 2 and info.end_row == 4 + assert info.inherited is False + assert info.to_mapping()["value"] == "group-value" + + +def test_inherited_cell(wb): + resolver = MergedCellResolver(openpyxl.load_workbook(wb, data_only=True), "S") + info = resolver.cell(4, "B") # non-anchor inside the merge + assert info.value == "group-value" # anchor value expanded + assert info.anchor_cell == "B2" + assert info.merged_range == "B2:B4" + assert info.inherited is True + + +def test_plain_cell(wb): + resolver = MergedCellResolver(openpyxl.load_workbook(wb, data_only=True), "S") + info = resolver.cell(2, "C") + assert info.value == "plain-value" + assert info.anchor_cell == "C2" + assert info.merged_range is None + assert info.inherited is False + assert info.start_row is None + + +def test_column_letter_or_index_accepted(wb): + resolver = MergedCellResolver(openpyxl.load_workbook(wb, data_only=True), "S") + assert resolver.cell(2, "B").value == resolver.cell(2, 2).value == "group-value" diff --git a/tests/standards/test_real_xlsx.py b/tests/standards/test_real_xlsx.py index 4ea7093..b05495b 100644 --- a/tests/standards/test_real_xlsx.py +++ b/tests/standards/test_real_xlsx.py @@ -159,3 +159,66 @@ def test_build_deterministic_on_real_inputs(built): ) assert len(f2.entries) == 237 assert len(s2.entries) == 234 + + +@pytest.mark.parametrize( + "annotation_id,expected_range,expected_applies", + [ + ("finance-remark-55-55", None, 1), + ("finance-remark-93-132", "J93:J132", 40), + ("finance-remark-168-169", "J168:J169", 2), + ], +) +def test_finance_remark_scope_reproduced(built, annotation_id, expected_range, expected_applies): + finance, _, _, _ = built + by_id = {a.annotation_id: a for a in finance.scoped_annotations} + assert annotation_id in by_id + annotation = by_id[annotation_id] + assert annotation.merged_range == expected_range + assert len(annotation.applies_to_standard_entry_ids) == expected_applies + assert all(e.standard_entry_id in annotation.applies_to_standard_entry_ids + for e in finance.entries + if annotation.start_row <= e.source.row <= annotation.end_row) + + +def test_finance_row93_inherited_definitions_with_provenance(built): + finance, _, _, _ = built + entry = finance.by_entry_id()["finance:业务.金融监管和服务.反洗钱业务信息.分类考核评级信息"] + l2 = entry.raw_fields["level_2_definition"] + l3 = entry.raw_fields["level_3_definition"] + assert l2["source_cell"] == "D93" and l2["merged_range"] == "D93:D132" + assert l3["source_cell"] == "F93" + # remark is a scoped annotation, NOT a leaf raw field + assert "remark" not in entry.raw_fields + + +def test_shougang_inherited_definitions_and_resource_preserved(built): + _, _, shougang, _ = built + for code in ("A1-1-1", "B1-2"): + entry = shougang.by_entry_id()[code] + for key in ("level_1_definition", "level_2_definition", "level_3_definition", "resource"): + assert key in entry.raw_fields, (code, key) + assert entry.raw_fields["resource"]["value"] + assert entry.raw_fields["level_1_definition"]["source_cell"].startswith("C") + + +@pytest.mark.parametrize( + "xlsx,mins", + [ + (RAW / "部分金融数据.xlsx", 1), # one title merge (A1:A2) only + (RAW / "带分级分类的个人基础信息样本190条.xlsx", 0), + (RAW / "关基设施数据分类分级-不包含训练-用于测试(1).xlsx", 0), + ], +) +def test_sample_sources_have_no_business_merged_cells(xlsx, mins): + openpyxl = pytest.importorskip("openpyxl") + if not xlsx.is_file(): + pytest.skip("raw sample workbook not present") + workbook = openpyxl.load_workbook(xlsx) + try: + ws = workbook[workbook.sheetnames[0]] + # sample sources must not use group-level merges; the finance sample's + # single A1:A2 is a title, not a business column + assert len(ws.merged_cells.ranges) <= mins + finally: + workbook.close() From 1b11a258d897615a0b7deb74c4629b924cf3cc6b Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?=E6=9B=BE=E7=AB=8B=E5=AE=8F?= <曾立宏@buaa.edu.cn> Date: Thu, 20 Aug 2026 16:17:17 +0800 Subject: [PATCH 4/4] fix(standards): scope-key annotation grouping for unmerged cells MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Two unmerged cells with identical text (merged_range=None) were grouped into ONE annotation spanning both rows (e.g. 55..200), which breaks the lossless scope contract. Group by (type, text, scope_key) where scope_key is the merged range ('J93:J132') or the cell itself ('J55') for an unmerged cell. - _scoped_annotations(): scope_key grouping; merged_range=None kept per cell - _push_annotation_sighting(): type annotation corrected to 8 items - shougang reader: guard empty 分级 cell (None -> '' instead of 'None') - regression test: J5 + J10 same text, unmerged -> 2 annotations, each its row - tests/standards 57 passed; full suite 299 passed, 2 skipped (pre-existing) --- src/agent/standards/build.py | 24 +++++++++++++++--------- src/agent/standards/sources.py | 3 ++- tests/standards/test_build_finance.py | 25 +++++++++++++++++++++++++ 3 files changed, 42 insertions(+), 10 deletions(-) diff --git a/src/agent/standards/build.py b/src/agent/standards/build.py index 8b36142..893a401 100644 --- a/src/agent/standards/build.py +++ b/src/agent/standards/build.py @@ -136,32 +136,38 @@ def _scoped_annotations( ) -> tuple[ScopedAnnotation, ...]: """Build gold-scope annotations from per-entry annotation sightings. - Groups sightings by (type, text, merged_range); a merged range with one - anchor value yields exactly ONE annotation covering every entry in that - range. ``start_row/end_row`` come from the merged-range provenance (or the - cell's own row for a single cell), never from the observed member subset. + Groups sightings by ``(type, text, scope_key)`` where scope_key is the + merged range ("J93:J132") — or the cell itself ("J55") for an unmerged + cell. Two unmerged cells with identical text are therefore DIFFERENT + annotations (each keeps its own row), never merged into one spanning + range. ``start_row/end_row`` come from the merged-range provenance or the + single cell's own row, never from the observed member subset. """ - groups: dict[tuple[str, str, str | None], list[tuple[int, str, str, int, int]]] = defaultdict(list) + groups: dict[tuple[str, str, str], list[tuple[int, str, str, int, int]]] = defaultdict(list) for row, entry_id, type_, text, source_cell, merged_range, start, end in rows: if not text: continue - groups[(type_, text, merged_range)].append((row, entry_id, source_cell, start, end)) + scope_key = merged_range or source_cell or f"cell-{row}" + groups[(type_, text, scope_key)].append((row, entry_id, source_cell, start, end)) annotations: list[ScopedAnnotation] = [] - for (type_, text, merged_range), members in groups.items(): + for (type_, text, scope_key), members in groups.items(): members.sort(key=lambda m: (m[3] if m[3] is not None else m[0], m[0])) start_rows = [m[3] if m[3] is not None else m[0] for m in members] end_rows = [m[4] if m[4] is not None else m[0] for m in members] start_row = min(start_rows) end_row = max(end_rows) source_cell = members[0][2] + # a merged range always contains ':' (e.g. "J93:J132"); a plain cell + # ("J55") is not a range -> merged_range=None keeps the single scope + merged_range = scope_key if ":" in scope_key else None annotation_id = f"{dataset}-{type_}-{start_row}-{end_row}" annotations.append( ScopedAnnotation( annotation_id=annotation_id, type=type_, text=text, - source_cell=source_cell, + source_cell=source_cell or scope_key, merged_range=merged_range, start_row=start_row, end_row=end_row, @@ -387,7 +393,7 @@ def build_shougang_standard( def _push_annotation_sighting( - rows: list[tuple[int, str, str, str, str, str]], + rows: list[tuple[int, str, str, str, str, str, int, int]], row: int, entry_id: str, type_: str, diff --git a/src/agent/standards/sources.py b/src/agent/standards/sources.py index 716fee9..f85bd08 100644 --- a/src/agent/standards/sources.py +++ b/src/agent/standards/sources.py @@ -254,7 +254,8 @@ def read_guanji_catalog(path: str | Path) -> ReaderResult: leaf = clean(leaf_h.value) description = clean(leaf_h_def.value) content = clean(resolver.cell(row, SHOUGANG_COLS["CONTENT"]).value) - raw_level = str(resolver.cell(row, SHOUGANG_COLS["LEVEL"]).value).strip() + _level_cell = resolver.cell(row, SHOUGANG_COLS["LEVEL"]).value + raw_level = str(_level_cell).strip() if _level_cell is not None else "" resource = clean(resolver.cell(row, SHOUGANG_COLS["RESOURCE"]).value) l1di = resolver.cell(row, SHOUGANG_COLS["L1_DEF"]) l2di = resolver.cell(row, SHOUGANG_COLS["L2_DEF"]) diff --git a/tests/standards/test_build_finance.py b/tests/standards/test_build_finance.py index f69b44a..fd548b1 100644 --- a/tests/standards/test_build_finance.py +++ b/tests/standards/test_build_finance.py @@ -195,3 +195,28 @@ def test_finance_empty_department_opinion_produces_no_annotation(): entries = [_entry(level_1="客户", level_2="个人", level_3="个人身份鉴别信息", leaf="特有账户信息", row=55)] standard, _ = build_finance_standard(entries, source_file="f", source_sheet="Table 1") assert standard.scoped_annotations == () + + +def test_finance_identical_text_in_two_unmerged_cells_is_two_annotations(): + # regression: J5 and J10 both carry the same text but are NOT merged; they + # must stay two separate scoped annotations (each its own row), never one + # annotation spanning rows 5..10. + def sighting(row): + return _entry( + level_1="客户", level_2="个人", level_3="个人自然信息", + leaf="个人基本概况信息" if row == 5 else "个人财产信息", + raw_level="3", row=row, + remark="相同的备注文字", + provenance={"remark": {"source_cell": "J%d" % row, "merged_range": None, + "start_row": row, "end_row": row}}, + ) + + standard, _ = build_finance_standard( + [sighting(5), sighting(10)], source_file="f", source_sheet="Table 1" + ) + by_id = {a.annotation_id: a for a in standard.scoped_annotations} + assert set(by_id) == {"finance-remark-5-5", "finance-remark-10-10"} + assert by_id["finance-remark-5-5"].merged_range is None + assert by_id["finance-remark-5-5"].start_row == 5 + assert by_id["finance-remark-10-10"].start_row == 10 + assert len(standard.scoped_annotations) == 2