diff --git a/artifacts/generated/provenance/finance_standard_alignment.json b/artifacts/generated/provenance/finance_standard_alignment.json new file mode 100644 index 0000000..864ef80 --- /dev/null +++ b/artifacts/generated/provenance/finance_standard_alignment.json @@ -0,0 +1,474 @@ +{ + "mismatched_samples": [ + { + "category_id": "finance:业务.交易信息.交易基本信息", + "field": "AMONEY", + "name": "交易基本信息", + "sample_level": "L3", + "sample_path": [ + "业务", + "交易信息", + "交易通用信息", + "交易基本信息" + ], + "source_rows": [ + 133 + ], + "standard_entry_ids": [ + "finance:业务.交易信息.交易通用信息.交易基本信息" + ], + "standard_levels": [ + "L2" + ] + }, + { + "category_id": "finance:业务.账户信息.基本信息", + "field": "HXTRADENO", + "name": "基本信息", + "sample_level": "L3", + "sample_path": [ + "业务", + "账户信息", + "基本信息" + ], + "source_rows": [ + 51 + ], + "standard_entry_ids": [ + "finance:业务.账户信息..基本信息" + ], + "standard_levels": [ + "L2" + ] + } + ], + "resolved_match_rate": 0.9962, + "sample_counts": { + "matched": 529, + "mismatched": 2, + "resolved": 531, + "standard_level_unavailable": 0, + "standard_missing": 0, + "total": 568, + "unresolved": 37 + }, + "sample_missing_standard_categories": [ + "finance:业务.交易信息.保险收费信息", + "finance:业务.交易信息.保险金信托给付信息", + "finance:业务.合约协议.交易类中间业务信息", + "finance:业务.合约协议.代理类中间业务信息", + "finance:业务.合约协议.保单基本信息", + "finance:业务.合约协议.保单责任信息", + "finance:业务.合约协议.信托产品信息", + "finance:业务.合约协议.信托募集信息", + "finance:业务.合约协议.信托运用重要信息", + "finance:业务.合约协议.债权转让合同信息", + "finance:业务.合约协议.催收信息", + "finance:业务.合约协议.其他类中间业务信息", + "finance:业务.合约协议.咨询顾问类中间业务信息", + "finance:业务.合约协议.固有业务信息", + "finance:业务.合约协议.垫款信息", + "finance:业务.合约协议.展期信息", + "finance:业务.合约协议.托管业务信息", + "finance:业务.合约协议.投资理财信息", + "finance:业务.合约协议.投资银行业务信息", + "finance:业务.合约协议.担保信息", + "finance:业务.合约协议.担保承诺类中间业务信息", + "finance:业务.合约协议.授信信息", + "finance:业务.合约协议.放还款信息", + "finance:业务.合约协议.核保信息", + "finance:业务.合约协议.签约信息", + "finance:业务.合约协议.自营资金投资信息", + "finance:业务.合约协议.贷款业务重要信息", + "finance:业务.合约协议.资产信息", + "finance:业务.合约协议.赔付结果信息", + "finance:业务.合约协议.违约信息", + "finance:业务.合约协议.逾期信息", + "finance:业务.合约协议.销售渠道费用信息", + "finance:业务.账户信息.介质信息", + "finance:业务.账户信息.冻结信息", + "finance:业务.账户信息.特有账户信息", + "finance:业务.金融监管和服务.LEI数据发布信息", + "finance:业务.金融监管和服务.LEI注册信息", + "finance:业务.金融监管和服务.个人征信机构管理信息", + "finance:业务.金融监管和服务.事件公告信息", + "finance:业务.金融监管和服务.事项审批信息", + "finance:业务.金融监管和服务.事项申请信息", + "finance:业务.金融监管和服务.企业征信机构管理信息", + "finance:业务.金融监管和服务.信息沟通信息", + "finance:业务.金融监管和服务.信用体系建设管理信息", + "finance:业务.金融监管和服务.信用评级信息", + "finance:业务.金融监管和服务.分类考核评级信息", + "finance:业务.金融监管和服务.咨询投诉管理信息", + "finance:业务.金融监管和服务.国际合作信息", + "finance:业务.金融监管和服务.失效居民身份证核查信息", + "finance:业务.金融监管和服务.宣传培训信息", + "finance:业务.金融监管和服务.居民身份证核查信息", + "finance:业务.金融监管和服务.异议反馈信息", + "finance:业务.金融监管和服务.征信维权信息", + "finance:业务.金融监管和服务.成果评审信息", + "finance:业务.金融监管和服务.房地产市场分析信息", + "finance:业务.金融监管和服务.房地产数据处理信息", + "finance:业务.金融监管和服务.房地产调查信息", + "finance:业务.金融监管和服务.报表处理信息", + "finance:业务.金融监管和服务.指标分析信息", + "finance:业务.金融监管和服务.指标管理信息", + "finance:业务.金融监管和服务.机构业务申请基本信息", + "finance:业务.金融监管和服务.机构业务许可证书管理信息", + "finance:业务.金融监管和服务.机构年检信息", + "finance:业务.金融监管和服务.案例管理信息", + "finance:业务.金融监管和服务.汇率信息", + "finance:业务.金融监管和服务.消费者教育信息", + "finance:业务.金融监管和服务.热点汇总信息", + "finance:业务.金融监管和服务.电子公告管理信息", + "finance:业务.金融监管和服务.监督检查信息", + "finance:业务.金融监管和服务.舆情采集信息", + "finance:业务.金融监管和服务.行政监管信息", + "finance:业务.金融监管和服务.调查协查信息", + "finance:业务.金融监管和服务.金融信用数据库管理信息", + "finance:业务.金融监管和服务.问卷调查信息", + "finance:业务.金融监管和服务.非居民身份信息核查信息", + "finance:客户.个人.个人信贷信息", + "finance:客户.个人.个人健康生理信息", + "finance:客户.个人.个人党政信息", + "finance:客户.个人.个人司法信息", + "finance:客户.个人.个人地理位置信息", + "finance:客户.个人.个人就学信息", + "finance:客户.个人.个人职业信息", + "finance:客户.个人.个人财产信息", + "finance:客户.个人.个人资质证书信息", + "finance:客户.个人.个人间关系信息", + "finance:客户.个人.交易类标签信息", + "finance:客户.个人.价值标签信息", + "finance:客户.个人.公私间关系信息", + "finance:客户.个人.关系标签信息", + "finance:客户.个人.基础标签信息", + "finance:客户.个人.弱隐私生物特征信息", + "finance:客户.个人.强隐私生物特征信息", + "finance:客户.个人.签约标签信息", + "finance:客户.个人.营销服务标签信息", + "finance:客户.个人.行为信息", + "finance:客户.个人.行为标签信息", + "finance:客户.个人.风险标签信息", + "finance:客户.单位.交易类标签信息", + "finance:客户.单位.价值标签信息", + "finance:客户.单位.企业信贷信息", + "finance:客户.单位.企业司法信息", + "finance:客户.单位.企业工商信息", + "finance:客户.单位.企业税务信息", + "finance:客户.单位.传统鉴别信息", + "finance:客户.单位.公私间关系信息", + "finance:客户.单位.单位联系信息", + "finance:客户.单位.单位财务信息", + "finance:客户.单位.单位间关系信息", + "finance:客户.单位.基础标签信息", + "finance:客户.单位.签约标签信息", + "finance:客户.单位.股东信息", + "finance:客户.单位.股东重要信息", + "finance:客户.单位.营销标签信息", + "finance:客户.单位.行为信息", + "finance:客户.单位.行为标签信息", + "finance:客户.单位.风险标签信息", + "finance:监管.数据报送.监管指标上报信息", + "finance:监管.数据报送.监管明细数据上报信息", + "finance:监管.数据报送.金融统计信息", + "finance:监管.数据收取.审计信息", + "finance:监管.数据收取.统计分析信息", + "finance:监管.数据收取.评价、处罚与违规信息", + "finance:监管.数据收取.预警信息", + "finance:经营管理.技术管理.信息资产管理信息", + "finance:经营管理.技术管理.办公软件资源", + "finance:经营管理.技术管理.安全管理信息", + "finance:经营管理.技术管理.开发信息", + "finance:经营管理.技术管理.测试信息", + "finance:经营管理.技术管理.系统运维信息", + "finance:经营管理.技术管理.规划信息", + "finance:经营管理.技术管理.质量管理信息", + "finance:经营管理.综合管理.一般员工信息(公开)", + "finance:经营管理.综合管理.业务发展规划信息", + "finance:经营管理.综合管理.业绩信息", + "finance:经营管理.综合管理.人力需求规划信息", + "finance:经营管理.综合管理.人员招聘报名信息", + "finance:经营管理.综合管理.人员招聘考试信息", + "finance:经营管理.综合管理.企事业财务管理信息", + "finance:经营管理.综合管理.信息披露信息", + "finance:经营管理.综合管理.党务纪检信息", + "finance:经营管理.综合管理.公文信息", + "finance:经营管理.综合管理.内部审计信息", + "finance:经营管理.综合管理.内部资金往来信息", + "finance:经营管理.综合管理.分类信息", + "finance:经营管理.综合管理.利率管理信息", + "finance:经营管理.综合管理.合规信息", + "finance:经营管理.综合管理.员工信息(非公开)", + "finance:经营管理.综合管理.品牌战略规划信息", + "finance:经营管理.综合管理.国库管理信息", + "finance:经营管理.综合管理.培训与资质信息", + "finance:经营管理.综合管理.基建财务管理信息", + "finance:经营管理.综合管理.基本信息(公开)", + "finance:经营管理.综合管理.基本信息(非公开)", + "finance:经营管理.综合管理.层级信息", + "finance:经营管理.综合管理.岗位角色信息", + "finance:经营管理.综合管理.工会信息", + "finance:经营管理.综合管理.市场营销规划信息", + "finance:经营管理.综合管理.技能信息", + "finance:经营管理.综合管理.支付清算汇总信息", + "finance:经营管理.综合管理.支付清算鉴别信息", + "finance:经营管理.综合管理.档案管理信息", + "finance:经营管理.综合管理.法务信息", + "finance:经营管理.综合管理.税务信息", + "finance:经营管理.综合管理.章程制度信息", + "finance:经营管理.综合管理.管理会计信息", + "finance:经营管理.综合管理.薪资信息", + "finance:经营管理.综合管理.证件信息", + "finance:经营管理.综合管理.财务会计信息", + "finance:经营管理.综合管理.财务支出信息", + "finance:经营管理.综合管理.资产负债管理信息", + "finance:经营管理.综合管理.资金渠道流通汇总信息", + "finance:经营管理.综合管理.资金规划信息", + "finance:经营管理.综合管理.邮件信息", + "finance:经营管理.营销服务.产品重要信息", + "finance:经营管理.营销服务.分类信息", + "finance:经营管理.营销服务.基本信息", + "finance:经营管理.营销服务.市场营销信息(公开)", + "finance:经营管理.营销服务.市场营销信息(非公开)", + "finance:经营管理.营销服务.新产品(项目)研发信息", + "finance:经营管理.营销服务.服务管理信息", + "finance:经营管理.营销服务.渠道管理信息", + "finance:经营管理.营销服务.特征信息", + "finance:经营管理.营销服务.第三方代理渠道信息(公开)", + "finance:经营管理.营销服务.管理信息", + "finance:经营管理.营销服务.线上自有渠道信息(公开)", + "finance:经营管理.营销服务.线下自有渠道信息(公开)", + "finance:经营管理.营销服务.营销管理信息", + "finance:经营管理.运营管理.单证入库信息", + "finance:经营管理.运营管理.单证发放信息", + "finance:经营管理.运营管理.单证日常管理信息", + "finance:经营管理.运营管理.单证核销信息", + "finance:经营管理.运营管理.单证设计信息", + "finance:经营管理.运营管理.参数/指标运维信息", + "finance:经营管理.运营管理.合作内容信息", + "finance:经营管理.运营管理.合作单位基本信息", + "finance:经营管理.运营管理.合作单位联系人信息", + "finance:经营管理.运营管理.客户及监管相关音影像信息", + "finance:经营管理.运营管理.技术安防信息", + "finance:经营管理.运营管理.日常管理相关音影像信息", + "finance:经营管理.运营管理.柜面服务信息", + "finance:经营管理.运营管理.物理安防信息", + "finance:经营管理.运营管理.电话服务信息", + "finance:经营管理.运营管理.网络服务信息", + "finance:经营管理.风险管理信息.风险偏好不定量指标", + "finance:经营管理.风险管理信息.风险偏好偏离信息", + "finance:经营管理.风险管理信息.风险偏好定最指标", + "finance:经营管理.风险管理信息.风险抵御水平信息", + "finance:经营管理.风险管理信息.风险控制信息", + "finance:经营管理.风险管理信息.风险监测信息", + "finance:经营管理.风险管理信息.风险缓释信息", + "finance:经营管理.风险管理信息.风险计量信息", + "finance:经营管理.风险管理信息.风险识别信息", + "finance:经营管理.风险管理信息.黑名单信息" + ], + "standard": "finance", + "standard_entries": 237, + "standard_level_unavailable_samples": [], + "standard_missing_samples": [], + "training_categories": 233, + "training_categories_observed": 20, + "training_categories_unobserved": 213, + "unresolved_by_status": { + "missing_leaf": 34, + "path_mismatch": 3 + }, + "unresolved_evidence": [ + { + "candidate_standard_categories": [], + "leaf_name": "交易清金额信息", + "status": "missing_leaf" + }, + { + "candidate_standard_categories": [], + "leaf_name": "单位基本信息", + "status": "missing_leaf" + }, + { + "candidate_standard_categories": [], + "leaf_name": "单位基本信息", + "status": "missing_leaf" + }, + { + "candidate_standard_categories": [], + "leaf_name": "单位基本信息", + "status": "missing_leaf" + }, + { + "candidate_standard_categories": [], + "leaf_name": "单位基本情况", + "status": "missing_leaf" + }, + { + "candidate_standard_categories": [], + "leaf_name": "单位基本情况", + "status": "missing_leaf" + }, + { + "candidate_standard_categories": [], + "leaf_name": "单位基本情况", + "status": "missing_leaf" + }, + { + "candidate_standard_categories": [], + "leaf_name": "单位基本情况", + "status": "missing_leaf" + }, + { + "candidate_standard_categories": [], + "leaf_name": "单位基本情况", + "status": "missing_leaf" + }, + { + "candidate_standard_categories": [], + "leaf_name": "单位基本情况", + "status": "missing_leaf" + }, + { + "candidate_standard_categories": [], + "leaf_name": "单位基本情况", + "status": "missing_leaf" + }, + { + "candidate_standard_categories": [], + "leaf_name": "单位基本情况", + "status": "missing_leaf" + }, + { + "candidate_standard_categories": [], + "leaf_name": "单位基本情况", + "status": "missing_leaf" + }, + { + "candidate_standard_categories": [], + "leaf_name": "单位基本情况", + "status": "missing_leaf" + }, + { + "candidate_standard_categories": [], + "leaf_name": "单位基本情况", + "status": "missing_leaf" + }, + { + "candidate_standard_categories": [], + "leaf_name": "单位基本情况", + "status": "missing_leaf" + }, + { + "candidate_standard_categories": [], + "leaf_name": "单位基本情况", + "status": "missing_leaf" + }, + { + "candidate_standard_categories": [], + "leaf_name": "单位基本情况", + "status": "missing_leaf" + }, + { + "candidate_standard_categories": [], + "leaf_name": "单位基本情况", + "status": "missing_leaf" + }, + { + "candidate_standard_categories": [], + "leaf_name": "单位基本情况", + "status": "missing_leaf" + }, + { + "candidate_standard_categories": [], + "leaf_name": "单位基本情况", + "status": "missing_leaf" + }, + { + "candidate_standard_categories": [], + "leaf_name": "单位基本情况", + "status": "missing_leaf" + }, + { + "candidate_standard_categories": [], + "leaf_name": "单位基本情况", + "status": "missing_leaf" + }, + { + "candidate_standard_categories": [], + "leaf_name": "单位联系人信息", + "status": "missing_leaf" + }, + { + "candidate_standard_categories": [], + "leaf_name": "单位联系人信息", + "status": "missing_leaf" + }, + { + "candidate_standard_categories": [], + "leaf_name": "单位联系人信息", + "status": "missing_leaf" + }, + { + "candidate_standard_categories": [], + "leaf_name": "单位联系人信息", + "status": "missing_leaf" + }, + { + "candidate_standard_categories": [], + "leaf_name": "单位联系人信息", + "status": "missing_leaf" + }, + { + "candidate_standard_categories": [], + "leaf_name": "单位联系人信息", + "status": "missing_leaf" + }, + { + "candidate_standard_categories": [], + "leaf_name": "单位联系人信息", + "status": "missing_leaf" + }, + { + "candidate_standard_categories": [], + "leaf_name": "单位联系人信息", + "status": "missing_leaf" + }, + { + "candidate_standard_categories": [], + "leaf_name": "单位联系人信息", + "status": "missing_leaf" + }, + { + "candidate_standard_categories": [], + "leaf_name": "基本信息(公开", + "status": "missing_leaf" + }, + { + "candidate_standard_categories": [], + "leaf_name": "基本信息(公开", + "status": "missing_leaf" + }, + { + "candidate_standard_categories": [ + "finance:业务.交易信息.交易清结算信息" + ], + "leaf_name": "交易清结算信息", + "status": "path_mismatch" + }, + { + "candidate_standard_categories": [ + "finance:经营管理.运营管理.网络服务信息" + ], + "leaf_name": "网络服务信息", + "status": "path_mismatch" + }, + { + "candidate_standard_categories": [ + "finance:经营管理.运营管理.网络服务信息" + ], + "leaf_name": "网络服务信息", + "status": "path_mismatch" + } + ] +} diff --git a/artifacts/generated/provenance/infra_standard_alignment.json b/artifacts/generated/provenance/infra_standard_alignment.json new file mode 100644 index 0000000..03b2695 --- /dev/null +++ b/artifacts/generated/provenance/infra_standard_alignment.json @@ -0,0 +1,254 @@ +{ + "mismatched_samples": [], + "resolved_match_rate": 1.0, + "sample_counts": { + "matched": 64, + "mismatched": 0, + "resolved": 64, + "standard_level_unavailable": 0, + "standard_missing": 0, + "total": 64, + "unresolved": 0 + }, + "sample_missing_standard_categories": [ + "A1-1-1", + "A1-1-2", + "A1-1-4", + "B1-2", + "B1-3-1", + "B1-3-2", + "B1-3-3", + "B1-4-1", + "B1-4-2", + "B1-4-3", + "B1-4-4", + "B1-5", + "B2-1-1", + "B2-1-2", + "B2-2-1", + "B2-2-2", + "B2-2-3", + "B2-2-4", + "B2-2-5", + "B2-2-6", + "B3-1-1", + "B3-1-2", + "B3-1-3", + "B3-1-4", + "B3-1-5", + "B3-2-1", + "B3-2-2", + "B3-3", + "B3-4", + "B3-5", + "B3-6", + "B3-6-1", + "B3-6-2", + "B3-7-1", + "B3-7-2", + "B4-1-1", + "B4-1-2", + "B4-1-3", + "B4-1-4", + "B4-1-5", + "B4-1-6", + "B4-1-7", + "B4-2", + "B4-3-1", + "B5-1-1", + "B5-1-2", + "B5-1-3", + "B5-1-4", + "B5-1-5", + "B5-1-6", + "B5-1-7", + "B5-1-8", + "B5-1-9", + "B5-2-1", + "B5-2-10", + "B5-2-2", + "B5-2-3", + "B5-2-4", + "B5-2-5", + "B5-2-6", + "B5-2-7", + "B5-2-8", + "B5-2-9", + "B6-1-1", + "B6-1-10", + "B6-1-11", + "B6-1-12", + "B6-1-13", + "B6-1-2", + "B6-1-3", + "B6-1-4", + "B6-1-5", + "B6-1-6", + "B6-1-7", + "B6-1-8", + "B6-1-9", + "B6-2", + "C1-1-1", + "C1-1-2", + "C1-1-3", + "C1-2-1", + "C1-2-10", + "C1-2-2", + "C1-2-3", + "C1-2-4", + "C1-2-5", + "C1-2-6", + "C1-2-7", + "C1-2-8", + "C1-2-9", + "C1-3-1", + "C1-3-2", + "C1-3-3", + "C1-3-4", + "C1-3-5", + "C1-3-6", + "C1-4-1", + "C1-4-2", + "C1-4-3", + "C1-4-4", + "C1-4-5", + "C1-4-6", + "C1-5", + "C2-1-1", + "C2-1-10", + "C2-1-11", + "C2-1-2", + "C2-1-3", + "C2-1-4", + "C2-1-5", + "C2-1-6", + "C2-1-7", + "C2-1-8", + "C2-1-9", + "C2-2-1", + "C2-2-10", + "C2-2-11", + "C2-2-12", + "C2-2-13", + "C2-2-14", + "C2-2-15", + "C2-2-16", + "C2-2-17", + "C2-2-2", + "C2-2-3", + "C2-2-4", + "C2-2-5", + "C2-2-6", + "C2-2-7", + "C2-2-8", + "C2-2-9", + "C2-3-1", + "C2-3-2", + "C2-3-3", + "C2-3-4", + "C2-3-5", + "C2-3-6", + "C2-3-7", + "C3-1-1", + "C3-1-2", + "C3-2-1", + "C3-2-2", + "C3-2-3", + "C3-2-4", + "C3-2-6", + "C3-2-7", + "C3-2-8", + "C3-3-1", + "C3-3-2", + "C3-3-3", + "C3-4-1", + "C3-4-2", + "C3-4-3", + "C3-4-4", + "C4-1", + "C5-1-1", + "C5-1-2", + "C5-1-3", + "C5-2-1", + "C5-2-2", + "C5-2-3", + "C5-2-4", + "C5-2-5", + "C6-1-1", + "C6-1-2", + "C6-1-3", + "C6-1-4", + "C6-2-1", + "C6-2-2", + "C6-2-3", + "C6-2-4", + "C6-2-5", + "C6-2-6", + "C6-2-7", + "C6-3-1", + "C7-1-1", + "C7-1-2", + "C7-1-3", + "C7-1-4", + "C7-2-1", + "C7-2-2", + "C7-2-3", + "C7-3-1", + "C7-3-2", + "C7-3-3", + "C7-3-4", + "C7-3-5", + "C7-4-1", + "C7-4-2", + "C7-4-3", + "C7-5-1", + "C7-5-2", + "C8-1-1", + "C8-1-2", + "C8-1-3", + "C8-1-4", + "C8-1-5", + "C8-1-6", + "C8-1-7", + "C8-2-1", + "C8-2-2", + "C8-3-1", + "C8-3-2", + "C8-3-3", + "C8-4-1", + "C8-5-1", + "C8-6-1", + "C8-6-2", + "C8-6-3", + "C8-6-4", + "C8-6-5", + "C8-7-1", + "C8-7-10", + "C8-7-2", + "C8-7-3", + "C8-7-4", + "C8-7-5", + "C8-7-6", + "C8-7-7", + "C8-7-8", + "C8-7-9", + "C8-8-1", + "C8-8-2", + "C8-8-3", + "C8-8-4", + "C8-8-5", + "C8-8-6", + "C8-8-7", + "C8-8-8", + "C8-9-1" + ], + "standard": "shougang", + "standard_entries": 234, + "standard_level_unavailable_samples": [], + "standard_missing_samples": [], + "training_categories": 234, + "training_categories_observed": 4, + "training_categories_unobserved": 230, + "unresolved_by_status": {}, + "unresolved_evidence": [] +} diff --git a/artifacts/generated/provenance/shougang_standard_alignment.json b/artifacts/generated/provenance/shougang_standard_alignment.json new file mode 100644 index 0000000..38473ca --- /dev/null +++ b/artifacts/generated/provenance/shougang_standard_alignment.json @@ -0,0 +1,5179 @@ +{ + "mismatched_samples": [], + "resolved_match_rate": 1.0, + "sample_counts": { + "matched": 18393, + "mismatched": 0, + "resolved": 18393, + "standard_level_unavailable": 0, + "standard_missing": 0, + "total": 19415, + "unresolved": 1022 + }, + "sample_missing_standard_categories": [ + "B1-2", + "B1-4-2", + "B1-4-3", + "B1-4-4", + "B1-5", + "B3-1-3", + "B3-1-4", + "B3-1-5", + "B3-3", + "B3-4", + "B3-5", + "B3-6", + "B3-7-2", + "B4-2", + "B5-1-5", + "B5-1-8", + "B5-1-9", + "B5-2-5", + "B5-2-9", + "B6-2", + "C1-2-10", + "C1-2-4", + "C1-2-9", + "C1-4-6", + "C1-5", + "C3-1-2", + "C3-2-6", + "C3-2-7", + "C3-4-2", + "C3-4-4", + "C4-1", + "C5-1-3", + "C5-2-4", + "C8-1-5", + "C8-7-1", + "C8-7-10", + "C8-7-2", + "C8-7-3", + "C8-7-4", + "C8-7-9", + "C8-8-7", + "C8-8-8" + ], + "standard": "shougang", + "standard_entries": 234, + "standard_level_unavailable_samples": [], + "standard_missing_samples": [], + "training_categories": 234, + "training_categories_observed": 192, + "training_categories_unobserved": 42, + "unresolved_by_status": { + "placeholder": 1022 + }, + "unresolved_evidence": [ + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + }, + { + "candidate_standard_categories": [], + "leaf_name": "——", + "status": "placeholder" + } + ] +} diff --git a/artifacts/generated/provenance/standard_build_summary.json b/artifacts/generated/provenance/standard_build_summary.json new file mode 100644 index 0000000..1181702 --- /dev/null +++ b/artifacts/generated/provenance/standard_build_summary.json @@ -0,0 +1,149 @@ +{ + "legacy_information_loss": { + "finance": { + "legacy_dict_entries": 237, + "legacy_dict_path_compression": { + "entries_at_legacy_depth": 6, + "entries_with_path_deeper_than_legacy_L1_L2_leaf": 231, + "note": "legacy financial_standards_dict stored L1-L2-leaf identity strings; the real standard has 三级子类 provenance nodes that the legacy digest dropped (canonical standard keeps every entry, standard_entry_id includes the real 三级)" + }, + "legacy_unparseable_level_values": [ + "3 4级", + "l级" + ], + "standard_vs_registry_ids": { + "missing_from_registry": 0, + "registry_ids": 233, + "standard_training_categories": 233 + } + }, + "shougang": { + "legacy_dict_losses": { + "catalog_entries": 234, + "legacy_dict_entries": 234, + "note": "guanji_dict kept only 'name(code)' + description: real hierarchy path and the 分级 column were dropped (registry path was [] and no class field existed)", + "with_grading_restored": 234, + "with_real_path_restored": 234 + }, + "standard_vs_registry_codes": { + "note": "B3-6 中厚板作业计划 exists in the raw catalog but was dropped by guanji_dict/registry", + "registry_codes": 233, + "registry_only": [], + "standard_codes": 234, + "standard_only": [ + "B3-6" + ] + } + } + }, + "pers_info": { + "alignment_headline": null, + "note": "no confirmed classification/grading standard; the 18-category registry remains dataset-derived and is NOT presented as a canonical standard; no standard_data_level is fabricated", + "standard_source": null, + "status": "missing_or_unknown" + }, + "phase": "phase1-canonical-standard", + "standards": { + "finance": { + "alignment_headline": { + "matched": 529, + "mismatched": 2, + "resolved": 531, + "resolved_match_rate": 0.9962, + "samples_total": 568, + "standard_missing": 0 + }, + "build": { + "dataset": "finance", + "entries_read": 237, + "id_strategy": "path", + "issues": [ + { + "detail": "row 150 category '市场营销信息(公开)': raw level 'l' kept as-is, standard_data_level=null (not guessed)", + "kind": "level_unparseable" + }, + { + "detail": "row 168 category '客户及监管相关音影像信息': raw level '3 4' kept as-is, standard_data_level=null (not guessed)", + "kind": "level_unparseable" + } + ], + "level_distribution": { + "''": 2, + "L1": 11, + "L2": 157, + "L3": 61, + "L4": 6 + }, + "source_file": "data/raw/金融行业数据安全分类分级标准指南.xlsx", + "source_sheet": "Table 1", + "standard_entries_out": 237, + "standard_name": "金融行业数据安全分类分级标准指南", + "training_categories": 233 + }, + "standard_entries": 237, + "standard_source": "finance", + "status": "built", + "training_categories": 233 + }, + "infra": { + "alignment_headline": { + "matched": 64, + "mismatched": 0, + "resolved": 64, + "resolved_match_rate": 1.0, + "samples_total": 64, + "standard_missing": 0 + }, + "build": { + "dataset": "shougang", + "entries_read": 234, + "id_strategy": "code", + "issues": [], + "level_distribution": { + "L1": 16, + "L2": 170, + "L3": 48 + }, + "source_file": "data/raw/关基-数据分类分级目录.xlsx", + "source_sheet": "数据分类分级", + "standard_entries_out": 234, + "standard_name": "首钢京唐数据分类分级目录(关基)", + "training_categories": 234 + }, + "standard_entries": 234, + "standard_source": "shougang", + "status": "built", + "training_categories": 234 + }, + "shougang": { + "alignment_headline": { + "matched": 18393, + "mismatched": 0, + "resolved": 18393, + "resolved_match_rate": 1.0, + "samples_total": 19415, + "standard_missing": 0 + }, + "build": { + "dataset": "shougang", + "entries_read": 234, + "id_strategy": "code", + "issues": [], + "level_distribution": { + "L1": 16, + "L2": 170, + "L3": 48 + }, + "source_file": "data/raw/关基-数据分类分级目录.xlsx", + "source_sheet": "数据分类分级", + "standard_entries_out": 234, + "standard_name": "首钢京唐数据分类分级目录(关基)", + "training_categories": 234 + }, + "standard_entries": 234, + "standard_source": "shougang", + "status": "built", + "training_categories": 234 + } + } +} diff --git a/docs/design/data_level_design.md b/docs/design/data_level_design.md new file mode 100644 index 0000000..371b062 --- /dev/null +++ b/docs/design/data_level_design.md @@ -0,0 +1,168 @@ +# data_level / classification / field_sensitive 角色区分(设计说明 · Phase 0) + +Status: 只读记录(2026-08-20,数据快照 v1 后新增 raw Excel 已核对)。 +原则:只记录仓库内可验证事实,不猜测等级语义;**不改任何代码 / 数据 / 现有训练行为**。 + +**Phase 0 冻结范围**:只冻结三个字段「是什么、来自哪里、当前代码是否训练」。**不决定** `data_level` 最终训不训练——那属于 Stage2 task contract 的决策,本文档保持开放。 + +## 1. 三个概念是什么(角色模型) + +### `classification`(`level_1..level_4`) +- **角色**:sample label(源端标注),并且是**当前**的 training target(经 canonical 解析为 `target.category_id`,Stage1/Stage2 唯一监督目标)。 +- **槽位口径(重要,避免概念混淆)**:`level_1..level_4` 是统一分类槽位。当前 processed 数据把**最终/最小粒度类别**放进 `level_4` 槽位,但不同源数据的**真实分类深度并不统一**: + - pers_info:源数据是单层分类(学籍管理信息…),canonical schema 把它放进 `level_4`;`level_1/2/3` 为空,**不是说源标准真有四级体系**。 + - finance:四层填充(`level_1..level_4`);shougang/infra:四层填充。 + - 因此**不能把 `level_4` 等价成"真实标准的第四层"**;建无损 standard(Phase 1)时按各源的真实深度建模。 + +### `data_level`(`L1..L4`) +- **角色**:**sample label——样本携带的分级标签**。是否同时可被确定为 standard knowledge,**分数据集**: + - finance / shougang / infra:有证据可追溯到分类分级标准/目录(finance↔《金融行业数据安全分类分级标准指南》"最低安全级别参考";shougang/infra↔首钢京唐数据分类分级目录"分级"列,192/192 零冲突)。 + - pers_info:**目前只有 sample label,尚无已确认的 standard knowledge 来源**(源 Excel 直接填 `L1/L2/L3`,仓库内无对应标准文档)。 + - 所以不做笼统定义"`data_level` = standard knowledge 的样本级表现"——pers_info 就是反例。 +- **训练目标状态**:当前实现未进入 training target(`src/agent`、`script/verl`、`script/canonical` 对 `data_level` 零引用,parquet 不导出);**目标设计倾向将 `data_level` 作为与 classification 并列的模型预测目标**,即模型需要根据字段语义、业务上下文和分级规则独立判断安全等级,而不是仅根据 category 查表返回标准等级。最终接口仍待 Stage2 task contract 冻结。 +- **训练目标总口径**: + ```text + 当前实现: + classification → training target + data_level → provenance only + 目标任务方向: + classification → 模型预测 + data_level → 模型独立预测 + standard_data_level → 分级参考 / 审计依据,不直接等同于字段最终等级 + + 最终输出格式、prompt 可见信息和 reward + → 待 Stage2 task contract 冻结 + ``` + +### `sample_data_level` 与 `standard_data_level`(两个变体,必须分开保存) +- **`sample_data_level`** = 原始数据中具体字段的实际分级标签(即 § 上文 `data_level` 的本义),作为训练/评测 gold 候选。 +- **`standard_data_level`** = 分类分级标准给 category 的参考/最低等级(如 finance 标准列原名就叫**"最低安全级别参考"**)。 +- 二者**必须分别保存,不应在 canonicalization 时相互覆盖**:finance 的"最低安全级别参考"本身就不适合直接建模成 `category → 唯一最终 data_level`。 +- 后续 Phase 1 的 canonical standard 建议明确字段名 `standard_data_level`(值如 `"L3"`),**不要**简单命名为 `data_level`,否则日后容易再次与 sample gold 混淆。 + +### `field_sensitive`(是/否) +- **角色**:**独立 field-level 源端标注**,不属于上述两者。 +- 仅存在于 `data/raw/关基设施数据分类分级-不包含训练-用于测试(1).xlsx`(infra 测试来源)的"字段是否敏感"列;mapping 未映射 → 未进入 processed/canonical/parquet。 +- 与 `data_level` **不同步**(64 行:`L3+是=32 / L3+否=14 / L2+是=14 / L2+否=4`),证据明确,故为独立概念。 + +### 三者关系(已验证统计事实) +- `data_level` 与分类深度无关(pers_info 单层分类却分布 L1–L3);在数据上 `data_level` 对叶子类别近似确定性映射(finance 18/20、infra 4/4、pers_info 17/18、shougang 192/192 类别为单一级别,全库仅 4 条跨级样本)。 + +## 2. 三个字段在管线各层的位置 + +| 层 | classification | data_level | field_sensitive | +| --- | --- | --- | --- | +| raw(data/raw/*.xlsx) | 源列:一级子类/一级分类/分类 等 | 源列:数据级别(finance) / 分级(infra,shougang,pers_info),原始值 `1..4`(pers_info 为 `L1..L3`) | 仅 `关基设施…用于测试(1).xlsx` 有"字段是否敏感"列 | +| preprocessing(`script/preprocessing/processor.py`) | `normalize_label` 归一化(剥 `(A1-1-3)` 代码后缀) | `normalize_level`:`1/2/3/4/LEVEL1..4 → L1..L4`,其余直接 raise(仅重命名+别名归一,无转换规则) | **未映射 → 丢弃** | +| canonical(`data/canonical//all.json`) | 原样保留(provenance);另解析出 `target.category_id` 为唯一训练身份 | 原样保留(provenance;已验证 4 数据集 100% 保真) | 不存在 | +| SFT / RL parquet(`data/sft|rl/…`) | 标签 = `target.category_id`;messages 只含 prompt-visible 元数据(默认 field_name/field_description/field_type) | **不导出**(SFT row 与 RL five-field 均无 data_level) | 不存在 | + +关键代码事实: +- `src/agent/task/contracts.py::SampleTarget` 仅含 `leaf_level/leaf_name/category_id/category_path`,不含 data_level。 +- `src/agent/training/common.py::canonical_target`:训练标签唯一来自 `target.category_id`;`classification.level_1..4` 明确 "provenance only, never fallback"。 +- 全仓 grep:`data_level` 在 `src/agent`、`script/verl`、`script/canonical` 中 0 命中;仅 预处理(processor/split/rft_export)、mapping、`script/analysis`(只读统计)触及。 + +## 3. 各数据集数据来源与已知事实 + +### finance(信托核心系统) +- 原始文件:`data/raw/部分金融数据.xlsx`(568 行 ↔ processed 568;按 (table,field) 566/566 data_level 一致);标准文档 `data/raw/金融行业数据安全分类分级标准指南.xlsx`。 +- 分类体系:标准指南 → `financial_standards_dict.json`(233 条)→ finance corpus 233 / registry(path ID)。 +- data_level 来源:`部分金融数据.xlsx`"数据级别"列(原始 1/2/3/4);与标准指南"最低安全级别参考"(1级~4级)**一致**:529/568 精确对齐,37 条可经同域相邻类对齐(如 单位联系人信息 L2 ↔ 单位联系信息 2级),仅 2 条离群(AMONEY、HXTRADENO)。 +- data_level 分布:`L1:25 / L2:460 / L3:76 / L4:7`;L4 全部为"个人身份鉴别信息/传统鉴别信息"。 +- field_sensitive:无。等级语义(1~4 级代表什么):**仓库内无定义文本 → 不猜测**。 + +### shougang(首钢京唐集团) +- 原始文件:`data/raw/关基-数据分类分级目录.xlsx`("首钢京唐数据分类分级目录",234 叶,含**"分级"列** per-leaf `1/2/3`,分布 1:16/2:170/3:48);测试子集导出 `data/raw/关基设施…用于测试(1).xlsx`。 +- 分类体系:目录 → `guanji_dict.json`(234 条,**只保留类别、丢掉了"分级"列**)→ shougang corpus 233 / registry(code ID `A1-1-1`)。 +- data_level 来源:目录"分级"列;shougang 已观测的 192 个 leaf category 中,sample `data_level` 与目录对应 category 的"分级"**192/192 一致、零冲突**(32 个目录类别在数据中未出现)。因此该目录是目前可确认的 `standard_data_level` 来源——但 sample 与 standard **概念上不预设必然相等**(见 §1 两个变体)。 +- data_level 分布:`L1:899 / L2:14,188 / L3:4,328`;占位符 `——` 不可训练(1,022 条)。 +- field_sensitive:主训练数据无;仅 64 行的测试导出文件带该列。等级语义:**无定义**(目录只有数字)。 + +### infra(钢铁基建,= shougang 测试子集) +- 原始文件:`data/raw/关基设施…用于测试(1).xlsx`(64 行);逐条 (db,table,field) 与 infra processed **64/64 一致**(含 data_level);⊂ shougang。data_level 的 `standard_data_level` 来源同样是该目录(经 shougang 复用),不另设独立映射。 +- 分类/registry:同 shougang(registry_source=shougang)。 +- data_level 分布:`L2:18 / L3:46`(原"分级"值 2/3)。 +- field_sensitive:**有**("字段是否敏感",是=46 / 否=18,与分级不同步,见 §1);预处理丢弃。 +- 等级语义:无定义。 + +### pers_info(高校个人信息) +- 原始文件:`data/raw/带分级分类的个人基础信息样本190条.xlsx`(189 行 → 176,去重后;176/176 data_level 一致)。 +- 分类体系:单层 `level_4` 18 类(放入 `level_4` 槽位,见 §1);corpus 来自数据集自身 universe(`build_report.source = null`)。 +- data_level 来源:该 Excel "分级"列直接 `L1/L2/L3`;**仓库无对应标准文档 → 仅有 sample label,无确认的 standard knowledge 来源**。 +- data_level 分布:`L1:31 / L2:98 / L3:47`;1 类跨级(基本信息年级信息和班级信息 L1×1/L2×8)。 +- field_sensitive:无。等级语义:仓库内 `education_dict.json` 为"公开/内部/重要/敏感"四档,与 L1~L3 无法对应且有反例(考核信息=内部数据却标 L3 等)→ 不猜测。 + +## 4. Phase 0 冻结的核心模型 + +```text +classification data_level + sample label sample label + ↓ ↓ +canonical category_id normalized data_level + │ │ + └──────────┬─────────────────┘ + ↓ + proposed Stage2 joint target + category + data_level + (Stage2 contract 待实现 / future target) + + + 原始分类分级标准 + │ + ↓ + canonical standard(Phase 1) + category_id / description / path + standard_data_level + grading_rules(若能获得) + │ + ↓ + 为模型提供分级规则/参考 + + 与 sample data_level 审计 + + +field_sensitive = 独立 field-level annotation,不属于上述两者 +``` + +> **最关键的独立分级设计原则**:`standard_data_level` 是 category 的标准参考等级,**不预设其必然等于每个具体字段的最终 `sample_data_level`**;若目标是独立分级,Stage2 **不应直接把候选类别对应的 gold `standard_data_level` 暴露给模型**,否则 data_level 任务会退化为 category→level 查表。 + +- **Phase 0 冻结**:`data_level` 是什么(sample label 的分级标签)、来自哪里(finance/shougang/infra 可追溯到标准/目录;pers_info 尚无标准来源)、当前代码未训练它(provenance only)。 +- **Phase 0 不决定**:`data_level` 最后训不训练;不把"当前未训练"写成"设计上不训练"。 +- **设计方向(非当前实现,待 Phase 1 起逐步验证)**:从原始分类分级标准构建 canonical standard(含 `standard_data_level`),再与样本 `data_level` 做一致性审计——因此建 standard 时按各源真实分类深度建模,勿把 `level_4` 槽位当作真实标准第四层。 + +## 5. 目标任务方向与前置缺口 + +**目标任务方向:优先采用任务形态 B —— 模型独立判断字段安全等级。** + +```text +Stage1: +field metadata + 全量 leaf categories +→ Top-5 categories + +Stage2: +field metadata ++ Top-5 category descriptions/examples ++ 领域信息 ++ 分级规则/标准说明 +→ category + data_level +``` + +其中: +- `category` 与 `data_level` **都是模型预测目标**。 +- `standard_data_level` 作为标准参考和审计字段保存。 +- **默认不直接把每个候选的 `standard_data_level` 暴露给 Stage2**,否则 data_level 任务会退化为 category→level 查表。 + +**Baseline A(降级为基线):lookup-assisted grading** +向模型提供候选 category 的 `standard_data_level`,用于评估"分类 + 标准读取"任务,并作为 independent grading(Target B)的对照基线。 + +**Target B:independent grading** +不直接给候选 gold level,模型依据字段语义、业务上下文和分级规则自行判级。 + +研究问题由此清晰为: +- **A:会不会选对标准项?** +- **B:会不会真正做分级判断?** + +**其他未决项**: +- holdout 缺口:L4 仅 finance 7 条且全在 train;L1 在多数 val/test 缺失 → 分级泛化评估需处理(补充/重抽样/保持冻结)。 +- `field_sensitive` 是否建设为字段级敏感标签,及其与 data_level 的关系定义。 +- 跨数据集 L1~L4 **不推断同义**:finance / pers_info / shougang 无证据表明相同 `L3` 用同一套业务定义,仅 infra=shougang 同源明确。 +- **研究风险(需专门实验验证)**:per-category `data_level` 近确定性映射(跨级样本全库仅 4 条)——若训练集里几乎每个 category 永远只有一个 level,即使目标是独立分级,模型仍可能学成**隐式 `category → level` 查表**而非真正分级规则。后续必须专门设计实验检查:构造同 category 多 level / 跨级样本;并至少跑三组对照——**A lookup-assisted grading vs B independent grading vs B−(对 grading rules 做消融)**——用于区分模型学到的是查表还是真正分级规则。 diff --git a/docs/design/phase1_canonical_standard.md b/docs/design/phase1_canonical_standard.md new file mode 100644 index 0000000..d4a756e --- /dev/null +++ b/docs/design/phase1_canonical_standard.md @@ -0,0 +1,160 @@ +# Phase 1:无损 canonical standard 层(设计说明 + 迁移 note) + +Status: 2026-08-20。只读建立数据事实层,**不改任何现有训练行为**;不推断 +L1–L4 语义;不跨数据集假设同义;不修改样本、prompt、parser、reward、SFT/RL +parquet;`standard_data_level` 仅作标准参考,绝不覆盖 processed/canonical 的 +sample `data_level`。 + +相关文件: +- Phase 0 语义:`docs/design/data_level_design.md` +- 标准构建实现:`src/agent/standards/`(contracts/sources/build/align) +- CLI:`script/standard/cli.py` → `python -m script.standard.cli` +- 产物:`data/standards/*.standard.json`、`artifacts/generated/provenance/` + +--- + +## 1. canonical standard schema + +每个 standard entry(`data/standards/.standard.json` 内,`entries[]`, +一条原始标准行 = 一个 entry,**事实层不做任何聚合**): + +```jsonc +{ + "standard_entry_id": "finance:业务.合约协议.贷款业务信息.基本信息 | A1-1-1", + "category_id": "finance:业务.合约协议.基本信息 | A1-1-1", + "name": "基本信息", + "path": ["业务", "合约协议", "贷款业务信息", "基本信息"], // 真实源层级深度,空层省略 + "description": "…", + "code": null | "A1-1-1", + "standard_data_level": "L1|L2|L3|L4|null", + "raw_level": "2 | l | 3 4", + "content": "…数据资源说明…", + "source": {"file": "…", "sheet": "…", "row": …}, + "raw_fields": { // 继承自合并组的事实,带 provenance + "level_2_definition": {"value": "…", "source_cell": "D93", "merged_range": "D93:D132", "start_row": 93, "end_row": 132, "inherited": true} + } +} +``` + +顶层:`dataset / id_strategy / standard_name / standard_source{file,sheet} / +fingerprint / entries[] / training_projection / scoped_annotations[]`。 + +- **`standard_entry_id`** = 原始标准中的**真实身份**(finance 含真实三级子类 + `finance:{L1}.{L2}.{L3}.{L4}`;shougang = guanji code)。唯一。 +- **`category_id`** = 当前训练/registry 兼容 alias(finance = L1-L2-leaf,与 + `DatasetConfig.identity_fields` 一致)。**不唯一**:多个标准 entry 可投影到同一 + 训练类别。 +- **`training_projection`** = 派生视图 `{category_id: [standard_entry_id…]}`, + 把 237 个 finance entry 投影到 233 个训练类别。 +- **`raw_fields`** = entry 从合并组**继承**的层级事实(finance 二级/三级定义; + shougang 一级/二级/三级定义 + 数据来源 resource),每项带 `source_cell` / + `merged_range` provenance——是网格事实,不是 leaf 私有标签。 +- **`scoped_annotations`** = 网格作用域注解(finance 备注 J、部门意见 K,当前 K 为空): + `{annotation_id, type, text, source_cell, merged_range, start_row, end_row, + applies_to_standard_entry_ids}`。合并格是一条源值作用于一个行范围,**绝不复制成 + 单个 leaf 的私有备注**。 +- `fingerprint` = entries(不含 source 行号)+ projection + annotations 的 + sha256;**与输入顺序无关**(按 entry id / annotation id 排序)。 + +### 1.1 列语义:hierarchy / leaf / scoped(merged-aware) + +| 源 | hierarchy-level(组级合并) | leaf-level(逐行) | scoped annotation | +| --- | --- | --- | --- | +| finance | B 一级子类 / C 二级子类 / D 二级定义 / E 三级子类 / F 三级定义 | G 四级子类 / H 内容 / I 安全级别 | J 备注(J55 单行 / J93:J132 40 行 / J168:J169 2 行)、K 部门意见(空) | +| shougang | B..G 一级分类+定义 / 二级分类+定义 / 三级分类+定义 | H 四级分类 / I 四级定义 / J 数据资源说明 / K 分级 / L 数据来源 | 无备注列 | + +- **merged 处理只作用于两个 standard source**(finance 指南、shougang 目录): + `MergedCellResolver` 对任意 cell 返回 value + anchor_cell + merged_range + + start/end row + inherited,reader 不再用手工 carry-forward。 +- **三个 sample source(部分金融数据 / 带分级分类… / 关基设施…用于测试)无业务级 + 合并**(`merged_ranges_total=0`;finance 样本仅标题 A1:A2 一条),样本层 zero 改动。 + +## 2. 各数据集 standard source 状态 + +| dataset | standard_source | 事实源文件 | 状态 | +| --- | --- | --- | --- | +| finance | `finance` | `data/raw/金融行业数据安全分类分级标准指南.xlsx`(sheet Table 1) | built(**237 entries / 233 training categories**) | +| shougang | `shougang` | `data/raw/关基-数据分类分级目录.xlsx`(sheet 数据分类分级) | built(234 entries) | +| infra | `shougang`(复用,不复制维护另一套) | — | 复用共享标准(64/64 对齐) | +| pers_info | `null`(missing / unknown) | 无已确认标准 | **不生成虚假 standard_data_level** | + +pers_info:仓库内无确认的分类分级标准;18 类 registry 维持当前 +dataset-derived 行为,但**不伪装成 canonical standard**——不生成 +`pers_info.standard.json`,不在 summary 中编造等级。 + +## 3. sample ↔ standard 对齐统计(严格 category alias 连接) + +| dataset | total | resolved | matched | mismatched | standard_missing | unresolved | resolved_match_rate | +| --- | --- | --- | --- | --- | --- | --- | --- | +| finance | 568 | 531 | 529 | 2 | 0 | 37 | 99.62% | +| shougang | 19,415 | 18,393 | 18,393 | 0 | 0 | 1,022 | 100% | +| infra | 64 | 64 | 64 | 0 | 0 | 0 | 100% | +| pers_info | — | — | — | — | — | — | 无标准 | + +- finance 未解析 37 条附 **`unresolved_evidence`**(evidence-only):每条含 + `{status, leaf_name, candidate_standard_categories[]}`(同名校对候选,不修复); + 其中 `missing_leaf 34 + path_mismatch 3`。 +- 对齐支持多 entry 类别:sample 等级命中该类别任一 entry 的等级即 matched。 +- 对齐只读:不修改 sample `data_level`,不自动修标签。 + +## 4. 发现的数据异常(只报告,不修复) + +1. **finance 原始标准 2 处不可解析等级**:row 150 `市场营销信息(公开)` 原始值 + `l`、row 168 `客户及监管相关音影像信息` 原始值 `3 4`。→ `standard_data_level=null` + + build issue,不做猜测。 +2. **shougang 目录中"三级即叶子"层级**:10 行 四列为 `——`、叶子码在 三级 + (合同归并 B1-2、合同跟踪 B1-5 等)。是真实类别,已按真实深度 3 层保存 + (不发明第 4 层)。 +3. **数据侧 `——` 与目录侧 `——` 语义不同**:样本 `level_4='——'` 是不可训练 + 占位;目录 `——` 是"该层无子结点"。分别处理。 +4. **legacy 信息损失**:finance dict 压平 4 层 path(232/233 类丢 三级 provenance + 层);shougang dict 丢全部 path+分级,并**漏掉 B3-6 中厚板作业计划**。 + canonical standard 均已恢复。 +5. **finance 5 条同训练类别的标准 entry**(业务/合约协议/基本信息 下的 合同通用/ + 贷款业务/中间业务/资金业务/其他支付业务)全部保留,等级一致(2),经 + `training_projection` 投影为 1 个训练类别——这是低损事实层与训练投影的边界, + 不是 bug。 + +## 5. 下一阶段(registry/corpus 接口变化) + +当前:`raw standard(Excel) → canonical standard(无损,本文档)→(下一步)LeafRegistry + Corpus`。 + +下一步需明确: +1. `src/agent/task/canonical_corpus.py` 改消费 `CanonicalStandard`:按 + `category_id`(training 投影)建 LeafRegistry,`path` 用标准真实深度; + 233 类保留、B3-6 是否收养(当前数据 0 样本,纯宇宙完整性)。 +2. `corpus_to_mapping` 是否附带 `standard_data_level`/`standard_entry_id` + 作为 Stage2 知识——属任务契约决策(Phase 0 §5 A/B),本阶段不预置。 +3. 现有 dict 产物降级为 legacy/derived 审计对照,不再作事实源。 + +## 6. 测试 + +``` +tests/standards/ + test_merged_resolver.py MergedCellResolver anchor/inherited/range(tmp workbook) + test_contracts.py round-trip(含 raw_fields / scoped_annotations)/ normalize / fingerprint + test_build_finance.py D/F 层级定义继承、J55/J93:J132/J168:J169 作用域、level 异常、确定性 + test_build_shougang.py C/E/G 定义 + resource 保留、三级叶、no_code、确定性 + test_align.py 对齐桶、多 entry 类别、不修改样本、unresolved evidence、路由 + test_checksum.py checksum manifest 校验(缺失/不符) + test_real_xlsx.py 真实 Excel + canonical 集成断言(raw 缺失时 skip) + —— 含:3 个 sample source 无业务级 merged cells;J55=1/J93:J132=40/J168:J169=2 +``` + +结果:`pytest tests/standards` → **56 passed**;全仓 **298 passed, 2 skipped** +(skip=本地无 verl,既有)。 + +产物可重生成:`python -m script.standard.cli`(拒绝无 `--overwrite` 覆盖; +先行全部构建/对齐、后写盘;重复构建字节级一致)。 + +### restore / 分发(Blocker-2 选型:B) + +- **事实源不进入 Git**:raw workbook 属于数据提供方(含首钢内部目录),gitignored。 +- **受控恢复流程**:从数据提供方/私有 artifact 取回两张表到 `data/raw/` 后,CLI + 先按 `script/standard/checksums.json`(已入库)校验 sha256,不符即拒绝构建 + (`--skip-checksum` 供离线调试显式绕过)。 +- **git 边界**:`data/standards/*.standard.json` 仍被 `/data/*` 排除(可再生层, + 与 processed/canonical 一致);入库的是 `src/agent/standards/`、`script/standard/` + (含 `checksums.json`)、`tests/standards/`、`docs/design/*.md` 与 + `artifacts/generated/provenance/`。fresh clone 后按 restore 流程即可重建完全一致的 + 事实层(fingerprint 可核对)。 diff --git a/script/standard/__init__.py b/script/standard/__init__.py new file mode 100644 index 0000000..e69de29 diff --git a/script/standard/checksums.json b/script/standard/checksums.json new file mode 100644 index 0000000..f9460dd --- /dev/null +++ b/script/standard/checksums.json @@ -0,0 +1,4 @@ +{ + "data/raw/金融行业数据安全分类分级标准指南.xlsx": "8af7e04bf6928ad1740e5370e43d61a3b26c0677ead2525cf4198e2c43d54219", + "data/raw/关基-数据分类分级目录.xlsx": "32e00ec50573b8de310313486e8acd4d09e7614d1a2195f01c55ec7604d9d078" +} diff --git a/script/standard/cli.py b/script/standard/cli.py new file mode 100644 index 0000000..61fd7f3 --- /dev/null +++ b/script/standard/cli.py @@ -0,0 +1,346 @@ +"""Phase 1 canonical standard CLI: build standards + alignment artifacts. + +Usage: + python -m script.standard.cli [--overwrite] [--skip-checksum] + +Reads the ORIGINAL standard workbooks (data/raw), verifies their sha256 +against the committed manifest (script/standard/checksums.json), and builds +the canonical standards + sample<->standard alignment: + + data/standards/finance.standard.json + data/standards/shougang.standard.json + artifacts/generated/provenance/finance_standard_alignment.json + artifacts/generated/provenance/shougang_standard_alignment.json + artifacts/generated/provenance/infra_standard_alignment.json + artifacts/generated/provenance/standard_build_summary.json + +Distribution (option B): raw workbooks are gitignored and must be restored +from the data provider first (see docs/design/phase1_canonical_standard.md +"restore" section). The CLI refuses to build on a missing file and on a +checksum mismatch (silently building from a wrong file would corrupt the +standard); --skip-checksum overrides the latter for offline tweaks. + +Fail-fast: every dataset is read + built + aligned before anything is +written; all outputs are written exactly once. Deterministic: JSON is written +with sort_keys and entries are pre-sorted by standard_entry_id; no timestamps +or machine-local paths. Raw workbooks are never training dependencies. +""" + +from __future__ import annotations + +import argparse +import hashlib +import json +from pathlib import Path +import sys + +PROJECT_ROOT = Path(__file__).resolve().parents[2] +sys.path.insert(0, str(PROJECT_ROOT / "src")) + +from agent.standards.align import align_dataset_to_standard, load_canonical_records +from agent.standards.build import ( + build_finance_standard, + build_shougang_standard, + resolve_standard_dataset, +) +from agent.standards.contracts import CanonicalStandard +from agent.standards.sources import ( + read_finance_standard_guide, + read_guanji_catalog, +) + +DEFAULT_RAW_DIR = PROJECT_ROOT / "data" / "raw" +DEFAULT_CANONICAL_DIR = PROJECT_ROOT / "data" / "canonical" +DEFAULT_STANDARD_DIR = PROJECT_ROOT / "data" / "standards" +DEFAULT_ARTIFACT_DIR = PROJECT_ROOT / "artifacts" / "generated" / "provenance" +CHECKSUM_MANIFEST = Path(__file__).with_name("checksums.json") + +FINANCE_XLSX = "金融行业数据安全分类分级标准指南.xlsx" +SHOUGANG_XLSX = "关基-数据分类分级目录.xlsx" + + +def _repo_relative(path: str | Path) -> str: + try: + return Path(path).resolve().relative_to(PROJECT_ROOT.resolve()).as_posix() + except ValueError: + return Path(path).as_posix() + + +def _write_json(payload, path: Path) -> None: + path.parent.mkdir(parents=True, exist_ok=True) + with path.open("w", encoding="utf-8", newline="\n") as handle: + json.dump(payload, handle, ensure_ascii=False, indent=2, sort_keys=True) + handle.write("\n") + + +def _sha256(path: Path) -> str: + digest = hashlib.sha256() + with path.open("rb") as handle: + for chunk in iter(lambda: handle.read(1 << 20), b""): + digest.update(chunk) + return digest.hexdigest() + + +def _verify_checksums( + paths: dict[str, Path], + manifest_path: Path, + *, + allow_skip: bool, +) -> None: + if not manifest_path.is_file(): + if not allow_skip: + raise FileNotFoundError( + f"checksum manifest not found: {manifest_path} " + "(pass --skip-checksum to proceed without verification)" + ) + return + with manifest_path.open(encoding="utf-8") as handle: + manifest = json.load(handle) + for rel, path in paths.items(): + expected_raw = manifest.get(rel) + if not expected_raw: + continue # file not covered by manifest + actual = _sha256(path) + if actual != expected_raw: + raise ValueError( + f"sha256 mismatch for {rel}: got {actual[:16]}… expected " + f"{expected_raw[:16]}… — refusing to build from a file that " + "differs from the recorded fact source " + "(pass --skip-checksum only if you know what you are doing)" + ) + + +def _registry_ids(dataset: str) -> set[str]: + from agent.task import LeafRegistry + + path = PROJECT_ROOT / "cfg" / "task" / "registry" / f"{dataset}.registry.json" + return set(LeafRegistry.from_path(path).ids) + + +def _legacy_information_loss( + finance_standard: CanonicalStandard, + shougang_standard: CanonicalStandard, +) -> dict[str, object]: + """Quantify what the legacy standards_map digests dropped vs the raw + standard canonical build (path depth, grading, missing codes).""" + + import json as _json + + finance_out: dict[str, object] = {} + finance_ids = {c.category_id for c in finance_standard.categories} + finance_reg_ids = _registry_ids("finance") + finance_out["standard_vs_registry_ids"] = { + "standard_training_categories": len(finance_ids), + "registry_ids": len(finance_reg_ids), + "missing_from_registry": len(finance_ids - finance_reg_ids), + } + legacy = _json.load( + ( + PROJECT_ROOT / "data" / "knowledge" / "standards_map" + / "financial_standards_dict.json" + ).open(encoding="utf-8") + ) + depth_lost = 0 + depth_kept = 0 + for entry in finance_standard.categories: + segments = 3 # legacy dict identity was L1-L2-leaf + if len(entry.path) > segments: + depth_lost += 1 + else: + depth_kept += 1 + finance_out["legacy_dict_path_compression"] = { + "entries_with_path_deeper_than_legacy_L1_L2_leaf": depth_lost, + "entries_at_legacy_depth": depth_kept, + "note": "legacy financial_standards_dict stored L1-L2-leaf identity " + "strings; the real standard has 三级子类 provenance nodes that the " + "legacy digest dropped (canonical standard keeps every entry, " + "standard_entry_id includes the real 三级)", + } + finance_out["legacy_dict_entries"] = len(legacy) + finance_out["legacy_unparseable_level_values"] = sorted( + str(v.get("class", "")) for v in legacy.values() if isinstance(v, dict) + and v.get("class") in ("l级", "3 4级") + ) + + shougang_out: dict[str, object] = {} + shougang_ids = {c.category_id for c in shougang_standard.categories} + shougang_reg_ids = _registry_ids("shougang") + shougang_out["standard_vs_registry_codes"] = { + "standard_codes": len(shougang_ids), + "registry_codes": len(shougang_reg_ids), + "standard_only": sorted(shougang_ids - shougang_reg_ids), + "registry_only": sorted(shougang_reg_ids - shougang_ids), + "note": "B3-6 中厚板作业计划 exists in the raw catalog but was dropped " + "by guanji_dict/registry", + } + legacy_g = _json.load( + ( + PROJECT_ROOT / "data" / "knowledge" / "standards_map" / "guanji_dict.json" + ).open(encoding="utf-8") + ) + codes_without_path = 0 + codes_with_level = 0 + for entry in shougang_standard.categories: + if entry.path: + codes_without_path += 1 + if entry.standard_data_level: + codes_with_level += 1 + shougang_out["legacy_dict_losses"] = { + "catalog_entries": len(shougang_standard.categories), + "legacy_dict_entries": len(legacy_g), + "with_real_path_restored": codes_without_path, + "with_grading_restored": codes_with_level, + "note": "guanji_dict kept only 'name(code)' + description: real " + "hierarchy path and the 分级 column were dropped (registry path was " + "[] and no class field existed)", + } + return {"finance": finance_out, "shougang": shougang_out} + + +def main(argv: list[str] | None = None) -> int: + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("--raw-dir", type=Path, default=DEFAULT_RAW_DIR) + parser.add_argument("--canonical-dir", type=Path, default=DEFAULT_CANONICAL_DIR) + parser.add_argument("--standard-dir", type=Path, default=DEFAULT_STANDARD_DIR) + parser.add_argument("--artifact-dir", type=Path, default=DEFAULT_ARTIFACT_DIR) + parser.add_argument("--overwrite", action="store_true") + parser.add_argument( + "--skip-checksum", + action="store_true", + help="build even when the checksum manifest is missing or a file " + "mismatches (use only for offline tweaks)", + ) + args = parser.parse_args(argv) + + raw_dir = Path(args.raw_dir) + finance_xlsx = raw_dir / FINANCE_XLSX + shougang_xlsx = raw_dir / SHOUGANG_XLSX + missing = [p for p in (finance_xlsx, shougang_xlsx) if not p.is_file()] + if missing: + raise FileNotFoundError( + "raw standard workbook(s) missing; restore them from the data " + "provider first (see docs/design/phase1_canonical_standard.md, " + "'restore' section). Missing: " + + ", ".join(str(p) for p in missing) + ) + _verify_checksums( + {FINANCE_XLSX: finance_xlsx, SHOUGANG_XLSX: shougang_xlsx}, + CHECKSUM_MANIFEST, + allow_skip=args.skip_checksum, + ) + + # 1. read + build + align EVERYTHING (pure computation, no writes) + finance_raw = read_finance_standard_guide(finance_xlsx) + shougang_raw = read_guanji_catalog(shougang_xlsx) + finance_standard, finance_report = build_finance_standard( + finance_raw.entries, + source_file=_repo_relative(finance_xlsx), + source_sheet="Table 1", + reader_issues=finance_raw.issues, + ) + shougang_standard, shougang_report = build_shougang_standard( + shougang_raw.entries, + source_file=_repo_relative(shougang_xlsx), + source_sheet="数据分类分级", + reader_issues=shougang_raw.issues, + ) + + canonical_records = { + dataset: load_canonical_records(args.canonical_dir / dataset / "all.json") + for dataset in ("finance", "shougang", "infra") + } + alignments = { + "finance": align_dataset_to_standard( + canonical_records["finance"], finance_standard + ), + "shougang": align_dataset_to_standard( + canonical_records["shougang"], shougang_standard + ), + "infra": align_dataset_to_standard( + canonical_records["infra"], shougang_standard + ), + } + loss = _legacy_information_loss(finance_standard, shougang_standard) + + # 2. refuse to overwrite without --overwrite + outputs = { + "standards/finance": args.standard_dir / "finance.standard.json", + "standards/shougang": args.standard_dir / "shougang.standard.json", + "provenance/finance_alignment": args.artifact_dir / "finance_standard_alignment.json", + "provenance/shougang_alignment": args.artifact_dir / "shougang_standard_alignment.json", + "provenance/infra_alignment": args.artifact_dir / "infra_standard_alignment.json", + "provenance/summary": args.artifact_dir / "standard_build_summary.json", + } + if not args.overwrite: + existing = [str(p) for p in outputs.values() if p.exists()] + if existing: + raise FileExistsError( + "refusing to overwrite existing canonical-standard artifacts: " + + ", ".join(existing) + + " (pass --overwrite to regenerate)" + ) + + # 3. build the summary first (fail-fast: nothing is written until every + # computed payload is ready) + standards = { + "finance": finance_standard, + "shougang": shougang_standard, + } + summary = { + "phase": "phase1-canonical-standard", + "standards": { + dataset: { + "standard_source": resolve_standard_dataset(dataset), + "status": "built" if resolve_standard_dataset(dataset) else "missing_or_unknown", + "standard_entries": ( + len(standards[resolve_standard_dataset(dataset)].entries) + if resolve_standard_dataset(dataset) in standards + else 0 + ), + "training_categories": ( + standards[resolve_standard_dataset(dataset)].trainable_category_count() + if resolve_standard_dataset(dataset) in standards + else 0 + ), + "build": ( + finance_report.to_mapping() + if dataset == "finance" + else shougang_report.to_mapping() + if dataset in ("shougang", "infra") + else None + ), + "alignment_headline": { + "samples_total": alignments[dataset]["sample_counts"]["total"], + "resolved": alignments[dataset]["sample_counts"]["resolved"], + "matched": alignments[dataset]["sample_counts"]["matched"], + "mismatched": alignments[dataset]["sample_counts"]["mismatched"], + "standard_missing": alignments[dataset]["sample_counts"]["standard_missing"], + "resolved_match_rate": alignments[dataset]["resolved_match_rate"], + }, + } + for dataset in ("finance", "shougang", "infra") + }, + "pers_info": { + "standard_source": None, + "status": "missing_or_unknown", + "note": "no confirmed classification/grading standard; the 18-category " + "registry remains dataset-derived and is NOT presented as a canonical " + "standard; no standard_data_level is fabricated", + "alignment_headline": None, + }, + "legacy_information_loss": loss, + } + # 4. write everything exactly once + _write_json(finance_standard.to_mapping(), outputs["standards/finance"]) + _write_json(shougang_standard.to_mapping(), outputs["standards/shougang"]) + for key in ("finance", "shougang", "infra"): + _write_json(alignments[key], outputs[f"provenance/{key}_alignment"]) + _write_json(summary, outputs["provenance/summary"]) + + for name, path in outputs.items(): + print(f"wrote: {_repo_relative(path)}") + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/src/agent/standards/__init__.py b/src/agent/standards/__init__.py new file mode 100644 index 0000000..0495e2c --- /dev/null +++ b/src/agent/standards/__init__.py @@ -0,0 +1,64 @@ +"""Canonical standard layer (Phase 1). + +Lossless, auditable, reproducible canonical form of the original +classification/grading standards. Distinct from sample labels: see +docs/design/data_level_design.md (Phase 0). +""" + +from .contracts import ( + LEVELS, + CanonicalStandard, + ScopedAnnotation, + SourceRef, + StandardCategory, + StandardCategoryBuilder, + clean, + compact, + normalize_standard_level, + strip_code, +) +from .sources import ( + CellInfo, + MergedCellResolver, + RawEntry, + ReaderResult, + read_finance_standard_guide, + read_guanji_catalog, +) +from .build import ( + BuildIssue, + StandardBuildReport, + build_finance_standard, + build_shougang_standard, + resolve_standard_dataset, +) +from .align import ( + align_dataset_to_standard, + load_canonical_records, +) + +__all__ = [ + "LEVELS", + "CanonicalStandard", + "ScopedAnnotation", + "SourceRef", + "StandardCategory", + "StandardCategoryBuilder", + "clean", + "compact", + "normalize_standard_level", + "strip_code", + "CellInfo", + "MergedCellResolver", + "RawEntry", + "ReaderResult", + "read_finance_standard_guide", + "read_guanji_catalog", + "BuildIssue", + "StandardBuildReport", + "build_finance_standard", + "build_shougang_standard", + "resolve_standard_dataset", + "align_dataset_to_standard", + "load_canonical_records", +] diff --git a/src/agent/standards/align.py b/src/agent/standards/align.py new file mode 100644 index 0000000..1fbd34e --- /dev/null +++ b/src/agent/standards/align.py @@ -0,0 +1,197 @@ +"""sample <-> canonical standard alignment audit (Phase 1). + +Reads canonical records (data/canonical//all.json — the same records +the training pipeline consumes, unchanged) and joins them to a +``CanonicalStandard`` by the training alias ``category_id``. + +Buckets (strict alias join): +- matched : sample level is among the entry levels of the category +- mismatched : entries have levels, sample level differs from all of them +- standard_level_unavailable : the category's entries have no parseable level +- standard_missing : sample category_id not in the standard +- sample_missing : standard category observed in no resolved sample +- unresolved : canonical records without a resolved target, with + evidence-only candidate standard entries by leaf name + +Pure computation: NEVER modifies sample data_level, never auto-repairs labels, +never guesses the meaning of a level. ``standard_data_level`` is only the +standard's category-level reference. +""" + +from __future__ import annotations + +import json +from collections import Counter +from pathlib import Path +from typing import Any, Mapping, Sequence + +from agent.standards.contracts import CanonicalStandard + + +def load_canonical_records(path: str | Path) -> list[dict[str, Any]]: + with Path(path).open(encoding="utf-8") as handle: + records = json.load(handle) + if not isinstance(records, list): + raise ValueError(f"{path} must be a JSON list") + return records + + +def align_dataset_to_standard( + records: Sequence[Mapping[str, Any]], + standard: CanonicalStandard, + *, + field_for_audit: str = "field_name", +) -> dict[str, Any]: + """Return a deterministic alignment report (no mutation of ``records``).""" + entries_by_category = standard.entries_by_category_id() + counts: Counter[str] = Counter() + unresolved_by_status: Counter[str] = Counter() + mismatched: list[dict[str, Any]] = [] + standard_missing: list[dict[str, Any]] = [] + level_unavailable: list[dict[str, Any]] = [] + unresolved_evidence: list[dict[str, Any]] = [] + resolved_categories: set[str] = set() + + for record in records: + counts["total"] += 1 + status = str(record.get("resolution_status", "") or "") + target = record.get("target") + if status != "resolved" or not isinstance(target, Mapping): + unresolved_by_status[status or "(no status)"] += 1 + counts["unresolved"] += 1 + unresolved_evidence.append( + _unresolved_evidence(record, status, standard) + ) + continue + category_id = str(target.get("category_id", "") or "") + sample_level = str(record.get("data_level", "") or "") + counts["resolved"] += 1 + resolved_categories.add(category_id) + + entries = entries_by_category.get(category_id) + if not entries: + counts["standard_missing"] += 1 + standard_missing.append( + { + "category_id": category_id, + "name": str(target.get("leaf_name", "") or ""), + "sample_level": sample_level, + "path": list(target.get("category_path") or ()), + "field": _audit_field(record, field_for_audit), + } + ) + continue + entry_levels = { + entry.standard_data_level + for entry in entries + if entry.standard_data_level is not None + } + if not entry_levels: + counts["standard_level_unavailable"] += 1 + level_unavailable.append( + { + "category_id": category_id, + "name": entries[0].name, + "sample_level": sample_level, + "raw_levels": sorted({e.raw_level for e in entries}), + "field": _audit_field(record, field_for_audit), + } + ) + continue + if sample_level in entry_levels: + counts["matched"] += 1 + else: + counts["mismatched"] += 1 + mismatched.append( + { + "category_id": category_id, + "name": entries[0].name, + "sample_level": sample_level, + "standard_levels": sorted(entry_levels), + "standard_entry_ids": [e.standard_entry_id for e in entries], + "source_rows": [e.source.row for e in entries], + "field": _audit_field(record, field_for_audit), + "sample_path": list(target.get("category_path") or ()), + } + ) + + # standard categories never observed as a resolved sample + sample_missing = sorted(set(entries_by_category) - resolved_categories) + + # near-alias candidates for standard-missing samples (leaf-name overlap), + # so the audit can distinguish "different standard branch" from "lost" + for item in standard_missing: + item["near_by_name"] = sorted( + { + entry.category_id + for entry in standard.entries + if entry.name == item["name"] and entry.category_id != item["category_id"] + } + ) + + resolved = counts["resolved"] + return { + "standard": standard.dataset, + "standard_entries": len(standard.entries), + "training_categories": len(entries_by_category), + "training_categories_observed": len(resolved_categories), + "training_categories_unobserved": len(sample_missing), + "sample_counts": { + "total": counts["total"], + "resolved": counts["resolved"], + "unresolved": counts["unresolved"], + "matched": counts["matched"], + "mismatched": counts["mismatched"], + "standard_missing": counts["standard_missing"], + "standard_level_unavailable": counts["standard_level_unavailable"], + }, + "resolved_match_rate": round(counts["matched"] / resolved, 4) if resolved else 0.0, + "unresolved_by_status": dict(sorted(unresolved_by_status.items())), + "unresolved_evidence": sorted( + unresolved_evidence, + key=lambda x: (x["status"], x["leaf_name"]), + ), + "mismatched_samples": sorted(mismatched, key=lambda x: (x["category_id"], x["field"])), + "standard_missing_samples": sorted(standard_missing, key=lambda x: x["category_id"]), + "standard_level_unavailable_samples": sorted( + level_unavailable, key=lambda x: (x["category_id"], x["field"]) + ), + "sample_missing_standard_categories": sample_missing, + } + + +def _unresolved_evidence( + record: Mapping[str, Any], + status: str, + standard: CanonicalStandard, +) -> dict[str, Any]: + """Evidence-only: which standard entries share the record's leaf name. + + Aids the alias/near-alignment audit for unresolved samples. Never repairs. + """ + classification = record.get("classification") + leaf = "" + if isinstance(classification, Mapping): + leaf = str(classification.get("level_4", "") or "") + candidates = sorted( + entry.category_id + for entry in standard.entries + if leaf and entry.name == leaf + ) + return { + "status": status, + "leaf_name": leaf, + "candidate_standard_categories": candidates, + } + + +def _audit_field(record: Mapping[str, Any], field_for_audit: str) -> str: + metadata = record.get("metadata") + if isinstance(metadata, Mapping): + value = metadata.get(field_for_audit, "") + if value: + return str(value) + return "" + + +__all__ = ["load_canonical_records", "align_dataset_to_standard"] diff --git a/src/agent/standards/build.py b/src/agent/standards/build.py new file mode 100644 index 0000000..893a401 --- /dev/null +++ b/src/agent/standards/build.py @@ -0,0 +1,444 @@ +"""Canonical standard builders (Phase 1). + +Turn raw standard rows (from merge-aware ``sources``) into a LOSSESS, +auditable ``CanonicalStandard``: +- ONE entry per real standard row — no aggregation in the fact layer. + ``standard_entry_id`` is the true source identity; ``category_id`` is the + legacy training/registry alias (projection; Phase 2 decides membership). +- Hierarchy facts that a leaf INHERITS from a merged group (finance 二级/三级 + 定义, shougang 一级/二级/三级 定义, resource) are kept per entry in + ``raw_fields`` WITH their source-cell / merged-range provenance. +- GRID-scoped annotations (finance 备注 J, 部门意见 K) are kept at standard + level as ``ScopedAnnotation`` carrying their original merged range and the + entry ids they apply to — never copied into a single leaf as private info. +- Grading columns are normalized L1..L4; unparseable values are reported and + kept raw, never fixed. Placeholder / malformed rows (shougang \"——\" with no + code, NaN) are skipped and reported; reader-level issues are merged in. + +Deterministic: every entry preserved and sorted by standard_entry_id; +annotations sorted by annotation_id; no reliance on first-seen order. +""" + +from __future__ import annotations + +import re +from collections import Counter, defaultdict +from dataclasses import dataclass, field +from pathlib import Path +from typing import Any, Iterable, Mapping, Sequence + +from agent.task.identity import qualified_category_id +from agent.standards.contracts import ( + CanonicalStandard, + ScopedAnnotation, + SourceRef, + StandardCategory, + StandardCategoryBuilder, + clean, + normalize_standard_level, + strip_code, +) + +_PLACEHOLDER_NAMES = {"——", "nan", "none", "-", ""} +_TRAILING_CODE_RE = re.compile( + r"[\(\[(【]\s*([A-Za-z]+\d*(?:-\d+)*)\s*[\)\])】]\s*$" +) + + +@dataclass(frozen=True) +class BuildIssue: + kind: str + detail: str + + def to_mapping(self) -> dict[str, str]: + return {"kind": self.kind, "detail": self.detail} + + +@dataclass +class StandardBuildReport: + dataset: str + id_strategy: str + standard_name: str + source_file: str + source_sheet: str + entries_read: int = 0 + standard_entries_out: int = 0 + training_categories: int = 0 + level_distribution: dict[str, int] = field(default_factory=dict) + issues: list[BuildIssue] = field(default_factory=list) + + def to_mapping(self) -> dict[str, Any]: + return { + "dataset": self.dataset, + "id_strategy": self.id_strategy, + "standard_name": self.standard_name, + "source_file": self.source_file, + "source_sheet": self.source_sheet, + "entries_read": self.entries_read, + "standard_entries_out": self.standard_entries_out, + "training_categories": self.training_categories, + "level_distribution": dict(self.level_distribution), + "issues": sorted( + (issue.to_mapping() for issue in self.issues), + key=lambda i: (i["kind"], i["detail"]), + ), + } + + +def _add_issue(report: StandardBuildReport, kind: str, detail: str) -> None: + report.issues.append(BuildIssue(kind=kind, detail=detail)) + + +def _dedupe_path(parts: Sequence[str]) -> tuple[str, ...]: + """Drop empty parts and consecutive duplicates (a leaf living at level 3 + would otherwise repeat its name in the path).""" + result: list[str] = [] + for part in parts: + part = part.strip() + if not part: + continue + if result and result[-1] == part: + continue + result.append(part) + return tuple(result) + + +def _entry_value(entry: Any, key: str) -> Any: + if hasattr(entry, key): + return getattr(entry, key) + if isinstance(entry, Mapping): + return entry.get(key) + return None + + +def _prov(entry: Any, key: str) -> Mapping[str, Any]: + provenance = _entry_value(entry, "provenance") or {} + value = provenance.get(key) + return value if isinstance(value, Mapping) else {} + + +def _raw_field(entry: Any, key: str): + """Build one raw_fields item ``{value, source_cell, merged_range, …}`` for + a non-empty inherited hierarchy field, else None.""" + value = clean(_entry_value(entry, key) or "") + if not value: + return None + info = dict(_prov(entry, key)) + info.pop("value", None) + item = {"value": value} + item.update(info) + return item + + +def _scoped_annotations( + dataset: str, + rows: Sequence[tuple[int, str, str, str, str, str, int, int]], +) -> tuple[ScopedAnnotation, ...]: + """Build gold-scope annotations from per-entry annotation sightings. + + Groups sightings by ``(type, text, scope_key)`` where scope_key is the + merged range ("J93:J132") — or the cell itself ("J55") for an unmerged + cell. Two unmerged cells with identical text are therefore DIFFERENT + annotations (each keeps its own row), never merged into one spanning + range. ``start_row/end_row`` come from the merged-range provenance or the + single cell's own row, never from the observed member subset. + """ + groups: dict[tuple[str, str, str], list[tuple[int, str, str, int, int]]] = defaultdict(list) + for row, entry_id, type_, text, source_cell, merged_range, start, end in rows: + if not text: + continue + scope_key = merged_range or source_cell or f"cell-{row}" + groups[(type_, text, scope_key)].append((row, entry_id, source_cell, start, end)) + + annotations: list[ScopedAnnotation] = [] + for (type_, text, scope_key), members in groups.items(): + members.sort(key=lambda m: (m[3] if m[3] is not None else m[0], m[0])) + start_rows = [m[3] if m[3] is not None else m[0] for m in members] + end_rows = [m[4] if m[4] is not None else m[0] for m in members] + start_row = min(start_rows) + end_row = max(end_rows) + source_cell = members[0][2] + # a merged range always contains ':' (e.g. "J93:J132"); a plain cell + # ("J55") is not a range -> merged_range=None keeps the single scope + merged_range = scope_key if ":" in scope_key else None + annotation_id = f"{dataset}-{type_}-{start_row}-{end_row}" + annotations.append( + ScopedAnnotation( + annotation_id=annotation_id, + type=type_, + text=text, + source_cell=source_cell or scope_key, + merged_range=merged_range, + start_row=start_row, + end_row=end_row, + applies_to_standard_entry_ids=tuple( + sorted(entry_id for _, entry_id, _, _, _ in members) + ), + ) + ) + ids = [a.annotation_id for a in annotations] + if len(set(ids)) != len(ids): + raise ValueError(f"duplicate annotation ids (internal grouping error): {ids}") + return tuple(sorted(annotations, key=lambda a: a.annotation_id)) + + +def _finalize_report( + report: StandardBuildReport, + entries: Sequence[StandardCategory], + reader_issues: Sequence[str], +) -> None: + level_counter: Counter[str] = Counter() + for entry in entries: + level_counter[entry.standard_data_level or "''"] += 1 + report.level_distribution = dict(sorted(level_counter.items())) + report.training_categories = len({entry.category_id for entry in entries}) + for detail in reader_issues: + report.issues.append(BuildIssue(kind="reader_issue", detail=str(detail))) + + +def build_finance_standard( + entries: Sequence[Any], + *, + source_file: str, + source_sheet: str, + dataset: str = "finance", + standard_name: str = "金融行业数据安全分类分级标准指南", + reader_issues: Sequence[str] = (), +) -> tuple[CanonicalStandard, StandardBuildReport]: + """Build the finance canonical standard from merge-aware raw entries. + + standard_entry_id = finance:L1.L2.L3.leaf (true identity incl. 三级); + category_id = finance:L1.L2.leaf (training alias). Inherited 二级/三级 + 定义 go to ``raw_fields``; 备注/部门意见 become ``scoped_annotations``. + """ + report = StandardBuildReport( + dataset=dataset, + id_strategy="path", + standard_name=standard_name, + source_file=source_file, + source_sheet=source_sheet, + entries_read=len(entries), + ) + built: list[StandardCategory] = [] + annotation_rows: list[tuple[int, str, str, str, str, str, int, int]] = [] + for entry in entries: + level_1 = clean(_entry_value(entry, "level_1") or "") + level_2 = clean(_entry_value(entry, "level_2") or "") + level_3 = clean(_entry_value(entry, "level_3") or "") + leaf = clean(_entry_value(entry, "leaf") or "") + content = clean(_entry_value(entry, "description") or "") + raw_level = str(_entry_value(entry, "raw_level") or "").strip() + row = _entry_value(entry, "row") + + if not leaf: + _add_issue(report, "empty_leaf_skipped", f"row {row}: empty leaf") + continue + standard_entry_id = qualified_category_id(dataset, (level_1, level_2, level_3, leaf)) + category_id = qualified_category_id(dataset, (level_1, level_2, leaf)) + path = _dedupe_path((level_1, level_2, level_3, leaf)) + level, raw_clean = normalize_standard_level(raw_level) + if raw_level and level is None: + _add_issue( + report, + "level_unparseable", + f"row {row} category {leaf!r}: raw level {raw_clean!r} kept as-is, " + f"standard_data_level=null (not guessed)", + ) + raw_fields: dict[str, Any] = {} + for key in ("level_2_definition", "level_3_definition"): + item = _raw_field(entry, key) + if item is not None: + raw_fields[key] = item + + built.append( + StandardCategoryBuilder( + standard_entry_id=standard_entry_id, + category_id=category_id, + name=leaf, + path=path, + description=content, + code=None, + raw_level=raw_level, + raw_fields=raw_fields, + source_file=source_file, + source_sheet=source_sheet, + source_row=row, + ).build() + ) + # scoped-annotation sightings (remark / department opinion) + remark = clean(_entry_value(entry, "remark") or "") + opinion = clean(_entry_value(entry, "department_opinion") or "") + if remark: + _push_annotation_sighting( + annotation_rows, row, standard_entry_id, "remark", remark, + _prov(entry, "remark"), + ) + if opinion: + _push_annotation_sighting( + annotation_rows, row, standard_entry_id, "department_opinion", + opinion, _prov(entry, "department_opinion"), + ) + + built.sort(key=lambda c: c.standard_entry_id) + _assert_unique_entry_ids(built, "finance") + report.standard_entries_out = len(built) + _finalize_report(report, built, reader_issues) + scoped = _scoped_annotations(dataset, annotation_rows) + return CanonicalStandard( + dataset=dataset, + id_strategy="path", + standard_source=SourceRef(file=source_file, sheet=source_sheet), + standard_name=standard_name, + entries=tuple(built), + scoped_annotations=scoped, + ), report + + +def build_shougang_standard( + entries: Sequence[Any], + *, + source_file: str, + source_sheet: str, + dataset: str = "shougang", + standard_name: str = "首钢京唐数据分类分级目录(关基)", + reader_issues: Sequence[str] = (), +) -> tuple[CanonicalStandard, StandardBuildReport]: + """Build the shougang canonical standard from the raw guanji catalog. + + standard_entry_id == category_id == guanji code. Inherited 一级/二级/三级 + 定义 and 数据来源(resource) are kept in ``raw_fields`` with provenance. + """ + report = StandardBuildReport( + dataset=dataset, + id_strategy="code", + standard_name=standard_name, + source_file=source_file, + source_sheet=source_sheet, + entries_read=len(entries), + ) + built: list[StandardCategory] = [] + for entry in entries: + level_1 = clean(_entry_value(entry, "level_1") or "") + level_2 = clean(_entry_value(entry, "level_2") or "") + level_3 = clean(_entry_value(entry, "level_3") or "") + raw_leaf = clean(_entry_value(entry, "leaf") or "") + description = clean(_entry_value(entry, "description") or "") + content = clean(_entry_value(entry, "content") or "") + raw_level = str(_entry_value(entry, "raw_level") or "").strip() + row = _entry_value(entry, "row") + + if not raw_leaf or raw_leaf.lower() in _PLACEHOLDER_NAMES or raw_leaf in _PLACEHOLDER_NAMES: + raw_leaf = clean(_entry_value(entry, "level_3") or "") + if not raw_leaf or raw_leaf in _PLACEHOLDER_NAMES: + _add_issue( + report, + "placeholder_skipped", + f"row {row}: no real leaf level (all ——); skipped and reported", + ) + continue + match = _TRAILING_CODE_RE.search(raw_leaf) + if match: + code = match.group(1) + name = strip_code(raw_leaf) + else: + _add_issue( + report, + "no_code", + f"row {row}: leaf {raw_leaf!r} has no category code; skipped " + f"(identity is code-based, not guessed)", + ) + continue + path = _dedupe_path((strip_code(level_1), strip_code(level_2), strip_code(level_3), name)) + level, raw_clean = normalize_standard_level(raw_level) + if raw_level and level is None: + _add_issue( + report, + "level_unparseable", + f"row {row} category {name!r}: raw level {raw_clean!r} kept as-is, " + f"standard_data_level=null (not guessed)", + ) + raw_fields: dict[str, Any] = {} + for key in ("level_1_definition", "level_2_definition", "level_3_definition", "resource"): + item = _raw_field(entry, key) + if item is not None: + raw_fields[key] = item + + built.append( + StandardCategoryBuilder( + standard_entry_id=code, + category_id=code, + name=name, + path=path, + description=description, + code=code, + raw_level=raw_level, + content=content, + raw_fields=raw_fields, + source_file=source_file, + source_sheet=source_sheet, + source_row=row, + ).build() + ) + built.sort(key=lambda c: c.standard_entry_id) + _assert_unique_entry_ids(built, "shougang") + report.standard_entries_out = len(built) + _finalize_report(report, built, reader_issues) + return CanonicalStandard( + dataset=dataset, + id_strategy="code", + standard_source=SourceRef(file=source_file, sheet=source_sheet), + standard_name=standard_name, + entries=tuple(built), + ), report + + +def _push_annotation_sighting( + rows: list[tuple[int, str, str, str, str, str, int, int]], + row: int, + entry_id: str, + type_: str, + text: str, + prov: Mapping[str, Any], +) -> None: + start = prov.get("start_row") + end = prov.get("end_row") + rows.append( + ( + row, + entry_id, + type_, + text, + str(prov.get("source_cell") or ""), + prov.get("merged_range"), + int(start) if start is not None else None, + int(end) if end is not None else None, + ) + ) + + +def _assert_unique_entry_ids(built: Sequence[StandardCategory], dataset: str) -> None: + unique = {entry.standard_entry_id for entry in built} + if len(unique) != len(built): + raise ValueError( + f"{dataset} standard_entry_id must be unique; got " + f"{len(built) - len(unique)} duplicate(s)" + ) + + +def resolve_standard_dataset(dataset: str) -> str | None: + """Which canonical standard owns a dataset's category facts.""" + return { + "finance": "finance", + "shougang": "shougang", + "infra": "shougang", + "pers_info": None, + }[dataset] + + +__all__ = [ + "BuildIssue", + "StandardBuildReport", + "build_finance_standard", + "build_shougang_standard", + "resolve_standard_dataset", +] diff --git a/src/agent/standards/contracts.py b/src/agent/standards/contracts.py new file mode 100644 index 0000000..db6adab --- /dev/null +++ b/src/agent/standards/contracts.py @@ -0,0 +1,385 @@ +"""Canonical standard contracts (Phase 1). + +Phase 0 frozen semantics respected here: +- ``sample_data_level`` (per-field label in processed/canonical) and + ``standard_data_level`` (grade the classification/grading standard assigns + to a category) are distinct and are NEVER merged or overwritten. +- ``standard_data_level`` is a category-level reference only; no natural- + language semantics of L1..L4 are asserted here, and levels are not assumed + equivalent across datasets. +- ``path`` stores the REAL source-hierarchy depth of the standard; empty + levels are omitted (no invented padding). +- category_id continues the existing stable identity strategy so the + standard stays joinable to the current registry/canonical targets. + +Determinism: categories are canonicalized by sorted category_id and JSON is +written with sort_keys; ``fingerprint()`` is a sha256 over the canonical +categories payload (no timestamps, no machine-local paths). +""" + +from __future__ import annotations + +import hashlib +import json +import re +from dataclasses import dataclass, field +from typing import Any, Iterable, Mapping, Sequence + +_WS_COLLAPSE_RE = re.compile(r"\s+") +_WS_REMOVE_RE = re.compile(r"\s+") +_TRAILING_CODE_RE = re.compile( + r"\s*[\(\[(【]\s*([A-Za-z]+\d*(?:-\d+)*)\s*[\)\])】]\s*$" +) + +LEVELS = ("L1", "L2", "L3", "L4") + +# Aliases accepted when normalizing a raw standard level value. Anything not +# covered here is kept raw and reported, never guessed. +_LEVEL_ALIASES: dict[str, str] = { + "1": "L1", "2": "L2", "3": "L3", "4": "L4", + "L1": "L1", "L2": "L2", "L3": "L3", "L4": "L4", + "LEVEL1": "L1", "LEVEL2": "L2", "LEVEL3": "L3", "LEVEL4": "L4", + "1级": "L1", "2级": "L2", "3级": "L3", "4级": "L4", +} + + +def clean(value: Any) -> str: + """Collapse every whitespace run to a single space and strip.""" + if value is None: + return "" + return _WS_COLLAPSE_RE.sub(" ", str(value).strip()) + + +def compact(value: Any) -> str: + """Remove every whitespace character (identity seed; matches identity.py).""" + if value is None: + return "" + return _WS_REMOVE_RE.sub("", str(value)) + + +def strip_code(text: str) -> str: + """Remove a trailing classification code such as (A1-1-1), (A), 【A】.""" + return _TRAILING_CODE_RE.sub("", text).strip() + + +def normalize_standard_level( + raw: Any, +) -> tuple[str | None, str]: + """Return (canonical L1..L4 | None, cleaned raw value). + + Unparseable values (e.g. 'l', '3 4' from the finance guide) map to None + and keep the raw text; callers must report them, never fix or guess. + """ + text = clean(raw) + if not text: + return None, "" + normalized = _LEVEL_ALIASES.get(text.upper()) + return (normalized, text) if normalized is not None else (None, text) + + +@dataclass(frozen=True) +class SourceRef: + """Traceable origin of one standard category.""" + + file: str = "" + sheet: str = "" + row: int | None = None + + def to_mapping(self) -> dict[str, Any]: + return {"file": self.file, "sheet": self.sheet, "row": self.row} + + @classmethod + def from_mapping(cls, value: Mapping[str, Any]) -> "SourceRef": + return cls( + file=str(value.get("file", "") or ""), + sheet=str(value.get("sheet", "") or ""), + row=value.get("row"), + ) + + +@dataclass(frozen=True) +class ScopedAnnotation: + """A GRID-scoped annotation from the source standard (e.g. finance 备注). + + A merged cell is a SINGLE source value that applies to a RANGE of rows + (and therefore to several standard entries). It must never be copied into + one leaf as if it were that leaf's private remark — the original merged + scope is preserved here. + """ + + annotation_id: str + type: str # e.g. "remark" / "department_opinion" + text: str + source_cell: str + merged_range: str | None + start_row: int + end_row: int + applies_to_standard_entry_ids: tuple[str, ...] = () + + def to_mapping(self) -> dict[str, Any]: + return { + "annotation_id": self.annotation_id, + "type": self.type, + "text": self.text, + "source_cell": self.source_cell, + "merged_range": self.merged_range, + "start_row": self.start_row, + "end_row": self.end_row, + "applies_to_standard_entry_ids": list(self.applies_to_standard_entry_ids), + } + + @classmethod + def from_mapping(cls, value: Mapping[str, Any]) -> "ScopedAnnotation": + return cls( + annotation_id=str(value.get("annotation_id", "") or ""), + type=str(value.get("type", "") or ""), + text=str(value.get("text", "") or ""), + source_cell=str(value.get("source_cell", "") or ""), + merged_range=value.get("merged_range"), + start_row=int(value.get("start_row", 0) or 0), + end_row=int(value.get("end_row", 0) or 0), + applies_to_standard_entry_ids=tuple( + str(item) for item in value.get("applies_to_standard_entry_ids", ()) + ), + ) + + +@dataclass(frozen=True) +class StandardCategory: + """One entry of the LOSSESS canonical standard (one raw standard row). + + - standard_entry_id: the TRUE identity of this entry in the raw standard + (finance includes the real 三级子类: ``finance:{L1}.{L2}.{L3}.{L4}``; + shougang = guanji code). Unique within one standard. + - category_id: the legacy training/registry alias (finance L1-L2-leaf = + ``DatasetConfig.identity_fields``; shougang = same code). NOT unique — + several distinct standard entries may project onto one training + category (Phase 2 decision; Phase 1 keeps every entry). + - raw_fields: source-specific EXTRA fields (hierarchy definitions, + resource) with per-value provenance (source_cell / merged_range). These + are grid facts that a leaf inherits from its merged group; they are NOT + leaf-private labels. Scoped annotations (remarks) live at standard level. + """ + + standard_entry_id: str + category_id: str + name: str + path: tuple[str, ...] = () + description: str = "" + code: str | None = None + standard_data_level: str | None = None + raw_level: str = "" + content: str = "" + source: SourceRef = field(default_factory=SourceRef) + raw_fields: Mapping[str, Mapping[str, Any]] = field(default_factory=dict) + + def to_mapping(self) -> dict[str, Any]: + mapping: dict[str, Any] = { + "standard_entry_id": self.standard_entry_id, + "category_id": self.category_id, + "name": self.name, + "path": list(self.path), + "description": self.description, + "code": self.code, + "standard_data_level": self.standard_data_level, + "raw_level": self.raw_level, + "source": self.source.to_mapping(), + } + if self.content: + mapping["content"] = self.content + if self.raw_fields: + mapping["raw_fields"] = { + key: dict(item) + for key, item in sorted(self.raw_fields.items()) + } + return mapping + + @classmethod + def from_mapping(cls, value: Mapping[str, Any]) -> "StandardCategory": + source = value.get("source") or {} + raw_fields = value.get("raw_fields") or {} + return cls( + standard_entry_id=str(value.get("standard_entry_id", "") or ""), + category_id=str(value.get("category_id", "") or ""), + name=str(value.get("name", "") or ""), + path=tuple(str(p) for p in value.get("path", ())), + description=str(value.get("description", "") or ""), + code=value.get("code"), + standard_data_level=value.get("standard_data_level"), + raw_level=str(value.get("raw_level", "") or ""), + content=str(value.get("content", "") or ""), + source=SourceRef.from_mapping( + source if isinstance(source, Mapping) else {} + ), + raw_fields=( + {str(k): dict(v) for k, v in raw_fields.items()} + if isinstance(raw_fields, Mapping) + else {} + ), + ) + + +@dataclass(frozen=True) +class CanonicalStandard: + """The LOSSESS canonical standard for one dataset. + + Contains one entry per real standard row (no aggregation); the + ``training_projection`` (category_id -> entry ids) is a derived downstream + alias view, never a fact-layer collapse. + """ + + dataset: str + id_strategy: str + standard_source: SourceRef + standard_name: str = "" + entries: tuple[StandardCategory, ...] = () + scoped_annotations: tuple[ScopedAnnotation, ...] = () + + @property + def categories(self) -> tuple[StandardCategory, ...]: + """Backward-compatible alias: the lossless entries.""" + return self.entries + + def by_entry_id(self) -> dict[str, StandardCategory]: + return {entry.standard_entry_id: entry for entry in self.entries} + + def entries_by_category_id(self) -> dict[str, list[StandardCategory]]: + grouped: dict[str, list[StandardCategory]] = {} + for entry in self.entries: + grouped.setdefault(entry.category_id, []).append(entry) + for values in grouped.values(): + values.sort(key=lambda e: e.standard_entry_id) + return grouped + + def training_projection(self) -> dict[str, list[str]]: + grouped: dict[str, list[str]] = {} + for entry in self.entries: + grouped.setdefault(entry.category_id, []).append(entry.standard_entry_id) + return { + key: sorted(values) + for key, values in sorted(grouped.items()) + } + + def trainable_category_count(self) -> int: + """Number of distinct training/registry categories (the 237-entries-of- + finance project onto 233 category_ids; this is that projection size).""" + return len(self.training_projection()) + + def to_mapping(self) -> dict[str, Any]: + return { + "dataset": self.dataset, + "id_strategy": self.id_strategy, + "standard_name": self.standard_name, + "standard_source": self.standard_source.to_mapping(), + "fingerprint": self.fingerprint(), + "entries": [entry.to_mapping() for entry in self.entries], + "training_projection": self.training_projection(), + "scoped_annotations": [ + annotation.to_mapping() + for annotation in sorted( + self.scoped_annotations, key=lambda a: a.annotation_id + ) + ], + } + + @classmethod + def from_mapping(cls, value: Mapping[str, Any]) -> "CanonicalStandard": + raw_entries = value.get("entries", ()) + entries = tuple( + StandardCategory.from_mapping(item) + for item in raw_entries + if isinstance(item, Mapping) + ) + source = value.get("standard_source") or {} + raw_annotations = value.get("scoped_annotations", ()) + annotations = tuple( + ScopedAnnotation.from_mapping(item) + for item in raw_annotations + if isinstance(item, Mapping) + ) + return cls( + dataset=str(value.get("dataset", "") or ""), + id_strategy=str(value.get("id_strategy", "") or ""), + standard_name=str(value.get("standard_name", "") or ""), + standard_source=SourceRef.from_mapping( + source if isinstance(source, Mapping) else {} + ), + entries=entries, + scoped_annotations=annotations, + ) + + def fingerprint(self) -> str: + payload = { + "dataset": self.dataset, + "id_strategy": self.id_strategy, + "entries": [ + {k: v for k, v in entry.to_mapping().items() if k != "source"} + for entry in sorted(self.entries, key=lambda e: e.standard_entry_id) + ], + "training_projection": self.training_projection(), + "scoped_annotations": [ + annotation.to_mapping() + for annotation in sorted( + self.scoped_annotations, key=lambda a: a.annotation_id + ) + ], + } + digest = hashlib.sha256() + digest.update( + json.dumps(payload, ensure_ascii=False, sort_keys=True).encode("utf-8") + ) + return digest.hexdigest() + + + +@dataclass(frozen=True) +class StandardCategoryBuilder: + """Deterministic canonical category from raw standard source fields. + + Kept as a small value object so build logic is trivially testable with + plain dicts (no Excel, no IO). + """ + + standard_entry_id: str + category_id: str + name: str + path: tuple[str, ...] + description: str = "" + code: str | None = None + raw_level: str = "" + content: str = "" + raw_fields: Mapping[str, Mapping[str, Any]] = field(default_factory=dict) + source_file: str = "" + source_sheet: str = "" + source_row: int | None = None + + def build(self) -> StandardCategory: + level, _ = normalize_standard_level(self.raw_level) # level kept, raw kept + return StandardCategory( + standard_entry_id=self.standard_entry_id, + category_id=self.category_id, + name=self.name, + path=self.path, + description=self.description, + code=self.code, + standard_data_level=level, + raw_level=clean(self.raw_level), + content=self.content, + raw_fields=dict(self.raw_fields), + source=SourceRef( + file=self.source_file, sheet=self.source_sheet, row=self.source_row + ), + ) + + +__all__ = [ + "LEVELS", + "clean", + "compact", + "strip_code", + "normalize_standard_level", + "SourceRef", + "StandardCategory", + "CanonicalStandard", + "StandardCategoryBuilder", +] diff --git a/src/agent/standards/sources.py b/src/agent/standards/sources.py new file mode 100644 index 0000000..f85bd08 --- /dev/null +++ b/src/agent/standards/sources.py @@ -0,0 +1,299 @@ +"""Raw standard readers (Phase 1): read the ORIGINAL standard workbooks into +plain raw-entry dicts — MERGED-RANGE-AWARE. + +The standard workbooks use vertical cell merges for GROUP/hierarchy columns: +finance 金融行业数据安全分类分级标准指南 (B..F hierarchy + J remark, K dept), +shougang 关基-数据分类分级目录 (B..G hierarchy). A merged cell is ONE source +value owned by a RANGE of rows; ``MergedCellResolver`` expands it to every +covered cell while preserving the anchor and scope, so a leaf inherits its +group's definitions/remark without misattributing them as leaf-private. + +Leaf columns (四级子类 / 内容 / 安全级别 / 分级 / 数据资源说明 / 数据来源) are +per-row cells (no merges). The three SAMPLE-source workbooks (部分金融数据 / +带分级分类… / 关基设施…用于测试) have no business-level merges and are NOT read +by this module. + +Excel layout (verified against data/raw at 2026-08-20): +- finance (sheet Table 1): B一级子类 C二级子类 D二级定义 E三级子类 F三级定义 + G四级子类 H内容 I安全级别 J备注 K部门意见. +- shougang (sheet 数据分类分级): B一级分类 C一级定义 D二级分类 E二级定义 + F三级分类(G(三级定义) H四级分类 I四级定义 J数据资源说明 K分级 L数据来源. + A "——"/empty 四级 cell means the leaf lives at 三级 (code carried by G). +""" + +from __future__ import annotations + +from dataclasses import dataclass, field +from pathlib import Path +from typing import Any, Mapping + +from agent.standards.contracts import clean + +FINANCE_SHEET = "Table 1" +SHOUGANG_SHEET = "数据分类分级" + +# finance columns (1-based): letter -> semantic key +FINANCE_COLS = { + "L1": "B", "L2": "C", "L2_DEF": "D", "L3": "E", "L3_DEF": "F", + "LEAF": "G", "DESC": "H", "LEVEL": "I", "REMARK": "J", "OPINION": "K", +} +SHOUGANG_COLS = { + "L1": "B", "L1_DEF": "C", "L2": "D", "L2_DEF": "E", "L3": "F", + "L3_DEF": "G", "LEAF": "H", "LEAF_DEF": "I", "CONTENT": "J", + "LEVEL": "K", "RESOURCE": "L", +} + + +@dataclass(frozen=True) +class CellInfo: + """One cell resolved through the merged-grid: value + scope provenance.""" + + value: Any = None + anchor_cell: str = "" + merged_range: str | None = None + start_row: int | None = None + end_row: int | None = None + inherited: bool = False + + def to_mapping(self) -> dict[str, Any]: + return { + "value": self.value, + "anchor_cell": self.anchor_cell, + "merged_range": self.merged_range, + "start_row": self.start_row, + "end_row": self.end_row, + "inherited": self.inherited, + } + + +class MergedCellResolver: + """Resolve any cell of a worksheet with merged-range awareness. + + Non-anchor cells inside a merged range return the ANCHOR's value plus the + merged scope (anchor_cell / merged_range / start..end / inherited=True). + Plain cells return their own value with inherited=False and no range. + """ + + def __init__(self, workbook, sheet: str): + if sheet not in workbook.sheetnames: + raise ValueError(f"sheet {sheet!r} not found") + self._sheet = workbook[sheet] + self._ranges = list(self._sheet.merged_cells.ranges) + + def close(self) -> None: + if self._sheet is not None: + pass # workbook managed by caller + + def cell(self, row: int, col) -> CellInfo: + """``col`` is a 1-based integer or a column letter (e.g. 'J').""" + from openpyxl.utils import column_index_from_string, get_column_letter + + if isinstance(col, str): + col = column_index_from_string(col) + cell = self._sheet.cell(row=row, column=col) + for rng in self._ranges: + if rng.min_row <= row <= rng.max_row and rng.min_col <= col <= rng.max_col: + anchor_value = self._sheet.cell(rng.min_row, rng.min_col).value + return CellInfo( + value=anchor_value, + anchor_cell=f"{get_column_letter(rng.min_col)}{rng.min_row}", + merged_range=str(rng), + start_row=rng.min_row, + end_row=rng.max_row, + inherited=not (row == rng.min_row and col == rng.min_col), + ) + return CellInfo( + value=cell.value, + anchor_cell=f"{get_column_letter(col)}{row}", + inherited=False, + ) + + +@dataclass(frozen=True) +class RawEntry: + """One leaf row of a raw standard, with its traceable origin and the + hierarchy/annotation fields it INHERITS from merged groups.""" + + level_1: str = "" + level_2: str = "" + level_3: str = "" + leaf: str = "" + description: str = "" + content: str = "" + raw_level: str = "" + resource: str = "" + level_1_definition: str = "" + level_2_definition: str = "" + level_3_definition: str = "" + remark: str = "" + department_opinion: str = "" + sheet: str = "" + row: int | None = None + provenance: Mapping[str, Mapping[str, Any]] = field(default_factory=dict) + + def to_mapping(self) -> dict[str, Any]: + return { + "level_1": self.level_1, + "level_2": self.level_2, + "level_3": self.level_3, + "leaf": self.leaf, + "description": self.description, + "content": self.content, + "raw_level": self.raw_level, + "resource": self.resource, + "level_1_definition": self.level_1_definition, + "level_2_definition": self.level_2_definition, + "level_3_definition": self.level_3_definition, + "remark": self.remark, + "department_opinion": self.department_opinion, + "sheet": self.sheet, + "row": self.row, + "provenance": dict(self.provenance), + } + + +@dataclass(frozen=True) +class ReaderResult: + entries: tuple[RawEntry, ...] = () + issues: tuple[str, ...] = () + + +def _open_xlsx(path: Path): + """Load a workbook with cached values (data_only) for merged-range reads.""" + try: + import openpyxl + except ImportError as exc: # pragma: no cover - optional dependency + raise RuntimeError("reading raw standard workbooks requires openpyxl") from exc + return openpyxl.load_workbook(path, data_only=True) + + +def _prov(cell_info: CellInfo) -> Mapping[str, Any]: + return { + "source_cell": cell_info.anchor_cell, + "merged_range": cell_info.merged_range, + "start_row": cell_info.start_row, + "end_row": cell_info.end_row, + "inherited": cell_info.inherited, + } + + +def read_finance_standard_guide(path: str | Path) -> ReaderResult: + """Extract leaf rows of the finance grading-standard guide (merge-aware).""" + workbook = _open_xlsx(Path(path)) + resolver = MergedCellResolver(workbook, FINANCE_SHEET) + entries: list[RawEntry] = [] + issues: list[str] = [] + try: + for row in range(3, (resolver._sheet.max_row or 2) + 1): + leaf = clean(resolver.cell(row, FINANCE_COLS["LEAF"]).value) + if not leaf: + continue # trailing rows / headers; not a leaf + l1i = resolver.cell(row, FINANCE_COLS["L1"]) + l2i = resolver.cell(row, FINANCE_COLS["L2"]) + l2di = resolver.cell(row, FINANCE_COLS["L2_DEF"]) + l3i = resolver.cell(row, FINANCE_COLS["L3"]) + l3di = resolver.cell(row, FINANCE_COLS["L3_DEF"]) + li = resolver.cell(row, FINANCE_COLS["DESC"]) + lvi = resolver.cell(row, FINANCE_COLS["LEVEL"]) + ri = resolver.cell(row, FINANCE_COLS["REMARK"]) + oi = resolver.cell(row, FINANCE_COLS["OPINION"]) + entries.append( + RawEntry( + level_1=clean(l1i.value), + level_2=clean(l2i.value), + level_3=clean(l3i.value), + leaf=leaf, + description=clean(li.value), + raw_level=str(lvi.value).strip() if lvi.value is not None else "", + level_2_definition=clean(l2di.value), + level_3_definition=clean(l3di.value), + remark=clean(ri.value), + department_opinion=clean(oi.value), + sheet=FINANCE_SHEET, + row=row, + provenance={ + "level_2_definition": _prov(l2di), + "level_3_definition": _prov(l3di), + "remark": _prov(ri), + "department_opinion": _prov(oi), + }, + ) + ) + finally: + workbook.close() + return ReaderResult(tuple(entries), tuple(issues)) + + +def read_guanji_catalog(path: str | Path) -> ReaderResult: + """Extract leaf rows of the shougang (关基) grading catalog (merge-aware). + + A \"——\"/empty 四级 cell means the leaf lives at 三级 (the code/name is + carried by the 三级 column, which is itself a merged group cell). + """ + workbook = _open_xlsx(Path(path)) + resolver = MergedCellResolver(workbook, SHOUGANG_SHEET) + entries: list[RawEntry] = [] + issues: list[str] = [] + try: + for row in range(3, (resolver._sheet.max_row or 2) + 1): + l1i = resolver.cell(row, SHOUGANG_COLS["L1"]) + l2i = resolver.cell(row, SHOUGANG_COLS["L2"]) + l3i = resolver.cell(row, SHOUGANG_COLS["L3"]) + leaf_h = resolver.cell(row, SHOUGANG_COLS["LEAF"]) + leaf_h_def = resolver.cell(row, SHOUGANG_COLS["LEAF_DEF"]) + if (leaf_h.value is None or clean(leaf_h.value) in ("", "——")): + # leaf sits at 三级: name+code and its definition come from G + if l3i.value is None or not clean(l3i.value): + issues.append(f"shougang row {row}: no real leaf level") + continue + leaf = clean(l3i.value) + description = clean( + resolver.cell(row, SHOUGANG_COLS["L3_DEF"]).value + ) + else: + leaf = clean(leaf_h.value) + description = clean(leaf_h_def.value) + content = clean(resolver.cell(row, SHOUGANG_COLS["CONTENT"]).value) + _level_cell = resolver.cell(row, SHOUGANG_COLS["LEVEL"]).value + raw_level = str(_level_cell).strip() if _level_cell is not None else "" + resource = clean(resolver.cell(row, SHOUGANG_COLS["RESOURCE"]).value) + l1di = resolver.cell(row, SHOUGANG_COLS["L1_DEF"]) + l2di = resolver.cell(row, SHOUGANG_COLS["L2_DEF"]) + l3di = resolver.cell(row, SHOUGANG_COLS["L3_DEF"]) + lresi = resolver.cell(row, SHOUGANG_COLS["RESOURCE"]) + entries.append( + RawEntry( + level_1=clean(l1i.value), + level_2=clean(l2i.value), + level_3=clean(l3i.value), + leaf=leaf, + description=description, + content=content, + raw_level=raw_level, + resource=resource, + level_1_definition=clean(l1di.value), + level_2_definition=clean(l2di.value), + level_3_definition=clean(l3di.value), + sheet=SHOUGANG_SHEET, + row=row, + provenance={ + "level_1_definition": _prov(l1di), + "level_2_definition": _prov(l2di), + "level_3_definition": _prov(l3di), + "resource": _prov(lresi), + }, + ) + ) + finally: + workbook.close() + return ReaderResult(tuple(entries), tuple(issues)) + + +__all__ = [ + "CellInfo", + "MergedCellResolver", + "RawEntry", + "ReaderResult", + "read_finance_standard_guide", + "read_guanji_catalog", +] diff --git a/tests/standards/test_align.py b/tests/standards/test_align.py new file mode 100644 index 0000000..1d5c0c2 --- /dev/null +++ b/tests/standards/test_align.py @@ -0,0 +1,121 @@ +"""Phase 1 canonical standard — sample<->standard alignment tests (hermetic).""" + +from __future__ import annotations + +from agent.standards.align import align_dataset_to_standard +from agent.standards.build import resolve_standard_dataset +from agent.standards.contracts import CanonicalStandard, SourceRef, StandardCategory + + +def _standard(entries: list[StandardCategory]) -> CanonicalStandard: + return CanonicalStandard( + dataset="ds", + id_strategy="code", + standard_source=SourceRef(), + entries=tuple(entries), + ) + + +def _cat(entry_id: str, level: str, category_id: str | None = None) -> StandardCategory: + return StandardCategory( + standard_entry_id=entry_id, + category_id=category_id or entry_id, + name=entry_id, + path=(entry_id,), + standard_data_level=level, + raw_level=level, + ) + + +def _record(category_id: str | None, level: str, status: str = "resolved", leaf: str | None = None): + record = {"data_level": level, "resolution_status": status} + if category_id is not None: + record["target"] = { + "category_id": category_id, + "leaf_name": category_id, + "category_path": [category_id], + } + if leaf is not None: + record["classification"] = {"level_4": leaf} + return record + + +def test_alignment_buckets(): + standard = _standard( + [ + _cat("A", "L1"), + _cat("B", "L3"), + _cat("C", "L1"), # observed in no sample -> sample_missing + ] + ) + records = [ + _record("A", "L1"), # matched + _record("B", "L2"), # mismatched + _record("D", "L3"), # resolved but standard_missing + _record(None, "L2", "placeholder"), # unresolved + ] + report = align_dataset_to_standard(records, standard) + assert report["sample_counts"]["total"] == 4 + assert report["sample_counts"]["resolved"] == 3 + assert report["sample_counts"]["unresolved"] == 1 + assert report["sample_counts"]["matched"] == 1 + assert report["sample_counts"]["mismatched"] == 1 + assert report["sample_counts"]["standard_missing"] == 1 + assert report["standard_missing_samples"][0]["category_id"] == "D" + assert report["sample_missing_standard_categories"] == ["C"] + + +def test_alignment_never_mutates_sample_level(): + standard = _standard([_cat("A", "L3")]) + records = [_record("A", "L1")] + report = align_dataset_to_standard(records, standard) + assert report["sample_counts"]["mismatched"] == 1 + assert records[0]["data_level"] == "L1" # untouched + + +def test_alignment_standard_level_unavailable_bucket(): + standard = CanonicalStandard( + dataset="ds", id_strategy="path", standard_source=SourceRef(), + entries=( + StandardCategory( + standard_entry_id="X", category_id="X", name="X", + standard_data_level=None, raw_level="3 4", + ), + ), + ) + report = align_dataset_to_standard([_record("X", "L3")], standard) + assert report["sample_counts"]["standard_level_unavailable"] == 1 + assert report["sample_counts"]["mismatched"] == 0 + + +def test_alignment_multi_entry_category_matches_any_entry_level(): + # one training category backed by two distinct standard entries (L2, L3): + # a sample at either level matches; a sample at an absent level mismatches + standard = _standard( + [ + _cat("finance:a.中.基本信息", "L2", category_id="finance:a.b.基本信息"), + _cat("finance:a.乙.基本信息", "L3", category_id="finance:a.b.基本信息"), + ] + ) + report = align_dataset_to_standard( + [_record("finance:a.b.基本信息", "L3")], standard + ) + assert report["sample_counts"]["matched"] == 1 + assert report["sample_counts"]["mismatched"] == 0 + + +def test_alignment_unresolved_evidence_is_evidence_only(): + standard = _standard([_cat("X", "L1", category_id="X")]) + records = [_record(None, "L2", "path_mismatch", leaf="X")] + report = align_dataset_to_standard(records, standard) + assert report["sample_counts"]["unresolved"] == 1 + assert report["unresolved_evidence"][0]["status"] == "path_mismatch" + assert report["unresolved_evidence"][0]["leaf_name"] == "X" + assert report["unresolved_evidence"][0]["candidate_standard_categories"] == ["X"] + + +def test_dataset_standard_routing(): + assert resolve_standard_dataset("finance") == "finance" + assert resolve_standard_dataset("shougang") == "shougang" + assert resolve_standard_dataset("infra") == "shougang" # shared, not a copy + assert resolve_standard_dataset("pers_info") is None # no confirmed standard diff --git a/tests/standards/test_build_finance.py b/tests/standards/test_build_finance.py new file mode 100644 index 0000000..fd548b1 --- /dev/null +++ b/tests/standards/test_build_finance.py @@ -0,0 +1,222 @@ +"""Phase 1 canonical standard — finance build tests (dict fixtures, hermetic). + +Builders accept plain dict entries ({"level_1","level_2","level_3","leaf", +"description","content","raw_level","sheet","row"}) with no Excel dependency. +""" + +from __future__ import annotations + +from agent.standards.build import build_finance_standard + + +def _entry(**kw): + base = dict(sheet="Table 1", row=1, content="", raw_level="") + base.update(kw) + return base + + +def _by_category(standard): + """Find the single entry of a category (helper for fixtures).""" + return {entry.category_id: entry for entry in standard.entries} + + +def test_finance_lossless_path_keeps_real_depth_and_no_padding(): + entries = [ + _entry( + level_1="客户", level_2="个人", level_3="个人自然信息", + leaf="个人基本概况信息", description="指个人基本情况数据", + raw_level="3", row=3, + ), + _entry( + level_1="业务", level_2="账户信息", level_3="", + leaf="基本信息", description="账户基本信息", raw_level="2", row=40, + ), + ] + standard, report = build_finance_standard( + entries, source_file="data/raw/g.xlsx", source_sheet="Table 1" + ) + by_category = _by_category(standard) + full = by_category["finance:客户.个人.个人基本概况信息"] + assert full.path == ("客户", "个人", "个人自然信息", "个人基本概况信息") # 4 real levels + assert full.standard_entry_id == "finance:客户.个人.个人自然信息.个人基本概况信息" + assert full.standard_data_level == "L3" + assert full.description == "指个人基本情况数据" + shallow = by_category["finance:业务.账户信息.基本信息"] + assert shallow.path == ("业务", "账户信息", "基本信息") # empty 三级 omitted, no padding + assert shallow.standard_entry_id == "finance:业务.账户信息..基本信息" # empty slot kept + assert shallow.standard_data_level == "L2" + assert report.issues == [] + + +def test_finance_identical_leaf_under_three_different_level3_kept_as_entries(): + # The Phase-1 contract keeps every real standard row LOSSESS: two rows with + # the same L1/L2/leaf but different 三级子类 stay two entries that share a + # training category_id (projection), instead of collapsing to one. + entries = [ + _entry( + level_1="客户", level_2="个人", level_3="个人自然信息", + leaf="个人基本概况信息", raw_level="3", row=3, + ), + _entry( + level_1="客户", level_2="个人", level_3="个人健康生理信息", + leaf="个人基本概况信息", raw_level="4", row=9, + ), + ] + standard, report = build_finance_standard( + entries, source_file="f", source_sheet="Table 1" + ) + assert len(standard.entries) == 2 + assert standard.trainable_category_count() == 1 + entry_ids = {e.standard_entry_id for e in standard.entries} + assert entry_ids == { + "finance:客户.个人.个人自然信息.个人基本概况信息", + "finance:客户.个人.个人健康生理信息.个人基本概况信息", + } + paths = {e.path for e in standard.entries} + assert ("客户", "个人", "个人自然信息", "个人基本概况信息") in paths + assert ("客户", "个人", "个人健康生理信息", "个人基本概况信息") in paths + # distinct levels are preserved per entry (no level-conflict collapse) + levels = {e.standard_data_level for e in standard.entries} + assert levels == {"L3", "L4"} + assert report.issues == [] + + +def test_finance_unparseable_levels_reported_not_fixed(): + entries = [ + _entry(level_1="经营管理", level_2="综合管理", level_3="", leaf="市场营销信息(非公开)", raw_level="l", row=150), + _entry(level_1="客户", level_2="个人", level_3="个人基本概况", leaf="个人健康生理影像信息", raw_level="3 4", row=168), + ] + standard, report = build_finance_standard( + entries, source_file="f", source_sheet="Table 1" + ) + for entry in standard.entries: + assert entry.standard_data_level is None + assert entry.raw_level in ("l", "3 4") + kinds = {i.kind for i in report.issues} + assert "level_unparseable" in kinds + assert len(report.issues) == 2 + + +def test_finance_build_deterministic_under_input_shuffle(): + entries = [ + _entry(level_1="客户", level_2="个人", level_3="个人自然信息", leaf="个人基本概况信息", raw_level="3", row=3), + _entry(level_1="业务", level_2="账户信息", level_3="", leaf="基本信息", raw_level="2", row=40), + _entry(level_1="经营管理", level_2="技术管理", level_3="系统管理信息", leaf="配置信息", raw_level="l", row=99), + ] + a, _ = build_finance_standard(list(entries), source_file="f", source_sheet="Table 1") + b, _ = build_finance_standard(list(reversed(entries)), source_file="f", source_sheet="Table 1") + assert a.fingerprint() == b.fingerprint() + assert [e.standard_entry_id for e in a.entries] == [e.standard_entry_id for e in b.entries] + assert a.to_mapping()["entries"] == b.to_mapping()["entries"] + + +def test_finance_deterministic_with_duplicate_category_id_entries(): + # Same training category, five distinct 三级 entries: order must not matter + # (the fact layer keeps every entry, so "first seen" decides nothing). + entries_a = [ + _entry(level_1="业务", level_2="合约协议", level_3=l3, leaf="基本信息", raw_level="2", row=row) + for l3, row in [("合同通用信息", 56), ("贷款业务信息", 57), ("中间业务信息", 67), ("资金业务信息", 74), ("其他支付业务信息", 79)] + ] + entries_b = list(reversed(entries_a)) + a, airep = build_finance_standard(entries_a, source_file="f", source_sheet="Table 1") + b, brep = build_finance_standard(entries_b, source_file="f", source_sheet="Table 1") + assert len(a.entries) == 5 == len(b.entries) + assert a.trainable_category_count() == 1 + assert a.fingerprint() == b.fingerprint() + # all five paths are preserved (no first-wins collapse) + paths = {e.path[-2] for e in a.entries} + assert paths == {"合同通用信息", "贷款业务信息", "中间业务信息", "资金业务信息", "其他支付业务信息"} + assert airep.standard_entries_out == 5 + + +def test_reader_issues_merged_into_build_report(): + entries = [_entry(level_1="客户", level_2="个人", level_3="个人自然信息", leaf="个人基本概况信息", raw_level="3", row=3)] + _, report = build_finance_standard( + entries, source_file="f", source_sheet="Table 1", + reader_issues=["finance row 999: too few columns"], + ) + issues = report.to_mapping()["issues"] + assert any(i["kind"] == "reader_issue" and "row 999" in i["detail"] for i in issues) + + +def _entry_with_hierarchy(levels, leaf, row, **extra): + e = _entry(level_1="业务", level_2="金融监管和服务", level_3="反洗钱业务信息", + leaf=leaf, raw_level="3", row=row) + e.update(extra) + return e + + +def test_finance_hierarchy_definitions_kept_in_raw_fields_with_provenance(): + prov = { + "level_2_definition": {"source_cell": "D93", "merged_range": "D93:D132", "start_row": 93, "end_row": 132, "inherited": True}, + "level_3_definition": {"source_cell": "F93", "merged_range": "F93:F99", "start_row": 93, "end_row": 99, "inherited": True}, + } + entries = [ + _entry_with_hierarchy( + ["业务", "金融监管和服务", "反洗钱业务信息"], "分类考核评级信息", 93, + level_2_definition="金融监管和服务域定义", + level_3_definition="反洗钱业务定义", + provenance=prov, + ) + ] + standard, _ = build_finance_standard(entries, source_file="f", source_sheet="Table 1") + entry = standard.entries[0] + assert entry.raw_fields["level_2_definition"] == { + "value": "金融监管和服务域定义", "source_cell": "D93", + "merged_range": "D93:D132", "start_row": 93, "end_row": 132, "inherited": True, + } + assert entry.raw_fields["level_3_definition"]["source_cell"] == "F93" + # remark is NOT a leaf-private raw field (it is a scoped annotation) + assert "remark" not in entry.raw_fields + + +def test_finance_scoped_annotations_group_by_merged_range(): + # three sightings: two share the J93:J132 merged range, one is a single cell + prov_merged = {"remark": {"source_cell": "J93", "merged_range": "J93:J132", "start_row": 93, "end_row": 132}} + prov_single = {"remark": {"source_cell": "J55", "merged_range": None, "start_row": 55, "end_row": 55}} + entries = [ + _entry_with_hierarchy(["业务", "金融监管和服务", "反洗钱业务信息"], "分类考核评级信息", 93, remark="宜从高设置", provenance=prov_merged), + _entry_with_hierarchy(["业务", "金融监管和服务", "反洗钱业务信息"], "行政监管信息", 100, remark="宜从高设置", provenance=prov_merged), + _entry(level_1="客户", level_2="个人", level_3="个人身份鉴别信息", leaf="特有账户信息", remark="宜从高设置", provenance=prov_single, row=55), + ] + standard, _ = build_finance_standard(entries, source_file="f", source_sheet="Table 1") + assert len(standard.scoped_annotations) == 2 + by_id = {a.annotation_id: a for a in standard.scoped_annotations} + merged = by_id["finance-remark-93-132"] + assert merged.merged_range == "J93:J132" + assert merged.start_row == 93 and merged.end_row == 132 + assert len(merged.applies_to_standard_entry_ids) == 2 + single = by_id["finance-remark-55-55"] + assert single.merged_range is None + assert len(single.applies_to_standard_entry_ids) == 1 + + +def test_finance_empty_department_opinion_produces_no_annotation(): + entries = [_entry(level_1="客户", level_2="个人", level_3="个人身份鉴别信息", leaf="特有账户信息", row=55)] + standard, _ = build_finance_standard(entries, source_file="f", source_sheet="Table 1") + assert standard.scoped_annotations == () + + +def test_finance_identical_text_in_two_unmerged_cells_is_two_annotations(): + # regression: J5 and J10 both carry the same text but are NOT merged; they + # must stay two separate scoped annotations (each its own row), never one + # annotation spanning rows 5..10. + def sighting(row): + return _entry( + level_1="客户", level_2="个人", level_3="个人自然信息", + leaf="个人基本概况信息" if row == 5 else "个人财产信息", + raw_level="3", row=row, + remark="相同的备注文字", + provenance={"remark": {"source_cell": "J%d" % row, "merged_range": None, + "start_row": row, "end_row": row}}, + ) + + standard, _ = build_finance_standard( + [sighting(5), sighting(10)], source_file="f", source_sheet="Table 1" + ) + by_id = {a.annotation_id: a for a in standard.scoped_annotations} + assert set(by_id) == {"finance-remark-5-5", "finance-remark-10-10"} + assert by_id["finance-remark-5-5"].merged_range is None + assert by_id["finance-remark-5-5"].start_row == 5 + assert by_id["finance-remark-10-10"].start_row == 10 + assert len(standard.scoped_annotations) == 2 diff --git a/tests/standards/test_build_shougang.py b/tests/standards/test_build_shougang.py new file mode 100644 index 0000000..d65acd8 --- /dev/null +++ b/tests/standards/test_build_shougang.py @@ -0,0 +1,141 @@ +"""Phase 1 canonical standard — shougang build tests (dict fixtures, hermetic).""" + +from __future__ import annotations + +from agent.standards.build import build_shougang_standard + + +def _entry(**kw): + base = dict(sheet="数据分类分级", row=1, content="", raw_level="", resource="") + base.update(kw) + return base + + +def test_shougang_code_name_path_level(): + entries = [ + _entry( + level_1="研发数据域(A)", level_2="产品研发(A1)", level_3="科研检验(A1-1)", + leaf="科研设备预约管理(A1-1-1)", description="指设备预约", + content="设备预约信息", raw_level="2", row=7, + ) + ] + standard, report = build_shougang_standard( + entries, source_file="data/raw/c.xlsx", source_sheet="数据分类分级" + ) + entry = standard.entries[0] + assert entry.standard_entry_id == "A1-1-1" + assert entry.category_id == "A1-1-1" + assert entry.code == "A1-1-1" + assert entry.name == "科研设备预约管理" + assert entry.path == ("研发数据域", "产品研发", "科研检验", "科研设备预约管理") + assert entry.description == "指设备预约" + assert entry.content == "设备预约信息" + assert entry.standard_data_level == "L2" + assert report.issues == [] + + +def test_shougang_leaf_at_level_three_no_invented_fourth_level(): + # catalog encodes 三级-level leaves with a literal "——" in the 四级 cell + entries = [ + _entry( + level_1="生产数据域(B)", level_2="生产合同(订单)(B1)", + level_3="合同归并(B1-2)", leaf="——", + description="指按照合同加工途径", raw_level="3", row=9, + ), + _entry( + level_1="管理数据域(C)", level_2="生产质量管理(C1)", + level_3="质保书管理(C1-5)", leaf="——", raw_level="1", row=108, + ), + ] + standard, report = build_shougang_standard( + entries, source_file="c", source_sheet="数据分类分级" + ) + by_id = standard.by_entry_id() + merged = by_id["B1-2"] + assert merged.name == "合同归并" + # real hierarchy depth is 3; the "——" marker does NOT become a 4th level + assert merged.path == ("生产数据域", "生产合同(订单)", "合同归并") + assert merged.standard_data_level == "L3" + assert by_id["C1-5"].standard_data_level == "L1" + + +def test_shougang_no_code_real_leaf_reported_and_skipped(): + entries = [ + _entry(level_1="管理数据域(C)", level_2="经营管理(C8)", level_3="综合管理", + leaf="无法归类的真实名称", raw_level="2", row=90), + ] + standard, report = build_shougang_standard( + entries, source_file="c", source_sheet="数据分类分级" + ) + assert len(standard.entries) == 0 + assert any(issue.kind == "no_code" for issue in report.issues) + + +def test_shougang_build_deterministic_under_input_shuffle(): + entries = [ + _entry(level_1="研发数据域(A)", level_2="产品研发(A1)", level_3="科研检验(A1-1)", leaf="科研设备预约管理(A1-1-1)", raw_level="2", row=7), + _entry(level_1="生产数据域(B)", level_2="生产合同(订单)(B1)", level_3="合同归并(B1-2)", leaf="——", raw_level="3", row=9), + _entry(level_1="管理数据域(C)", level_2="生产质量管理(C1)", level_3="检化验管理(C1-4)", leaf="认证管理(C1-4-4)", raw_level="3", row=105), + _entry(level_1="管理数据域(C)", level_2="经营管理(C8)", level_3="无码", leaf="无码类别", raw_level="2", row=90), + ] + a, _ = build_shougang_standard(list(entries), source_file="c", source_sheet="数据分类分级") + b, _ = build_shougang_standard(list(reversed(entries)), source_file="c", source_sheet="数据分类分级") + assert a.fingerprint() == b.fingerprint() + assert [e.standard_entry_id for e in a.entries] == [e.standard_entry_id for e in b.entries] + + +def test_reader_issues_merged_into_build_report(): + entries = [_entry(level_1="研发数据域(A)", level_2="产品研发(A1)", level_3="科研检验(A1-1)", leaf="科研设备预约管理(A1-1-1)", raw_level="2", row=7)] + _, report = build_shougang_standard( + entries, source_file="c", source_sheet="数据分类分级", + reader_issues=["shougang row 7: too few columns"], + ) + issues = report.to_mapping()["issues"] + assert any(i["kind"] == "reader_issue" and "row 7" in i["detail"] for i in issues) + + +def test_shougang_hierarchy_definitions_and_resource_preserved(): + prov = { + "level_1_definition": {"source_cell": "C3", "merged_range": "C3:C6"}, + "level_2_definition": {"source_cell": "E3", "merged_range": "E3:E4"}, + "level_3_definition": {"source_cell": "G3", "merged_range": "G3:G3"}, + "resource": {"source_cell": "L3", "merged_range": None, "start_row": None, "end_row": None, "inherited": False}, + } + entries = [ + _entry( + level_1="研发数据域(A)", level_2="产品研发(A1)", level_3="科研检验(A1-1)", + leaf="科研设备预约管理(A1-1-1)", raw_level="2", row=3, + level_1_definition="研发域定义", level_2_definition="产品研发定义", + level_3_definition="科研检验定义", resource="科研实验室", + provenance=prov, + ) + ] + standard, _ = build_shougang_standard(entries, source_file="c", source_sheet="数据分类分级") + entry = standard.entries[0] + assert entry.raw_fields["level_1_definition"]["source_cell"] == "C3" + assert entry.raw_fields["level_2_definition"]["merged_range"] == "E3:E4" + assert entry.raw_fields["level_3_definition"]["value"] == "科研检验定义" + assert entry.raw_fields["resource"] == { + "value": "科研实验室", "source_cell": "L3", "merged_range": None, + "start_row": None, "end_row": None, "inherited": False, + } + + +def test_shougang_level_three_leaf_keeps_inherited_definitions_and_resource(): + entries = [ + _entry( + level_1="生产数据域(B)", level_2="生产合同(订单)(B1)", + level_3="合同归并(B1-2)", leaf="——", raw_level="3", row=9, + level_1_definition="生产域定义", level_2_definition="合同域定义", + level_3_definition="合同归并定义", resource="制造管理系统", + provenance={ + "level_1_definition": {"source_cell": "C7", "merged_range": "C7:C82"}, + "resource": {"source_cell": "L9", "merged_range": None}, + }, + ) + ] + standard, _ = build_shougang_standard(entries, source_file="c", source_sheet="数据分类分级") + entry = standard.by_entry_id()["B1-2"] + assert entry.raw_fields["level_1_definition"]["value"] == "生产域定义" + assert entry.raw_fields["resource"]["value"] == "制造管理系统" + assert entry.raw_fields["level_3_definition"]["value"] == "合同归并定义" diff --git a/tests/standards/test_checksum.py b/tests/standards/test_checksum.py new file mode 100644 index 0000000..0bf16a1 --- /dev/null +++ b/tests/standards/test_checksum.py @@ -0,0 +1,51 @@ +"""Phase 1 canonical standard — raw-source checksum verification (hermetic). + +Covers Blocker-2 option B: CLI refuses to build from a missing manifest / +mismatched file, so a silently wrong raw workbook can never produce a +"canonical" standard. Uses tmp files only (no repo data dependency). +""" + +from __future__ import annotations + +import hashlib +import json + +import pytest + +from script.standard.cli import _verify_checksums + + +def _write(path, text: bytes): + path.write_bytes(text) + return hashlib.sha256(text).hexdigest() + + +def test_checksum_mismatch_raises(tmp_path): + manifest = tmp_path / "checksums.json" + good = tmp_path / "a.xlsx" + expected = _write(good, b"A" * 100) + + other = tmp_path / "b.xlsx" + _write(other, b"B" * 100) + manifest.write_text(json.dumps({"a.xlsx": expected}), encoding="utf-8") + + with pytest.raises(ValueError, match="sha256 mismatch"): + _verify_checksums({"a.xlsx": other}, manifest, allow_skip=False) + + +def test_checksum_match_passes(tmp_path): + manifest = tmp_path / "checksums.json" + good = tmp_path / "a.xlsx" + expected = _write(good, b"A" * 100) + manifest.write_text(json.dumps({"a.xlsx": expected}), encoding="utf-8") + _verify_checksums({"a.xlsx": good}, manifest, allow_skip=False) # no raise + + +def test_checksum_missing_manifest_requires_override(tmp_path): + manifest = tmp_path / "checksums.json" # deliberately absent + good = tmp_path / "a.xlsx" + _write(good, b"A" * 100) + with pytest.raises(FileNotFoundError, match="checksum manifest not found"): + _verify_checksums({"a.xlsx": good}, manifest, allow_skip=False) + # explicit override is allowed for offline tweaks + _verify_checksums({"a.xlsx": good}, manifest, allow_skip=True) diff --git a/tests/standards/test_contracts.py b/tests/standards/test_contracts.py new file mode 100644 index 0000000..3c79949 --- /dev/null +++ b/tests/standards/test_contracts.py @@ -0,0 +1,160 @@ +"""Phase 1 canonical standard — contract unit tests (hermetic, no IO / Excel).""" + +from __future__ import annotations + +import copy + +import pytest + +from agent.standards.contracts import ( + CanonicalStandard, + SourceRef, + StandardCategory, + compact, + normalize_standard_level, + strip_code, +) + + +def _category(entry_id: str, level: str | None = None, row: int = 1, **kw) -> StandardCategory: + return StandardCategory( + standard_entry_id=entry_id, + category_id=kw.get("category_id", entry_id), + name=kw.get("name", entry_id), + path=kw.get("path", (entry_id,)), + standard_data_level=level, + raw_level=level or "", + source=SourceRef(file="f.xlsx", sheet="s", row=row), + ) + + +def test_normalize_standard_level_valid_forms(): + for raw, expected in { + "1": "L1", "2": "L2", "3": "L3", "4": "L4", + "L1": "L1", "LEVEL3": "L3", "1级": "L1", "4级": "L4", + }.items(): + level, kept = normalize_standard_level(raw) + assert level == expected + assert kept == raw.strip() + assert normalize_standard_level(None) == (None, "") + assert normalize_standard_level("") == (None, "") + + +def test_normalize_standard_level_unparseable_never_guessed(): + # finance guide anomalies: 'l' (typo of 1) and '3 4' (ambiguous) must stay + # None and keep the raw text — the builder reports them, never fixes. + for raw in ("l", "3 4", "敏感级", "高"): + level, kept = normalize_standard_level(raw) + assert level is None + assert kept == raw + + +def test_compact_and_strip_code(): + assert compact("经营 管理 技术") == "经营管理技术" + assert strip_code("科研设备预约管理(A1-1-1)") == "科研设备预约管理" + assert strip_code("生产数据域\n(B)") == "生产数据域" + assert strip_code("基本信息(公开)") == "基本信息(公开)" # parens without code kept + + +def test_standard_round_trip_stable(): + standard = CanonicalStandard( + dataset="finance", + id_strategy="path", + standard_name="指南", + standard_source=SourceRef(file="data/raw/f.xlsx", sheet="Table 1", row=None), + entries=( + _category("finance:业务.交易信息.交易通用信息.交易基本信息", "L2"), + _category("finance:业务.账户信息..基本信息", "L1"), + ), + ) + mapping = standard.to_mapping() + mapping["entries"] = list(reversed(mapping["entries"])) # scramble order + rebuilt = CanonicalStandard.from_mapping(mapping) + assert rebuilt.dataset == standard.dataset + assert rebuilt.id_strategy == standard.id_strategy + assert rebuilt.fingerprint() == standard.fingerprint() + + +def test_round_trip_preserves_level_and_source(): + entry = StandardCategory( + standard_entry_id="A1-1-1", + category_id="A1-1-1", + name="科研设备预约管理", + path=("研发数据域", "产品研发", "科研检验", "科研设备预约管理"), + description="描述", + code="A1-1-1", + standard_data_level="L3", + raw_level="3", + content="资源说明", + source=SourceRef(file="x.xlsx", sheet="数据分类分级", row=7), + raw_fields={ + "level_3_definition": { + "value": "科研检验定义", "source_cell": "G7", "merged_range": None, + } + }, + ) + rebuilt = StandardCategory.from_mapping(copy.deepcopy(entry.to_mapping())) + assert rebuilt == entry + + +def test_round_trip_preserves_scoped_annotations(): + from agent.standards.contracts import ScopedAnnotation + + annotation = ScopedAnnotation( + annotation_id="finance-remark-93-132", + type="remark", + text="鉴于其特殊性", + source_cell="J93", + merged_range="J93:J132", + start_row=93, + end_row=132, + applies_to_standard_entry_ids=("a", "b"), + ) + standard = CanonicalStandard( + dataset="finance", id_strategy="path", standard_source=SourceRef(), + entries=(_category("a", "L2"), _category("b", "L3")), + scoped_annotations=(annotation,), + ) + mapping = standard.to_mapping() + rebuilt = CanonicalStandard.from_mapping(copy.deepcopy(mapping)) + assert rebuilt.scoped_annotations == (annotation,) + assert rebuilt.fingerprint() == standard.fingerprint() + + +def test_fingerprint_includes_raw_fields_and_annotations(): + a = _category("A", "L1") + b = StandardCategory( + standard_entry_id="A", category_id="A", name="A", path=("A",), + raw_fields={"level_2_definition": {"value": "x", "source_cell": "D2"}}, + ) + sa = CanonicalStandard("s", "code", standard_source=SourceRef(), entries=(a,)) + sb = CanonicalStandard("s", "code", standard_source=SourceRef(), entries=(b,)) + assert sa.fingerprint() != sb.fingerprint() # raw_fields are content, not source + + +def test_fingerprint_independent_of_source_rows(): + a = _category("A", "L1", row=1) + b = _category("A", "L1", row=99) # same content, different source row + sa = CanonicalStandard("s", "code", standard_source=SourceRef(), entries=(a,)) + sb = CanonicalStandard("s", "code", standard_source=SourceRef(), entries=(b,)) + assert sa.fingerprint() == sb.fingerprint() + + +def test_training_projection_groups_entries_by_category_id(): + standard = CanonicalStandard( + dataset="finance", + id_strategy="path", + standard_source=SourceRef(), + entries=( + _category("finance:业务.合约协议.合同通用信息.基本信息", "L2", category_id="finance:业务.合约协议.基本信息"), + _category("finance:业务.合约协议.贷款业务信息.基本信息", "L2", category_id="finance:业务.合约协议.基本信息"), + _category("finance:业务.交易信息.交易通用信息.交易基本信息", "L2", category_id="finance:业务.交易信息.交易基本信息"), + ), + ) + projection = standard.training_projection() + assert standard.trainable_category_count() == 2 + assert projection["finance:业务.合约协议.基本信息"] == [ + "finance:业务.合约协议.合同通用信息.基本信息", + "finance:业务.合约协议.贷款业务信息.基本信息", + ] + assert len(standard.entries_by_category_id()["finance:业务.合约协议.基本信息"]) == 2 diff --git a/tests/standards/test_merged_resolver.py b/tests/standards/test_merged_resolver.py new file mode 100644 index 0000000..6e72488 --- /dev/null +++ b/tests/standards/test_merged_resolver.py @@ -0,0 +1,64 @@ +"""Phase 1 canonical standard — MergedCellResolver unit tests (hermetic). + +Builds a tiny workbook on the fly, so these tests need openpyxl but no repo +data (skipped when openpyxl is missing). +""" + +from __future__ import annotations + +import pytest + +openpyxl = pytest.importorskip("openpyxl") + +from agent.standards.sources import MergedCellResolver # noqa: E402 + + +@pytest.fixture() +def wb(tmp_path): + """A sheet with one vertical merge B2:B4 and one plain cell C2.""" + path = tmp_path / "t.xlsx" + workbook = openpyxl.Workbook() + ws = workbook.active + ws.title = "S" + ws["B2"] = "group-value" + ws["C2"] = "plain-value" + ws["C3"] = "plain-other" + ws.merge_cells("B2:B4") + workbook.save(path) + workbook.close() + return path + + +def test_anchor_cell(wb): + resolver = MergedCellResolver(openpyxl.load_workbook(wb, data_only=True), "S") + info = resolver.cell(2, "B") + assert info.value == "group-value" + assert info.anchor_cell == "B2" + assert info.merged_range == "B2:B4" + assert info.start_row == 2 and info.end_row == 4 + assert info.inherited is False + assert info.to_mapping()["value"] == "group-value" + + +def test_inherited_cell(wb): + resolver = MergedCellResolver(openpyxl.load_workbook(wb, data_only=True), "S") + info = resolver.cell(4, "B") # non-anchor inside the merge + assert info.value == "group-value" # anchor value expanded + assert info.anchor_cell == "B2" + assert info.merged_range == "B2:B4" + assert info.inherited is True + + +def test_plain_cell(wb): + resolver = MergedCellResolver(openpyxl.load_workbook(wb, data_only=True), "S") + info = resolver.cell(2, "C") + assert info.value == "plain-value" + assert info.anchor_cell == "C2" + assert info.merged_range is None + assert info.inherited is False + assert info.start_row is None + + +def test_column_letter_or_index_accepted(wb): + resolver = MergedCellResolver(openpyxl.load_workbook(wb, data_only=True), "S") + assert resolver.cell(2, "B").value == resolver.cell(2, 2).value == "group-value" diff --git a/tests/standards/test_real_xlsx.py b/tests/standards/test_real_xlsx.py new file mode 100644 index 0000000..b05495b --- /dev/null +++ b/tests/standards/test_real_xlsx.py @@ -0,0 +1,224 @@ +"""Phase 1 canonical standard — integration tests against real raw workbooks. + +These read the ORIGINAL standard workbooks under data/raw (gitignored) and the +canonical dataset layer. They are skipped when the raw files are absent (CI / +fresh clone) so the suite stays green without the data provider's files. +""" + +from __future__ import annotations + +from pathlib import Path + +import pytest + +from agent.standards.align import align_dataset_to_standard, load_canonical_records +from agent.standards.build import ( + build_finance_standard, + build_shougang_standard, + resolve_standard_dataset, +) +from agent.standards.sources import ( + read_finance_standard_guide, + read_guanji_catalog, +) +from agent.task import LeafRegistry + +ROOT = Path(__file__).resolve().parents[2] +RAW = ROOT / "data" / "raw" +FIN_XLSX = RAW / "金融行业数据安全分类分级标准指南.xlsx" +SHG_XLSX = RAW / "关基-数据分类分级目录.xlsx" +CANON = ROOT / "data" / "canonical" +REG = ROOT / "cfg" / "task" / "registry" + +pytestmark = pytest.mark.skipif( + not (FIN_XLSX.is_file() and SHG_XLSX.is_file()), + reason="raw standard workbooks not present (data/raw is gitignored)", +) + + +@pytest.fixture(scope="module") +def built(): + finance_raw = read_finance_standard_guide(FIN_XLSX) + shougang_raw = read_guanji_catalog(SHG_XLSX) + finance, finance_report = build_finance_standard( + finance_raw.entries, source_file=str(FIN_XLSX), source_sheet="Table 1", + reader_issues=finance_raw.issues, + ) + shougang, shougang_report = build_shougang_standard( + shougang_raw.entries, source_file=str(SHG_XLSX), source_sheet="数据分类分级", + reader_issues=shougang_raw.issues, + ) + return finance, finance_report, shougang, shougang_report + + +def test_finance_standard_is_lossless_237_entries_233_projection(built): + finance, finance_report, _, _ = built + registry = LeafRegistry.from_path(REG / "finance.registry.json") + assert len(finance.entries) == 237 # one per real standard row — NO collapse + assert finance.trainable_category_count() == 233 + assert finance_report.standard_entries_out == 237 + assert finance_report.training_categories == 233 + # training alias set == registry identity set (join stays compatible) + assert {e.category_id for e in finance.entries} == set(registry.ids) + # real hierarchy path depth preserved (dict compressed it) + depths = {len(e.path) for e in finance.entries} + assert 4 in depths and 3 in depths + + +def test_finance_five_level4_leaf_entries_kept_with_distinct_level3(built): + finance, _, _, _ = built + bucket = [ + e for e in finance.entries + if e.category_id == "finance:业务.合约协议.基本信息" + ] + assert len(bucket) == 5 # 合同通用/贷款业务/中间业务/资金业务/其他支付业务 + assert len({e.standard_entry_id for e in bucket}) == 5 + # every entry keeps its own source row (nothing collapsed) + assert len({e.source.row for e in bucket}) == 5 + + +def test_finance_unparseable_levels_reported_not_fixed(built): + _, finance_report, _, _ = built + unparseable = [i for i in finance_report.issues if i.kind == "level_unparseable"] + assert len(unparseable) == 2 + assert all("standard_data_level=null (not guessed)" in i.detail for i in unparseable) + + +def test_finance_alignment_reproduces_known_outliers(built): + finance, _, _, _ = built + records = load_canonical_records(CANON / "finance" / "all.json") + report = align_dataset_to_standard(records, finance) + counts = report["sample_counts"] + assert counts["total"] == 568 + assert counts["resolved"] == 531 + assert counts["matched"] == 529 + assert counts["mismatched"] == 2 + assert counts["standard_missing"] == 0 + fields = {m["field"] for m in report["mismatched_samples"]} + assert fields == {"AMONEY", "HXTRADENO"} + + +def test_finance_unresolved_evidence_lists_candidates(built): + finance, _, _, _ = built + records = load_canonical_records(CANON / "finance" / "all.json") + report = align_dataset_to_standard(records, finance) + assert report["sample_counts"]["unresolved"] == 37 + assert report["unresolved_by_status"] == {"missing_leaf": 34, "path_mismatch": 3} + # evidence-only mapping exists and never repairs + assert report["unresolved_evidence"] + for item in report["unresolved_evidence"]: + assert "candidate_standard_categories" in item + + +def test_shougang_standard_covers_registry_plus_lost_b3_6(built): + _, _, shougang, _ = built + registry = LeafRegistry.from_path(REG / "shougang.registry.json") + standard_codes = {e.standard_entry_id for e in shougang.entries} + assert len(shougang.entries) == 234 + assert set(registry.ids) <= standard_codes + assert standard_codes - set(registry.ids) == {"B3-6"} # 中厚板作业计划 lost by legacy + + +def test_shougang_alignment_100_percent(built): + _, _, shougang, _ = built + records = load_canonical_records(CANON / "shougang" / "all.json") + report = align_dataset_to_standard(records, shougang) + counts = report["sample_counts"] + assert counts["total"] == 19415 + assert counts["resolved"] == 18393 + assert counts["matched"] == 18393 + assert counts["mismatched"] == 0 + assert counts["standard_missing"] == 0 + + +def test_infra_reuses_shougang_standard(built): + _, _, shougang, _ = built + assert resolve_standard_dataset("infra") == "shougang" + records = load_canonical_records(CANON / "infra" / "all.json") + report = align_dataset_to_standard(records, shougang) + counts = report["sample_counts"] + assert counts["total"] == 64 + assert counts["matched"] == 64 + assert counts["mismatched"] == 0 + + +def test_pers_info_has_no_canonical_standard(): + assert resolve_standard_dataset("pers_info") is None + + +def test_build_deterministic_on_real_inputs(built): + _, _, _, _ = built + # rebuild from the same raw readers — identical fingerprints + f2, _ = build_finance_standard( + read_finance_standard_guide(FIN_XLSX).entries, + source_file=str(FIN_XLSX), source_sheet="Table 1", + ) + s2, _ = build_shougang_standard( + read_guanji_catalog(SHG_XLSX).entries, + source_file=str(SHG_XLSX), source_sheet="数据分类分级", + ) + assert len(f2.entries) == 237 + assert len(s2.entries) == 234 + + +@pytest.mark.parametrize( + "annotation_id,expected_range,expected_applies", + [ + ("finance-remark-55-55", None, 1), + ("finance-remark-93-132", "J93:J132", 40), + ("finance-remark-168-169", "J168:J169", 2), + ], +) +def test_finance_remark_scope_reproduced(built, annotation_id, expected_range, expected_applies): + finance, _, _, _ = built + by_id = {a.annotation_id: a for a in finance.scoped_annotations} + assert annotation_id in by_id + annotation = by_id[annotation_id] + assert annotation.merged_range == expected_range + assert len(annotation.applies_to_standard_entry_ids) == expected_applies + assert all(e.standard_entry_id in annotation.applies_to_standard_entry_ids + for e in finance.entries + if annotation.start_row <= e.source.row <= annotation.end_row) + + +def test_finance_row93_inherited_definitions_with_provenance(built): + finance, _, _, _ = built + entry = finance.by_entry_id()["finance:业务.金融监管和服务.反洗钱业务信息.分类考核评级信息"] + l2 = entry.raw_fields["level_2_definition"] + l3 = entry.raw_fields["level_3_definition"] + assert l2["source_cell"] == "D93" and l2["merged_range"] == "D93:D132" + assert l3["source_cell"] == "F93" + # remark is a scoped annotation, NOT a leaf raw field + assert "remark" not in entry.raw_fields + + +def test_shougang_inherited_definitions_and_resource_preserved(built): + _, _, shougang, _ = built + for code in ("A1-1-1", "B1-2"): + entry = shougang.by_entry_id()[code] + for key in ("level_1_definition", "level_2_definition", "level_3_definition", "resource"): + assert key in entry.raw_fields, (code, key) + assert entry.raw_fields["resource"]["value"] + assert entry.raw_fields["level_1_definition"]["source_cell"].startswith("C") + + +@pytest.mark.parametrize( + "xlsx,mins", + [ + (RAW / "部分金融数据.xlsx", 1), # one title merge (A1:A2) only + (RAW / "带分级分类的个人基础信息样本190条.xlsx", 0), + (RAW / "关基设施数据分类分级-不包含训练-用于测试(1).xlsx", 0), + ], +) +def test_sample_sources_have_no_business_merged_cells(xlsx, mins): + openpyxl = pytest.importorskip("openpyxl") + if not xlsx.is_file(): + pytest.skip("raw sample workbook not present") + workbook = openpyxl.load_workbook(xlsx) + try: + ws = workbook[workbook.sheetnames[0]] + # sample sources must not use group-level merges; the finance sample's + # single A1:A2 is a title, not a business column + assert len(ws.merged_cells.ranges) <= mins + finally: + workbook.close()