| 123456789101112131415161718192021222324252627282930313233343536373839404142434445464748495051525354555657585960616263646566676869707172737475767778798081828384858687888990919293949596979899100101102103104105106107108109110111112113114115116117118119120121122123124125126127128129130131132133134135136137138139140141142143144145146147148149150151152153154155156157158159160161162163164165166167168169170171172173174175176177178179180181182183184185186187188189190191192193194195196197198199200201202203204205206207208209210211212213214215216217218219220221222223224225226227228229230231232233234235236237238239240241242243244245246247248249250251252253254255256257258259260261262263264265266267268269270271272273274275276277278279280281282283284285286287288289290291292293294295296297298299300301302303304305306307308309310311312313314315316317318319320321322323324325326327328329330331332333334335336337338339340341342343344345346347348349350351352353354355356357358359360361362363364365366367368369370371372373374375376377378379380381382383384385386387388389390391392393394395396397398399400401402403404405406407408409410411412413414415416417418419420421422423424425426427428429430431432433434435436437438439440441442443444445446447448449450451452453454455456457458459460461462463464465466467468469470471472473474475476477478479480481482483484485486487488489490491492493494495496497498499500501502503504505506507508509510511512513514515516517518519520521522523524525526527528529530531532533534535536537538539540541542543544545546547548549550551552553554555556557558559560561562563564565566567568569570571572573574575576577578579580581582583584585586587588589590591592593594595596597598599600601602603604605606607608609610611612613614615616617618619620621622623624625626627628629630631632633634635636637638639640641642643644645646647648649650651652653654655656657658659660661662663664665666667668669670671672673674675676677678679680681682683684685686687688689690691692693694695696697698699700 |
- # -*- coding: utf-8 -*-
- """schemas.py — per 场 SOP artifact 链七 JSON schema 的单一权威源 (docs/SOP.md 附录B).
- 被 scripts/sop_check.py 与 (Batch2) 生成器共用; 改 schema 只改这里。
- 最小判据级 schema (非 jsonschema 库): 每 artifact 声明必填键 + 类型/枚举约束,
- validate(doc, name) 返回 issue 字符串列表 (空 = PASS)。
- """
- import re
- import json
- from fnmatch import fnmatch
- VERDICTS = {'定论', '准定论·预警', '候选', '参考', 'INSUFFICIENT', '撤回'}
- CLAIM_CLASSES = {'绝对性能', '相对排名', '趋势', '事件归因', '资源', '限电', '其他',
- '部件健康', '数据质量', '损失', '安全性'} # 后四类 2026-07-16 收编存量真实用法 (红队: 枚举定义了却无人校验)
- # artifact 名 → 顶层必填键 (dict 类 artifact)
- REQUIRED_KEYS = {
- 'survey.json': ['data_layers', 'window_matrix', 'n_machines_claimed', 'n_with_data',
- 'data_attribution', 'materials_read'],
- 'goem.json': ['fields'], # per 字段: 实证方法+结论+证据数字
- 'gate_record': ['section', 'verdict', 'assertions'], # gate_<section>.json
- 'clean_gate.json': ['clip_counts', 'sentinel_to_nan', 'dead_columns', 'rows_in_out'],
- 'mast_usability.json': ['shear_physical', 'direction_consistency', 'monthly_ratio', 'verdict'],
- 'findings_item': ['verdict', 'coverage', 'claim_class', 'basis', 'falsifiability'],
- 'review_item': ['reproduced', 'provenance'],
- 'input_fingerprint.json': ['farm', 'n_machines', 'window_min', 'window_max', 'layers'], # §9.1.1
- 'disk_manifest.json': ['source_roots', 'layers', 'by_ext'], # ON-1 对账门 onboard 端 (scripts/intake_scan.py)
- 'cleaned_fingerprint.json': ['section', 'parquet', 'sha256', 'rows', 'contract'], # §3 清洗快照留存(审计)
- }
- # §4 分析模块登记表 (configs/analysis_modules.yaml) — 每模块标准卡必填键 (单源, 改 schema 只改这里)
- MODULE_REQUIRED = ['id', 'name', 'domain', 'role', 'question', 'inputs', 'preconditions',
- 'method', 'threshold_basis', 'verdict_ceiling', 'falsifiability', 'impl', 'n', 'status']
- MODULE_DOMAINS = {'M0', 'M1', 'M2', 'M3', '横切'} # 机理域 (2026-07-08 D→M namespace; 交付九域=D1-D9, 审计门=AG0-AG9, 见 SOP §4 crosswalk)
- MODULE_ROLES = {'Gate', 'Diag', 'Integ', 'Act', 'Confirm'}
- MODULE_STATUS = {'稳定', '暂行', '待首跑', '条件性', 'N/A'}
- def norm_verdict(v):
- s = str(v).strip().strip('【】')
- # 判据函数筛查输出形态 ≡ 候选 (SOP §9.1.5 映射表 2026-07-16; 候选=内部流转禁入业主报告正文)
- return '候选' if s == 'CANDIDATE_RANKING' else s
- # 这些键 空 list/dict = 合法结果 (无死列 / 零 clip / 零哨兵 ≠ 漏算), 只查"在场"不要求非空
- ALLOW_EMPTY = {'dead_columns', 'clip_counts', 'sentinel_to_nan'}
- def validate_keys(doc, name):
- req = REQUIRED_KEYS.get(name, [])
- if not isinstance(doc, dict):
- return [f'{name}: 顶层非 dict']
- issues = []
- for k in req:
- if k in ALLOW_EMPTY:
- if k not in doc: # 空合法, 仅查键在场
- issues.append(f'{name}: 键缺: {k}')
- elif not doc.get(k):
- issues.append(f'{name}: 键缺/空: {k}')
- return issues
- def validate_module(rec, tag='module'):
- """单条分析模块卡校验: 必填键非空 + domain/role/status 枚举 + §0.3 准入 (n<3 须非稳定)."""
- if not isinstance(rec, dict):
- return [f'{tag}: 顶层非 dict']
- mid = rec.get('id', '?')
- issues = []
- for k in MODULE_REQUIRED:
- v = rec.get(k)
- if v is None or (isinstance(v, (str, list, dict)) and len(v) == 0):
- issues.append(f'module[{mid}]: 键缺/空: {k}')
- dom = rec.get('domain')
- if dom is not None and dom not in MODULE_DOMAINS:
- issues.append(f'module[{mid}]: domain 非法 {dom!r} (∈ {sorted(MODULE_DOMAINS)})')
- roles = rec.get('role') or []
- if isinstance(roles, list):
- bad = [r for r in roles if r not in MODULE_ROLES]
- if bad:
- issues.append(f'module[{mid}]: role 非法 {bad} (∈ {sorted(MODULE_ROLES)})')
- st = rec.get('status')
- if st is not None and st not in MODULE_STATUS:
- issues.append(f'module[{mid}]: status 非法 {st!r} (∈ {sorted(MODULE_STATUS)})')
- n = rec.get('n')
- if isinstance(n, int) and n < 3 and st == '稳定': # §0.3 准入闸: n<3 必标 [暂行]/条件性
- issues.append(f'module[{mid}]: n={n}<3 但 status=稳定 (§0.3 须标 暂行/条件性/待首跑)')
- return issues
- def validate_modules_doc(doc):
- """整份 analysis_modules.yaml: doc['modules'] 列表逐条 + id 唯一. 返回 issue 列表 (空=PASS)."""
- if not isinstance(doc, dict) or not isinstance(doc.get('modules'), list):
- return ['analysis_modules: 顶层须 dict 且含 modules 列表']
- issues, seen = [], set()
- for rec in doc['modules']:
- issues += validate_module(rec)
- mid = rec.get('id')
- if mid in seen:
- issues.append(f'module[{mid}]: id 重复')
- seen.add(mid)
- return issues
- def _name_tokens(name):
- """从文件/层名抽可匹配 token: 英数字 run(≥3) + CJK 连续子串(≥3). 鲁棒于缩写/空格/标点差异
- (disk 'FD116A-…-风力发电机组产品说明书' vs survey 'FD116A产品说明书' → 共享 '产品说明书'/'fd116a' 即认领)."""
- s = re.sub(r'\.[a-z0-9]{1,5}$', '', str(name).lower()) # 去扩展名: 防 'pdf'/'csv' 等通用 ext 假认领
- toks = set(re.findall(r'[a-z0-9]{3,}', s))
- for run in re.findall(r'[一-鿿]{3,}', s):
- run = run[:24]
- for L in range(3, len(run) + 1):
- for i in range(len(run) - L + 1):
- toks.add(run[i:i + L])
- return toks
- def reconcile_survey(dman, survey):
- """ON-1 对账 (docs/SOP.md §1.1): disk_manifest 枚举的每个数据层/资料/压缩包都必须在 survey.json
- 被认领 (claim-text 含其名 token), 否则 = 漏读 → FAIL. 唯一豁免 = survey.skipped:[{path,reason}] (reason 必填).
- 纯函数 (不碰文件系统; 枚举副作用在 scripts/intake_scan.py). 与 report 端 §9.1.1 指纹门对称:
- 一道证"进来的原始全枚举", 一道证"出去的清洗后可复算" → 两端闭环防漏读。
- """
- issues = []
- if not isinstance(dman, dict) or not isinstance(survey, dict):
- return ['reconcile: disk_manifest/survey 非 dict']
- skip_globs = []
- for s in (survey.get('skipped') or []):
- if not isinstance(s, dict) or not s.get('reason') or not (s.get('path') or s.get('glob')):
- issues.append(f'reconcile: skipped 条目缺 path/reason → 豁免必显式带理由 ({s})')
- else:
- skip_globs.append(s.get('path') or s.get('glob'))
- survey_wo_skip = {k: v for k, v in survey.items() if k != 'skipped'}
- claim = json.dumps(survey_wo_skip, ensure_ascii=False).lower()
- def basename(p):
- return str(p).replace('\\', '/').rstrip('/').rsplit('/', 1)[-1]
- def accounted(name, path):
- if any(t in claim for t in _name_tokens(name)):
- return True
- # 短名盲区补丁 (2026-07-14 古城 "6.zip"/海装 "总览.xls" 实证): stem<3字符/CJK<3 无 token 可抽
- # → 永 FAIL 无法认领。补精确文件名子串分支, 边界字符防 "6.zip" 误配 "16.zip"。
- bn = basename(name or path).lower()
- if bn and (f'/{bn}' in claim or f'"{bn}' in claim):
- return True
- return any(fnmatch(str(path), g) or fnmatch(str(name), g) or g in str(path) for g in skip_globs)
- for ly in dman.get('layers', []):
- pth = ly.get('path', '')
- nm = ly.get('name') or basename(pth)
- if not accounted(nm, pth):
- issues.append(f'reconcile: 数据层 "{pth}" ({ly.get("n_data_files")}文件) 未在 survey 认领 '
- f'→ 漏读 (§1.1①递归列全数据层); 在 survey 认领或加 skipped:{{path,reason}}')
- for mt in dman.get('materials', []):
- pth = mt.get('path', '')
- nm = mt.get('name') or basename(pth)
- if not accounted(nm, pth):
- issues.append(f'reconcile: 资料 "{pth}" 未在 survey.materials_read 认领 '
- f'→ 漏读 (§1.1②读全非数据资料); 在 survey 认领或加 skipped:{{path,reason}}')
- for ar in dman.get('archives', []):
- pth = ar.get('path', '')
- nm = ar.get('name') or basename(pth)
- if not accounted(nm, pth):
- issues.append(f'reconcile: 压缩包 "{pth}" 未在 survey 认领 → 漏读 (压缩易藏整层); '
- f'在 survey 认领或加 skipped:{{path,reason}}')
- return issues
- def validate_finding(it, tag='finding'):
- """AN-1 单条 finding 判据 (docs/SOP.md §4.0 + §0.2 降级表)."""
- issues = []
- v = norm_verdict(it.get('verdict', ''))
- if v not in VERDICTS:
- return [f'{tag}: verdict "{it.get("verdict")}" ∉ 六枚举 {sorted(VERDICTS)}']
- # 旧格式 findings (如白山/淄川/诺木洪 CMS 线) 把 coverage/basis 写成 str, 新 schema 的
- # 降级门读 cov.get(...)/basis.get(...) → 对旧数据 AttributeError 崩。checker 对旧数据崩
- # = checker 健壮性 bug, 不逼旧 findings 迁移 → 非 dict 时以空 dict 跑 (dict-key 门天然跳过),
- # 各留一条软提示防静默吞掉 (provenance: 不改别线 committed 结论, 只让 checker 不崩)。
- cov = it.get('coverage') or {}
- if not isinstance(cov, dict):
- issues.append(f'{tag}: coverage 为 {type(cov).__name__} 非 dict (旧格式) → n_machines/window_days 等无法机检, 建议迁 dict (软提示)')
- cov = {}
- for k in ('n_machines', 'window_days'):
- if not cov.get(k):
- issues.append(f'{tag}: coverage.{k} 缺/空')
- if v not in ('INSUFFICIENT', '撤回') and not cov.get('per_day_evidence'):
- issues.append(f'{tag}: 非INSUFFICIENT结论缺 per_day_evidence (单日快照骗3次, #18)')
- basis = it.get('basis') or {}
- if not isinstance(basis, dict):
- issues.append(f'{tag}: basis 为 {type(basis).__name__} 非 dict (旧格式) → 降级门(mast/curtail/window等)无法机检, 建议迁 dict (软提示)')
- basis = {}
- cc = str(it.get('claim_class', ''))
- # claim_class = §0.2 降级表的键; 自声明无校验 = 改标签即关掉整表 (2026-07-16 红队 live 复现后闸)
- if v not in ('INSUFFICIENT', '撤回'):
- if not cc:
- issues.append(f'{tag}: 缺 claim_class (§0.2 降级表交叉断言的键, 缺省=绕过降级)')
- elif cc not in CLAIM_CLASSES:
- issues.append(f'{tag}: claim_class "{cc}" ∉ 枚举 {sorted(CLAIM_CLASSES)} (自由文本=降级表逃逸)')
- # 2026-08-10: 补 '撤回' 豁免 — 本文件 line197 与 line204 两处同类规则均豁免 '撤回',
- # 唯此处漏, 导致**已撤回的 finding 仍被当生效结论校验**(如东 findings[22] 长期 FAIL)。
- # 撤回 = 该 claim 已作废, 不应再受内容规则约束; 三处规则的豁免集合就此对齐。
- if cc == '绝对性能' and str(basis.get('mast_usability', '')).upper() != 'PASS' and v not in ('INSUFFICIENT', '撤回'):
- issues.append(f'{tag}: 绝对性能∧mast非PASS → 必 INSUFFICIENT (实为 {v})')
- if cc in ('绝对性能', '损失', '相对排名') and not basis.get('curtail_strip') and v not in ('INSUFFICIENT', '撤回'):
- issues.append(f'{tag}: {cc}类缺 basis.curtail_strip(剥限电证据) → 限电=性能/损失头号混杂(¥620万教训); '
- f'补 {{method,n_removed}} 或显式 N/A (软警告, 审计 D4-4)')
- wm = basis.get('window_months')
- if cc == '资源' and isinstance(wm, (int, float)) and wm < 12 and v in ('定论', '准定论·预警', '候选'):
- issues.append(f'{tag}: 资源类单季窗({wm}月) → ≤参考 (实为 {v})')
- cp = it.get('cp_max', it.get('cp'))
- if isinstance(cp, (int, float)) and cp > 0.593 and v != 'INSUFFICIENT':
- issues.append(f'{tag}: Cp={cp}>0.593 Betz → INSUFFICIENT 物理铁证 (#32)')
- if not it.get('falsifiability'):
- issues.append(f'{tag}: 缺 falsifiability ("若X则假")')
- # 2026-07-16 Codex 对账三件: '其他'防错分类逃逸 + self_ref 降级行机器化 + [暂行] 跨场引用闸 (§0.3 待补兑现)
- if cc == '其他' and v not in ('INSUFFICIENT', '撤回') and not basis.get('claim_class_note'):
- issues.append(f'{tag}: claim_class=其他 须 basis.claim_class_note 注明为何不属既有类 (错分类=关降级表)')
- if basis.get('self_ref') and cc in ('相对排名', '趋势') and v in ('定论', '准定论·预警', '候选'):
- issues.append(f'{tag}: self_ref 风速上做{cc} → ≤参考 (§0.2 降级表, 实为 {v})')
- for i, ref in enumerate(it.get('cross_farm_refs') or []):
- if isinstance(ref, dict) and ref.get('tabu') and v != 'INSUFFICIENT':
- issues.append(f'{tag}: cross_farm_refs[{i}] 引 [暂行] 规则作 verdict 支撑 → 上限 INSUFFICIENT (§0.3 跨场引用闸, 实为 {v})')
- # §0.2 降级表机器覆盖补齐两行 (2026-07-17 红队 P1-6 附带; 字段自愿填=半机器化, 生产侧渐进接)
- if basis.get('reactive_fleet_relative') and not basis.get('q_setpoint_log') and v not in ('参考', 'INSUFFICIENT', '撤回'):
- issues.append(f'{tag}: 无功 fleet-relative 离群无园区 Q-setpoint 日志 → 归因 ≤【参考】 (§0.2 无功行, 实为 {v})')
- cwm = basis.get('calib_window_months')
- if isinstance(cwm, (int, float)) and cwm < 3 and not basis.get('calib_season_matched') and v in ('定论', '准定论·预警', '候选'):
- issues.append(f'{tag}: 变化型筛查标定窗 {cwm} 月 <1 季节周期且未同季匹配 → ≤【参考】 (§0.2 mset/NBM 行, 实为 {v})')
- # §4.6 温度域可选字段软校验 (2026-07-02 用户裁决清债; 缺省不报, 给了就须合法)
- avc = it.get('acute_vs_chronic')
- if avc is not None and avc not in ('急性', '慢性'):
- issues.append(f'{tag}: acute_vs_chronic "{avc}" ∉ {{急性,慢性}} (超限跳机码=急性, 无码持续偏离=慢性)')
- for bkey in ('oem_threshold_crossed', 'temp_clean_mech_not_cleared'):
- bv = it.get(bkey)
- if bv is not None and not isinstance(bv, bool):
- issues.append(f'{tag}: {bkey} 须 bool (OEM绝对阈交叉 / 温度干净≠机械已清→转CMS)')
- return issues
- def validate_review(it, tag='review'):
- """RV-2 单条评审判据 (docs/SOP.md §6.2)."""
- issues = []
- rep = it.get('reproduced')
- if not rep or not all(isinstance(r, dict) and r.get('claim') and r.get('cmd') for r in rep):
- issues.append(f'{tag}: reproduced 缺或条目缺 claim/cmd (subagent 数字必复现, #23/#30)')
- elif not all(r.get('output_digest') for r in rep):
- issues.append(f'{tag}: reproduced 条目缺 output_digest (无输出摘要=复现不可核对, §6.2; 2026-07-16 Codex 外审补)')
- prov = it.get('provenance')
- if not prov or not all(isinstance(r, dict) and r.get('src_doc') and r.get('origin') for r in prov):
- issues.append(f'{tag}: provenance 缺或条目缺 src_doc/origin (归属=audit单点故障, zyx)')
- elif not any(r.get('independence') or r.get('独立性') or r.get('独立性声明') for r in prov):
- issues.append(f'{tag}: provenance 无任何独立性声明字段 (independence/独立性; "多源印证"隐含独立性 claim 必显式, §6.2; 2026-07-16 Codex 外审补)')
- return issues
- # §7.1 逐台结论总表 (压轴章) 字段契约 (2026-06-20 ③ 解锁): finding 可选 per-机字段单源.
- PROBLEM_NATURE = {'性能', '安全性', '可靠性', '部件', '可靠性/部件', '电量', '风资源',
- '控制', '外部(电网)', '整体', '数据'}
- def validate_per_machine(findings):
- """③ 逐台字段可选契约 (在场则类型核): affected_machines(list[str]|str) / recommendation(str) / problem_nature(str∈枚举).
- 场级/健康 finding 合法缺省 (非每条都有问题机); 仅当显式提供时核类型 — 防 producer 写脏 (与 §9.1.2 同源: 字段也要结构化)。
- """
- issues = []
- for idx, it in enumerate(findings or []):
- if not isinstance(it, dict):
- continue
- am = it.get('affected_machines')
- if am is not None and not isinstance(am, (list, str)):
- issues.append(f'finding[{idx}]: affected_machines 须 list/str (实为 {type(am).__name__})')
- rec = it.get('recommendation')
- if rec is not None and not isinstance(rec, str):
- issues.append(f'finding[{idx}]: recommendation 须 str')
- pn = it.get('problem_nature')
- if pn is not None and pn not in PROBLEM_NATURE:
- issues.append(f'finding[{idx}]: problem_nature "{pn}" ∉ {sorted(PROBLEM_NATURE)}')
- return issues
- # _OV_MACHINE_RE + validate_owner_view_consistency (业主版散文↔结构化一致性闸, RV-2 a) 已删 — 取消业主版 (2026-06-20), owner_view 已无消费者。
- def validate_conservation(findings, n_total, window_days=None, slack_days=2):
- """9.1.4 守恒校验 (docs/SOP.md §9.1.4): 异常机数≤总机数 / 结论窗⊆数据窗 / (有分项)总=Σ分项.
- findings: findings.json 的 findings 列表; n_total: 指纹场站总机数; window_days: 数据窗跨度天.
- 防"算术不闭合"造假 — 与 §5 D8 窗对齐互补 (D8 防窗错配, 此条防数算不闭合).
- """
- issues = []
- for idx, it in enumerate(findings or []):
- if not isinstance(it, dict):
- continue
- tag = f'finding[{idx}]({str(it.get("title", ""))[:18]})'
- cov = it.get('coverage') or {}
- if not isinstance(cov, dict): # 2026-09-07 实逮: 旧条目 coverage 为字符串 → 此处 .get 抛异常, PB-1 9.1.4 整段 fail-closed 报"解析失败", 真问题反而看不见
- issues.append(f'{tag}: coverage 非结构化 (str), 守恒无法核 (9.1.4)')
- continue
- nm = cov.get('n_machines')
- if isinstance(nm, (int, float)) and isinstance(n_total, (int, float)) and nm > n_total:
- issues.append(f'{tag}: 异常机数 {nm} > 场站总机 {n_total} → 守恒不闭合 (9.1.4)')
- wd = cov.get('window_days')
- if (isinstance(wd, (int, float)) and isinstance(window_days, (int, float))
- and window_days > 0 and wd > window_days + slack_days):
- issues.append(f'{tag}: 结论窗 {wd}天 > 数据窗 {window_days}天 → 时间不⊆数据 (9.1.4)')
- comp, tot = it.get('components'), it.get('total')
- if isinstance(comp, list) and comp and isinstance(tot, (int, float)):
- s = sum(c.get('value', 0) for c in comp if isinstance(c, dict))
- if abs(s - tot) > max(1e-6, abs(tot) * 0.01):
- issues.append(f'{tag}: 分项和 {s} ≠ 总 {tot} (9.1.4 守恒, 容差1%)')
- wf = it.get('waterfall')
- if wf is not None:
- issues.extend(validate_waterfall(wf, f'{tag}.waterfall'))
- return issues
- def validate_waterfall(wf, tag='waterfall'):
- """9.1.4 损失瀑布守恒 (docs/SOP.md §4.1 模块1.2b + §9.1.4): 让能量/¥账长牙.
- Σ分项 == total(容差1%) / 各项≥0(物理) / 恰一 actual(实发)项 / 残差打包必声明 basis /
- 若声明 economic_total: 必 == Σ(损失项 value × tariff) (防手填¥, 须 电量×电价 可复算).
- """
- issues = []
- if not isinstance(wf, dict):
- return [f'{tag}: 非 dict']
- total, comps = wf.get('total'), wf.get('components')
- if not isinstance(total, (int, float)):
- issues.append(f'{tag}: 缺 total 数值')
- if not isinstance(comps, list) or not comps:
- return issues + [f'{tag}: 缺 components 非空列表']
- s, actual_n = 0.0, 0
- for c in comps:
- if not isinstance(c, dict) or not isinstance(c.get('value'), (int, float)):
- issues.append(f'{tag}: 分项 {c} 缺数值 value')
- continue
- v = c['value']
- if v < 0:
- issues.append(f'{tag}: 分项 "{c.get("name")}" {v}<0 (实发/损失不可负, 物理)')
- s += v
- actual_n += 1 if c.get('kind') == 'actual' else 0
- if c.get('kind') == 'residual' and not c.get('basis'):
- issues.append(f'{tag}: 残差项 "{c.get("name")}" 无 basis → 禁硬拆打包残差 (§4.1 1.2b)')
- if isinstance(total, (int, float)) and abs(s - total) > max(1e-6, abs(total) * 0.01):
- issues.append(f'{tag}: 分项和 {round(s, 3)} ≠ total {total} (差 {round(s - total, 3)}) → 守恒不闭合 (9.1.4)')
- if actual_n != 1:
- issues.append(f'{tag}: 须恰 1 个 kind=actual(实发)项, 实为 {actual_n} (瀑布=实发+Σ损失)')
- econ = wf.get('economic_total')
- if isinstance(econ, (int, float)):
- e, ok = 0.0, True
- for c in comps:
- if isinstance(c, dict) and c.get('kind') == 'loss':
- v, tar = c.get('value'), c.get('tariff')
- if isinstance(v, (int, float)) and isinstance(tar, (int, float)):
- e += v * tar
- else:
- ok = False
- if not ok:
- issues.append(f'{tag}: 声明 economic_total 但损失项缺 tariff → ¥不可复算 (须电量×电价, 禁手填)')
- elif abs(e - econ) > max(1e-6, abs(econ) * 0.01):
- issues.append(f'{tag}: economic_total {econ} ≠ Σ(损失×电价) {round(e, 2)} → ¥手填嫌疑 (9.1.4 ¥派生闭合)')
- return issues
- def validate_repro(findings, require_for=('定论',)):
- """9.1.5 可复现脚本声明 (docs/SOP.md §9.1.5): verdict∈require_for 的 finding 必带 repro:{script, expected}.
- 双模型评审铁律: 仅"脚本存在"= theater (可 echo 假值绕过), 故 expected 期望值必填 (供 verify-repro 重跑比对).
- 本函数只校验声明完整性 (纯函数); 脚本实存 + 重跑数值比对在 report_integrity.verify_repro (有副作用, 不在此).
- """
- issues = []
- for idx, it in enumerate(findings or []):
- if not isinstance(it, dict):
- continue
- if norm_verdict(it.get('verdict', '')) not in require_for:
- continue
- tag = f'finding[{idx}]({str(it.get("title", ""))[:18]})'
- repro = it.get('repro')
- if not isinstance(repro, dict) or not repro.get('script'):
- issues.append(f'{tag}: verdict=定论 但缺 repro.script (9.1.5 定论必附可复现脚本)')
- continue
- if repro.get('expected') is None:
- issues.append(f'{tag}: repro 缺 expected 期望值 (9.1.5 防空壳脚本: 仅脚本存在可 echo 假值绕过, 须可重跑比对)')
- return issues
- # §0.8 ON-0 分析前置锁定门: analysis_lock.yaml 自洽闸 (机器位 scripts/analysis_lock_check.py)
- LOCK_REQUIRED = ['farm', 'frozen', 'windows', 'rated_kw', 'turbines', 'capacity_mw',
- 'sources', 'metrics', 'cp_ceiling', 'gates']
- _LOCK_TODO = {'todo', '待填', '待定', 'tbd', '?', ''}
- def validate_required_outputs(doc, produced):
- """§0.8 required_outputs 完整性自检 — 必算清单缺一报错 (解 "分析项时有时无" = V-019 数据层翻版).
- lock.required_outputs = 本次分析必产出项清单 (声明式, 写死在锁里)。produced = 实际产出项
- (list/set/dict, 来自运行 manifest 或产物目录扫描的文件名)。清单上每一项必在 produced 出现 (精确或子串命中),
- 缺一 → issue, **不允许静默跳过**。这把 V-019 "声称做了实际没做" 的防造假机制延伸到分析项层:
- 不是事后发现少了, 是清单强制它不能少。纯函数, 返回 issue 列表 (空=PASS)。
- """
- if not isinstance(doc, dict):
- return ['analysis_lock: 顶层非 dict']
- req = doc.get('required_outputs')
- if not req:
- return ['analysis_lock: 缺 required_outputs (必算清单) → 无法做完整性自检, '
- '"分析项时有时无" 无机器拦截 (§0.8/V-019 数据层)']
- if not isinstance(req, (list, tuple)):
- return [f'analysis_lock: required_outputs 须为列表, 实为 {type(req).__name__}']
- if isinstance(produced, dict):
- produced = list(produced.keys())
- prod = {str(p).strip() for p in (produced or [])}
- issues = []
- for item in req:
- key = (item.get('id') if isinstance(item, dict) else str(item)).strip()
- if not key:
- continue
- if not (key in prod or any(key in p for p in prod)):
- issues.append(f'analysis_lock·required_outputs: 必算项 "{key}" 未产出 '
- f'(缺一报错, 禁静默跳过; §0.8/V-019 数据层)')
- return issues
- def validate_analysis_lock(doc, cap_tol=0.005):
- """§0.8 ON-0 分析前置锁定门 — analysis_lock.yaml 自洽闸 (解 "同数据两次结果不一").
- 根因实证: 利用小时 1330 vs 2389 / CF 27% vs 15% / 改造前后 +0.3 vs +23.6 —
- 窗/口径/源/容量/基准未锁 → 同数据两次结果不一。本函数机器化 §0.8 B 自洽闸:
- A 冻结六项+frozen+gates 在场 (未冻结=不得出数)。
- ① Σ台=场: Σ rated_kw[turbines[t]] == capacity_mw.全场 (容差) — catches 27.5 vs 48.92 类装机口径错。
- ② 口径完整: metric 标 须标口径:true 必给区分口径值/注 (L4 含状态过滤+风速来源)。
- ③ INSUFFICIENT 一致: metrics/cp_ceiling 标 状态:INSUFFICIENT 的量必在 gates.INSUFFICIENT标记 列出 (防漂移)。
- ④ Betz: cp_ceiling 数值 Cp ≥0.593 = 物理违背 (#32)。
- 纯函数, 返回 issue 列表 (空=PASS)。report↔lock 口径交叉在 analysis_lock_check.py (有副作用, 不在此)。
- """
- if not isinstance(doc, dict):
- return ['analysis_lock: 顶层非 dict']
- issues = []
- for k in LOCK_REQUIRED:
- v = doc.get(k)
- if k not in doc or v is None or (isinstance(v, (str, list, dict)) and len(v) == 0):
- issues.append(f'analysis_lock: 冻结项缺/空: {k} (§0.8 A 六项+frozen+gates)')
- elif isinstance(v, str) and v.strip().lower() in _LOCK_TODO:
- issues.append(f'analysis_lock: {k}="{v}" 占位未填 → 未冻结不得出数 (§0.5/§0.8)')
- # ★ freeze 须用户裁决 (§0.8 freeze 权威=用户非Claude自冻; 解"没找我确认"洞, 2026-06-30 加)
- # 防: Claude 自拟+自冻+盖哈希=把临场决定伪装成有权威的契约。frozen 有值则 frozen_by 必含"用户裁决"。
- frz = doc.get('frozen')
- if frz is not None and not (isinstance(frz, str) and frz.strip().lower() in _LOCK_TODO):
- fb = str(doc.get('frozen_by') or '')
- if '用户裁决' not in fb:
- issues.append('analysis_lock: frozen 已设但 frozen_by 非"用户裁决" → 自冻锁(Claude/非用户裁决)'
- '不得当冻结锁出数 (§0.8 freeze 权威=用户; Claude自冻+哈希=临场决定伪装契约)')
- # ① Σ台=场 (§0.8 B①) — 装机口径错 → CF/利用小时全错
- rated, turb, cap = doc.get('rated_kw'), doc.get('turbines'), doc.get('capacity_mw')
- if isinstance(rated, dict) and isinstance(turb, dict) and isinstance(cap, dict):
- unknown = sorted({str(m) for m in turb.values() if m not in rated})
- if unknown:
- issues.append(f'analysis_lock①Σ台=场: 机型 {unknown} 在 turbines 用到但 rated_kw 未定义')
- elif isinstance(cap.get('全场'), (int, float)):
- sum_mw = sum(float(rated[m]) for m in turb.values()) / 1000.0
- tot = float(cap['全场'])
- if abs(sum_mw - tot) > max(cap_tol, tot * cap_tol):
- issues.append(f'analysis_lock①Σ台=场不闭合: Σ台 {sum_mw:.3f}MW ≠ capacity_mw.全场 {tot}MW '
- f'(差 {sum_mw - tot:+.3f}; §0.8 B① 装机口径错→CF/利用小时全错)')
- # ② 口径完整 (§0.8 L4)
- metrics = doc.get('metrics') if isinstance(doc.get('metrics'), dict) else {}
- for mn, mv in metrics.items():
- if isinstance(mv, dict) and mv.get('须标口径') is True:
- extra = [x for x in mv if x not in ('须标口径', '定义', '注', '值')]
- if len(extra) < 2 and not mv.get('注'):
- issues.append(f'analysis_lock②口径: "{mn}" 标须标口径 却未给区分口径值/注 (L4 须含状态过滤+风速来源)')
- # ③ INSUFFICIENT 一致 (防 一处诚实留口、另一处被当定论引用)
- gates = doc.get('gates') if isinstance(doc.get('gates'), dict) else {}
- declared = json.dumps(gates.get('INSUFFICIENT标记') or [], ensure_ascii=False)
- cpc = doc.get('cp_ceiling') if isinstance(doc.get('cp_ceiling'), dict) else {}
- for src_name, src in (('metrics', metrics), ('cp_ceiling', cpc)):
- for mn, mv in (src.items() if isinstance(src, dict) else []):
- if isinstance(mv, dict) and str(mv.get('状态', '')).upper() == 'INSUFFICIENT' and str(mn) not in declared:
- issues.append(f'analysis_lock③: {src_name}."{mn}" 标 状态:INSUFFICIENT 但未在 gates.INSUFFICIENT标记 列出 → 口径漂移风险')
- # ④ Betz
- for k, v in (cpc.items() if isinstance(cpc, dict) else []):
- if isinstance(v, (int, float)) and 'cp' in str(k).lower() and v >= 0.593:
- issues.append(f'analysis_lock④Betz: cp_ceiling.{k}={v} ≥0.593 物理不可能 (#32 Cp>Betz 铁证)')
- return issues
- # ============================================================================
- # 附录A Case 状态机 (docs/运维Agent转化评估_v0.1.md 附录A → 代码化; 运维 agent 柱1「智能工单」)
- # ----------------------------------------------------------------------------
- # findings.json = 每批全量快照(无记性, B16 前兆每月被"重新发现"); case = 跨批次状态延续 +
- # 反向闭环(evidence→ticket→verify→CLOSED/RELAPSED→回灌 记分卡/基准库/误报台账)。
- # 本节把附录A 10 态状态机 + 6 条机器不变量落成纯函数闸(风格同上: 返 issue 列表, 空=PASS)。
- # 工单卡(TICKETED 载体)/回验卡(VERIFY 期)= case 生命周期的两个切面, 各自 schema 在此定义。
- # 正交两轴不进 state (附录A.2): tier(T1/T2/T3 紧迫度) / verdict(§0.2 证据强度) = case 属性。
- # ============================================================================
- CASE_STATES = {'NORMAL', 'WATCH', 'CANDIDATE', 'CONFIRMED', 'TICKETED',
- 'IN_REPAIR', 'VERIFY', 'CLOSED', 'REFUTED', 'RELAPSED'}
- # 合法状态迁移边 (附录A.2; 非法跳转拒绝 = 不变量⑤)
- CASE_EDGES = {
- 'NORMAL': {'WATCH'},
- 'WATCH': {'CANDIDATE', 'NORMAL'},
- 'CANDIDATE': {'CONFIRMED', 'NORMAL', 'REFUTED'},
- 'CONFIRMED': {'TICKETED', 'REFUTED'},
- 'TICKETED': {'IN_REPAIR', 'VERIFY'},
- 'IN_REPAIR': {'VERIFY'},
- 'VERIFY': {'CLOSED', 'CONFIRMED'}, # verify_fail → 回 CONFIRMED (未根治, 变桨 SOP A014 反弹式)
- 'CLOSED': {'RELAPSED'},
- 'REFUTED': set(), # 终态 (留档防再登顶)
- 'RELAPSED': {'CONFIRMED'},
- }
- # 确诊过之后 own-data 证据须留存 (不变量①: 禁 fleet-z 直升终判)
- CASE_CONFIRMED_OR_BEYOND = {'CONFIRMED', 'TICKETED', 'IN_REPAIR', 'VERIFY', 'CLOSED', 'RELAPSED'}
- # 工单已签发之后 (不变量②: 人工裁决点①机器不可绕)
- CASE_TICKETED_OR_BEYOND = {'TICKETED', 'IN_REPAIR', 'VERIFY', 'CLOSED', 'RELAPSED'}
- CASE_TERMINAL = {'CLOSED', 'REFUTED'} # 非终态判据 (附录A.4⑥ 至多一个活跃档)
- CASE_REQUIRED = ['case_id', 'farm', 'turbine', 'system', 'state', 'opened_at', 'updated_at']
- CASE_TIERS = {None, 'T1', 'T2', 'T3'}
- def validate_ticket_card(t, tag='ticket'):
- """工单卡 schema (附录A.3 ticket + P4 七段结构化): TICKETED 载体.
- 必: ticket_id / issued_by / actions(非空, = P4 §3 排查步骤). 可选: window_suggested(B1 排程建议) /
- parts / priority(P0/P1/P2). 无动作的工单 = 糊涂账 (P4 硬纪律)。"""
- if not isinstance(t, dict):
- return [f'{tag}: 非 dict']
- issues = []
- for k in ('ticket_id', 'issued_by'):
- if not t.get(k):
- issues.append(f'{tag}: 缺 {k}')
- acts = t.get('actions')
- if not isinstance(acts, list) or not acts:
- issues.append(f'{tag}: actions 须非空列表 (P4 §3 排查步骤; 无动作的工单=糊涂账)')
- pr = t.get('priority')
- if pr is not None and pr not in ('P0', 'P1', 'P2'):
- issues.append(f'{tag}: priority "{pr}" ∉ {{P0,P1,P2}} (P4 优先级)')
- parts = t.get('parts')
- if parts is not None and not isinstance(parts, list):
- issues.append(f'{tag}: parts 须 list')
- return issues
- def validate_verify_card(vf, tag='verify'):
- """回验卡 schema (附录A.3 verify): VERIFY 期量化闭环 = 反向闭环的一半价值 (P4 §7).
- 必: baseline(修前判据量) / target(回归带) / result∈{pass,fail,pending}.
- result=pass → readings 须非空 (无实测读数不得判 pass; 防"修完就是修好"糊涂账)。"""
- if not isinstance(vf, dict):
- return [f'{tag}: 非 dict']
- issues = []
- for k in ('baseline', 'target'):
- if vf.get(k) is None:
- issues.append(f'{tag}: 缺 {k} (修前基线/回归带; 无基线无法判回归)')
- res = vf.get('result')
- if res not in ('pass', 'fail', 'pending'):
- issues.append(f'{tag}: result "{res}" ∉ {{pass,fail,pending}}')
- rd = vf.get('readings')
- if res == 'pass' and (not isinstance(rd, list) or not rd):
- issues.append(f'{tag}: result=pass 但 readings 空 → 无实测读数不得判回归 (P4 §7 回验量化)')
- wd = vf.get('window_days')
- if wd is not None and not isinstance(wd, (int, float)):
- issues.append(f'{tag}: window_days 须数值')
- return issues
- def validate_analysis(an, tag='analysis'):
- """深挖层 schema (Phase-2 机制归因 + 独立审): case.analysis 可选扩展。
- 必: verdict(六枚举) + mechanism(机制, 无机制=空 verdict). 可选: review/field_needs/ref/date/by.
- ★纪律: analysis.verdict 是更深表征(可 > case.verdict, 如 D29 案候选→深挖准定论·预警),
- 但**不改 case.state** — 升 CONFIRMED 须 own-data(不变量①), 走 apply_human_action。"""
- if not isinstance(an, dict):
- return [f'{tag}: 非 dict']
- issues = []
- if norm_verdict(an.get('verdict', '')) not in VERDICTS:
- issues.append(f'{tag}: verdict "{an.get("verdict")}" ∉ 六枚举')
- if not an.get('mechanism'):
- issues.append(f'{tag}: 缺 mechanism (深挖必给机制归因, 否则=空 verdict)')
- return issues
- def validate_case(case, tag='case'):
- """附录A Case 状态机单档校验 (纯函数, 空=PASS). 6 条机器不变量 = 附录A.4.
- 注: 不变量⑤"追加式禁改写"的 append-only 语义无法从单份快照证 (需跨版本 diff, 在有副作用的
- 回灌器里查); 本函数证其可证部分 = 迁移边合法 + state↔审计链一致 + 结构完整。"""
- if not isinstance(case, dict):
- return [f'{tag}: 顶层非 dict']
- cid = case.get('case_id', '?')
- tg = f'{tag}[{cid}]'
- issues = []
- # 字段在场 (附录A.3)
- for k in CASE_REQUIRED:
- if not case.get(k):
- issues.append(f'{tg}: 缺 {k}')
- state = case.get('state')
- if state is not None and state not in CASE_STATES:
- issues.append(f'{tg}: state "{state}" ∉ 十态 {sorted(CASE_STATES)}')
- return issues # state 非法, 后续不变量无从判
- # 正交轴类型 (附录A.2: 紧迫度/证据强度 ≠ 状态)
- if case.get('tier') not in CASE_TIERS:
- issues.append(f'{tg}: tier "{case.get("tier")}" ∉ {{None,T1,T2,T3}} (紧迫度≠状态, 正交轴)')
- v = case.get('verdict')
- if v is not None and norm_verdict(v) not in VERDICTS:
- issues.append(f'{tg}: verdict "{v}" ∉ 六枚举 (证据强度≠状态, 正交轴)')
- # evidence 追加流水 (附录A.3): discriminator_id 外键 + verdict 合法
- ev = case.get('evidence') or []
- if not isinstance(ev, list):
- issues.append(f'{tg}: evidence 须 list')
- ev = []
- for j, e in enumerate(ev):
- if not isinstance(e, dict):
- issues.append(f'{tg}: evidence[{j}] 非 dict')
- continue
- if not e.get('discriminator_id'):
- issues.append(f'{tg}: evidence[{j}] 缺 discriminator_id (注册表外键, 判据溯源)')
- ev_v = e.get('verdict')
- if ev_v is not None and norm_verdict(ev_v) not in VERDICTS:
- issues.append(f'{tg}: evidence[{j}] verdict "{ev_v}" ∉ 六枚举')
- # history append-only 审计链: 每条迁移边合法 (不变量⑤可证部分) + by 枚举
- hist = case.get('history') or []
- if not isinstance(hist, list):
- issues.append(f'{tg}: history 须 list')
- hist = []
- for j, h in enumerate(hist):
- if not isinstance(h, dict):
- issues.append(f'{tg}: history[{j}] 非 dict')
- continue
- fr, to, by = h.get('from_state'), h.get('to_state'), h.get('by')
- if by not in ('engine', 'human'):
- issues.append(f'{tg}: history[{j}] by "{by}" ∉ {{engine,human}}')
- if fr not in CASE_STATES or to not in CASE_STATES:
- issues.append(f'{tg}: history[{j}] from/to 非十态 ({fr}→{to})')
- elif to not in CASE_EDGES.get(fr, set()):
- issues.append(f'{tg}: history[{j}] 非法迁移 {fr}→{to} (附录A.2 未定义边; 不变量⑤)')
- if hist and isinstance(hist[-1], dict) and hist[-1].get('to_state') and state:
- if hist[-1]['to_state'] != state:
- issues.append(f'{tg}: state={state} ≠ history 末态 {hist[-1]["to_state"]} (审计链不一致)')
- # ── 6 条机器不变量 (附录A.4) ──
- # ① CONFIRMED 及以后 → evidence 含 ≥1 own_data 级 (禁 fleet-z 直升终判, 验证金字塔③)
- if state in CASE_CONFIRMED_OR_BEYOND and not any(
- isinstance(e, dict) and e.get('own_data') for e in ev):
- issues.append(f'{tg}①: state={state} 但 evidence 无 own_data 级判据 → 禁 fleet-z 直升 (金字塔③终判)')
- # ② TICKETED 及以后 → history 含 by=human 的 ticket_issued + 工单卡在场合法 (人工裁决点①机器不可绕)
- if state in CASE_TICKETED_OR_BEYOND:
- if not any(isinstance(h, dict) and h.get('by') == 'human'
- and h.get('event') == 'ticket_issued' for h in hist):
- issues.append(f'{tg}②: state={state} 但 history 无 by=human 的 ticket_issued (人工裁决点①机器不可绕)')
- issues += validate_ticket_card(case.get('ticket') or {}, f'{tg}.ticket')
- # ③ CLOSED → verify.result=pass (无回验不销案; 昌邑 A27 悬案按构造杜绝)
- if state == 'CLOSED':
- vfr = (case.get('verify') or {}).get('result')
- if vfr != 'pass':
- issues.append(f'{tg}③: state=CLOSED 但 verify.result≠pass ("{vfr}") → 无回验不销案 (附录A.4③)')
- if state in ('VERIFY', 'CLOSED'):
- issues += validate_verify_card(case.get('verify') or {}, f'{tg}.verify')
- # ④ REFUTED → refuted_basis 非空 (防同信号再登顶)
- if state == 'REFUTED' and not case.get('refuted_basis'):
- issues.append(f'{tg}④: state=REFUTED 但 refuted_basis 空 → 防同信号再登顶 (附录A.4④)')
- # 深挖层 (可选扩展): case.analysis 在场则校验 schema (Phase-2 机制 + 独立审)
- if case.get('analysis') is not None:
- issues += validate_analysis(case['analysis'], f'{tg}.analysis')
- # ⑥ RELAPSED → relapse_of 指原 case (D7 复发率数据源; 跨档"至多一活跃"在 validate_cases)
- if state == 'RELAPSED' and not case.get('relapse_of'):
- issues.append(f'{tg}⑥: state=RELAPSED 但缺 relapse_of (指原 case; D7 复发率数据源)')
- return issues
- def validate_cases(cases, tag='cases'):
- """整份 cases.json 跨档校验 (附录A.4⑥): 逐档 validate_case + case_id 唯一
- + 同 (farm,turbine,system) 至多一个非终态 case (防重复开案)."""
- if not isinstance(cases, list):
- return [f'{tag}: 顶层须 list']
- issues, seen_id, active = [], set(), {}
- for c in cases:
- issues += validate_case(c)
- if not isinstance(c, dict):
- continue
- cid = c.get('case_id')
- if cid in seen_id:
- issues.append(f'{tag}: case_id 重复 {cid}')
- seen_id.add(cid)
- if c.get('state') not in CASE_TERMINAL:
- key = (c.get('farm'), c.get('turbine'), c.get('system'))
- if key in active:
- issues.append(f'{tag}: {key} 同时 >1 非终态 case ({active[key]} + {cid}) → 防重复开案 (附录A.4⑥)')
- active[key] = cid
- return issues
|