# -*- coding: utf-8 -*- """schemas.py — per 场 SOP artifact 链七 JSON schema 的单一权威源 (docs/SOP.md 附录B). 被 scripts/sop_check.py 与 (Batch2) 生成器共用; 改 schema 只改这里。 最小判据级 schema (非 jsonschema 库): 每 artifact 声明必填键 + 类型/枚举约束, validate(doc, name) 返回 issue 字符串列表 (空 = PASS)。 """ import re import json from fnmatch import fnmatch VERDICTS = {'定论', '准定论·预警', '候选', '参考', 'INSUFFICIENT', '撤回'} CLAIM_CLASSES = {'绝对性能', '相对排名', '趋势', '事件归因', '资源', '限电', '其他', '部件健康', '数据质量', '损失', '安全性'} # 后四类 2026-07-16 收编存量真实用法 (红队: 枚举定义了却无人校验) # artifact 名 → 顶层必填键 (dict 类 artifact) REQUIRED_KEYS = { 'survey.json': ['data_layers', 'window_matrix', 'n_machines_claimed', 'n_with_data', 'data_attribution', 'materials_read'], 'goem.json': ['fields'], # per 字段: 实证方法+结论+证据数字 'gate_record': ['section', 'verdict', 'assertions'], # gate_
.json 'clean_gate.json': ['clip_counts', 'sentinel_to_nan', 'dead_columns', 'rows_in_out'], 'mast_usability.json': ['shear_physical', 'direction_consistency', 'monthly_ratio', 'verdict'], 'findings_item': ['verdict', 'coverage', 'claim_class', 'basis', 'falsifiability'], 'review_item': ['reproduced', 'provenance'], 'input_fingerprint.json': ['farm', 'n_machines', 'window_min', 'window_max', 'layers'], # §9.1.1 'disk_manifest.json': ['source_roots', 'layers', 'by_ext'], # ON-1 对账门 onboard 端 (scripts/intake_scan.py) 'cleaned_fingerprint.json': ['section', 'parquet', 'sha256', 'rows', 'contract'], # §3 清洗快照留存(审计) } # §4 分析模块登记表 (configs/analysis_modules.yaml) — 每模块标准卡必填键 (单源, 改 schema 只改这里) MODULE_REQUIRED = ['id', 'name', 'domain', 'role', 'question', 'inputs', 'preconditions', 'method', 'threshold_basis', 'verdict_ceiling', 'falsifiability', 'impl', 'n', 'status'] MODULE_DOMAINS = {'M0', 'M1', 'M2', 'M3', '横切'} # 机理域 (2026-07-08 D→M namespace; 交付九域=D1-D9, 审计门=AG0-AG9, 见 SOP §4 crosswalk) MODULE_ROLES = {'Gate', 'Diag', 'Integ', 'Act', 'Confirm'} MODULE_STATUS = {'稳定', '暂行', '待首跑', '条件性', 'N/A'} def norm_verdict(v): s = str(v).strip().strip('【】') # 判据函数筛查输出形态 ≡ 候选 (SOP §9.1.5 映射表 2026-07-16; 候选=内部流转禁入业主报告正文) return '候选' if s == 'CANDIDATE_RANKING' else s # 这些键 空 list/dict = 合法结果 (无死列 / 零 clip / 零哨兵 ≠ 漏算), 只查"在场"不要求非空 ALLOW_EMPTY = {'dead_columns', 'clip_counts', 'sentinel_to_nan'} def validate_keys(doc, name): req = REQUIRED_KEYS.get(name, []) if not isinstance(doc, dict): return [f'{name}: 顶层非 dict'] issues = [] for k in req: if k in ALLOW_EMPTY: if k not in doc: # 空合法, 仅查键在场 issues.append(f'{name}: 键缺: {k}') elif not doc.get(k): issues.append(f'{name}: 键缺/空: {k}') return issues def validate_module(rec, tag='module'): """单条分析模块卡校验: 必填键非空 + domain/role/status 枚举 + §0.3 准入 (n<3 须非稳定).""" if not isinstance(rec, dict): return [f'{tag}: 顶层非 dict'] mid = rec.get('id', '?') issues = [] for k in MODULE_REQUIRED: v = rec.get(k) if v is None or (isinstance(v, (str, list, dict)) and len(v) == 0): issues.append(f'module[{mid}]: 键缺/空: {k}') dom = rec.get('domain') if dom is not None and dom not in MODULE_DOMAINS: issues.append(f'module[{mid}]: domain 非法 {dom!r} (∈ {sorted(MODULE_DOMAINS)})') roles = rec.get('role') or [] if isinstance(roles, list): bad = [r for r in roles if r not in MODULE_ROLES] if bad: issues.append(f'module[{mid}]: role 非法 {bad} (∈ {sorted(MODULE_ROLES)})') st = rec.get('status') if st is not None and st not in MODULE_STATUS: issues.append(f'module[{mid}]: status 非法 {st!r} (∈ {sorted(MODULE_STATUS)})') n = rec.get('n') if isinstance(n, int) and n < 3 and st == '稳定': # §0.3 准入闸: n<3 必标 [暂行]/条件性 issues.append(f'module[{mid}]: n={n}<3 但 status=稳定 (§0.3 须标 暂行/条件性/待首跑)') return issues def validate_modules_doc(doc): """整份 analysis_modules.yaml: doc['modules'] 列表逐条 + id 唯一. 返回 issue 列表 (空=PASS).""" if not isinstance(doc, dict) or not isinstance(doc.get('modules'), list): return ['analysis_modules: 顶层须 dict 且含 modules 列表'] issues, seen = [], set() for rec in doc['modules']: issues += validate_module(rec) mid = rec.get('id') if mid in seen: issues.append(f'module[{mid}]: id 重复') seen.add(mid) return issues def _name_tokens(name): """从文件/层名抽可匹配 token: 英数字 run(≥3) + CJK 连续子串(≥3). 鲁棒于缩写/空格/标点差异 (disk 'FD116A-…-风力发电机组产品说明书' vs survey 'FD116A产品说明书' → 共享 '产品说明书'/'fd116a' 即认领).""" s = re.sub(r'\.[a-z0-9]{1,5}$', '', str(name).lower()) # 去扩展名: 防 'pdf'/'csv' 等通用 ext 假认领 toks = set(re.findall(r'[a-z0-9]{3,}', s)) for run in re.findall(r'[一-鿿]{3,}', s): run = run[:24] for L in range(3, len(run) + 1): for i in range(len(run) - L + 1): toks.add(run[i:i + L]) return toks def reconcile_survey(dman, survey): """ON-1 对账 (docs/SOP.md §1.1): disk_manifest 枚举的每个数据层/资料/压缩包都必须在 survey.json 被认领 (claim-text 含其名 token), 否则 = 漏读 → FAIL. 唯一豁免 = survey.skipped:[{path,reason}] (reason 必填). 纯函数 (不碰文件系统; 枚举副作用在 scripts/intake_scan.py). 与 report 端 §9.1.1 指纹门对称: 一道证"进来的原始全枚举", 一道证"出去的清洗后可复算" → 两端闭环防漏读。 """ issues = [] if not isinstance(dman, dict) or not isinstance(survey, dict): return ['reconcile: disk_manifest/survey 非 dict'] skip_globs = [] for s in (survey.get('skipped') or []): if not isinstance(s, dict) or not s.get('reason') or not (s.get('path') or s.get('glob')): issues.append(f'reconcile: skipped 条目缺 path/reason → 豁免必显式带理由 ({s})') else: skip_globs.append(s.get('path') or s.get('glob')) survey_wo_skip = {k: v for k, v in survey.items() if k != 'skipped'} claim = json.dumps(survey_wo_skip, ensure_ascii=False).lower() def basename(p): return str(p).replace('\\', '/').rstrip('/').rsplit('/', 1)[-1] def accounted(name, path): if any(t in claim for t in _name_tokens(name)): return True # 短名盲区补丁 (2026-07-14 古城 "6.zip"/海装 "总览.xls" 实证): stem<3字符/CJK<3 无 token 可抽 # → 永 FAIL 无法认领。补精确文件名子串分支, 边界字符防 "6.zip" 误配 "16.zip"。 bn = basename(name or path).lower() if bn and (f'/{bn}' in claim or f'"{bn}' in claim): return True return any(fnmatch(str(path), g) or fnmatch(str(name), g) or g in str(path) for g in skip_globs) for ly in dman.get('layers', []): pth = ly.get('path', '') nm = ly.get('name') or basename(pth) if not accounted(nm, pth): issues.append(f'reconcile: 数据层 "{pth}" ({ly.get("n_data_files")}文件) 未在 survey 认领 ' f'→ 漏读 (§1.1①递归列全数据层); 在 survey 认领或加 skipped:{{path,reason}}') for mt in dman.get('materials', []): pth = mt.get('path', '') nm = mt.get('name') or basename(pth) if not accounted(nm, pth): issues.append(f'reconcile: 资料 "{pth}" 未在 survey.materials_read 认领 ' f'→ 漏读 (§1.1②读全非数据资料); 在 survey 认领或加 skipped:{{path,reason}}') for ar in dman.get('archives', []): pth = ar.get('path', '') nm = ar.get('name') or basename(pth) if not accounted(nm, pth): issues.append(f'reconcile: 压缩包 "{pth}" 未在 survey 认领 → 漏读 (压缩易藏整层); ' f'在 survey 认领或加 skipped:{{path,reason}}') return issues def validate_finding(it, tag='finding'): """AN-1 单条 finding 判据 (docs/SOP.md §4.0 + §0.2 降级表).""" issues = [] v = norm_verdict(it.get('verdict', '')) if v not in VERDICTS: return [f'{tag}: verdict "{it.get("verdict")}" ∉ 六枚举 {sorted(VERDICTS)}'] # 旧格式 findings (如白山/淄川/诺木洪 CMS 线) 把 coverage/basis 写成 str, 新 schema 的 # 降级门读 cov.get(...)/basis.get(...) → 对旧数据 AttributeError 崩。checker 对旧数据崩 # = checker 健壮性 bug, 不逼旧 findings 迁移 → 非 dict 时以空 dict 跑 (dict-key 门天然跳过), # 各留一条软提示防静默吞掉 (provenance: 不改别线 committed 结论, 只让 checker 不崩)。 cov = it.get('coverage') or {} if not isinstance(cov, dict): issues.append(f'{tag}: coverage 为 {type(cov).__name__} 非 dict (旧格式) → n_machines/window_days 等无法机检, 建议迁 dict (软提示)') cov = {} for k in ('n_machines', 'window_days'): if not cov.get(k): issues.append(f'{tag}: coverage.{k} 缺/空') if v not in ('INSUFFICIENT', '撤回') and not cov.get('per_day_evidence'): issues.append(f'{tag}: 非INSUFFICIENT结论缺 per_day_evidence (单日快照骗3次, #18)') basis = it.get('basis') or {} if not isinstance(basis, dict): issues.append(f'{tag}: basis 为 {type(basis).__name__} 非 dict (旧格式) → 降级门(mast/curtail/window等)无法机检, 建议迁 dict (软提示)') basis = {} cc = str(it.get('claim_class', '')) # claim_class = §0.2 降级表的键; 自声明无校验 = 改标签即关掉整表 (2026-07-16 红队 live 复现后闸) if v not in ('INSUFFICIENT', '撤回'): if not cc: issues.append(f'{tag}: 缺 claim_class (§0.2 降级表交叉断言的键, 缺省=绕过降级)') elif cc not in CLAIM_CLASSES: issues.append(f'{tag}: claim_class "{cc}" ∉ 枚举 {sorted(CLAIM_CLASSES)} (自由文本=降级表逃逸)') # 2026-08-10: 补 '撤回' 豁免 — 本文件 line197 与 line204 两处同类规则均豁免 '撤回', # 唯此处漏, 导致**已撤回的 finding 仍被当生效结论校验**(如东 findings[22] 长期 FAIL)。 # 撤回 = 该 claim 已作废, 不应再受内容规则约束; 三处规则的豁免集合就此对齐。 if cc == '绝对性能' and str(basis.get('mast_usability', '')).upper() != 'PASS' and v not in ('INSUFFICIENT', '撤回'): issues.append(f'{tag}: 绝对性能∧mast非PASS → 必 INSUFFICIENT (实为 {v})') if cc in ('绝对性能', '损失', '相对排名') and not basis.get('curtail_strip') and v not in ('INSUFFICIENT', '撤回'): issues.append(f'{tag}: {cc}类缺 basis.curtail_strip(剥限电证据) → 限电=性能/损失头号混杂(¥620万教训); ' f'补 {{method,n_removed}} 或显式 N/A (软警告, 审计 D4-4)') wm = basis.get('window_months') if cc == '资源' and isinstance(wm, (int, float)) and wm < 12 and v in ('定论', '准定论·预警', '候选'): issues.append(f'{tag}: 资源类单季窗({wm}月) → ≤参考 (实为 {v})') cp = it.get('cp_max', it.get('cp')) if isinstance(cp, (int, float)) and cp > 0.593 and v != 'INSUFFICIENT': issues.append(f'{tag}: Cp={cp}>0.593 Betz → INSUFFICIENT 物理铁证 (#32)') if not it.get('falsifiability'): issues.append(f'{tag}: 缺 falsifiability ("若X则假")') # 2026-07-16 Codex 对账三件: '其他'防错分类逃逸 + self_ref 降级行机器化 + [暂行] 跨场引用闸 (§0.3 待补兑现) if cc == '其他' and v not in ('INSUFFICIENT', '撤回') and not basis.get('claim_class_note'): issues.append(f'{tag}: claim_class=其他 须 basis.claim_class_note 注明为何不属既有类 (错分类=关降级表)') if basis.get('self_ref') and cc in ('相对排名', '趋势') and v in ('定论', '准定论·预警', '候选'): issues.append(f'{tag}: self_ref 风速上做{cc} → ≤参考 (§0.2 降级表, 实为 {v})') for i, ref in enumerate(it.get('cross_farm_refs') or []): if isinstance(ref, dict) and ref.get('tabu') and v != 'INSUFFICIENT': issues.append(f'{tag}: cross_farm_refs[{i}] 引 [暂行] 规则作 verdict 支撑 → 上限 INSUFFICIENT (§0.3 跨场引用闸, 实为 {v})') # §0.2 降级表机器覆盖补齐两行 (2026-07-17 红队 P1-6 附带; 字段自愿填=半机器化, 生产侧渐进接) if basis.get('reactive_fleet_relative') and not basis.get('q_setpoint_log') and v not in ('参考', 'INSUFFICIENT', '撤回'): issues.append(f'{tag}: 无功 fleet-relative 离群无园区 Q-setpoint 日志 → 归因 ≤【参考】 (§0.2 无功行, 实为 {v})') cwm = basis.get('calib_window_months') if isinstance(cwm, (int, float)) and cwm < 3 and not basis.get('calib_season_matched') and v in ('定论', '准定论·预警', '候选'): issues.append(f'{tag}: 变化型筛查标定窗 {cwm} 月 <1 季节周期且未同季匹配 → ≤【参考】 (§0.2 mset/NBM 行, 实为 {v})') # §4.6 温度域可选字段软校验 (2026-07-02 用户裁决清债; 缺省不报, 给了就须合法) avc = it.get('acute_vs_chronic') if avc is not None and avc not in ('急性', '慢性'): issues.append(f'{tag}: acute_vs_chronic "{avc}" ∉ {{急性,慢性}} (超限跳机码=急性, 无码持续偏离=慢性)') for bkey in ('oem_threshold_crossed', 'temp_clean_mech_not_cleared'): bv = it.get(bkey) if bv is not None and not isinstance(bv, bool): issues.append(f'{tag}: {bkey} 须 bool (OEM绝对阈交叉 / 温度干净≠机械已清→转CMS)') return issues def validate_review(it, tag='review'): """RV-2 单条评审判据 (docs/SOP.md §6.2).""" issues = [] rep = it.get('reproduced') if not rep or not all(isinstance(r, dict) and r.get('claim') and r.get('cmd') for r in rep): issues.append(f'{tag}: reproduced 缺或条目缺 claim/cmd (subagent 数字必复现, #23/#30)') elif not all(r.get('output_digest') for r in rep): issues.append(f'{tag}: reproduced 条目缺 output_digest (无输出摘要=复现不可核对, §6.2; 2026-07-16 Codex 外审补)') prov = it.get('provenance') if not prov or not all(isinstance(r, dict) and r.get('src_doc') and r.get('origin') for r in prov): issues.append(f'{tag}: provenance 缺或条目缺 src_doc/origin (归属=audit单点故障, zyx)') elif not any(r.get('independence') or r.get('独立性') or r.get('独立性声明') for r in prov): issues.append(f'{tag}: provenance 无任何独立性声明字段 (independence/独立性; "多源印证"隐含独立性 claim 必显式, §6.2; 2026-07-16 Codex 外审补)') return issues # §7.1 逐台结论总表 (压轴章) 字段契约 (2026-06-20 ③ 解锁): finding 可选 per-机字段单源. PROBLEM_NATURE = {'性能', '安全性', '可靠性', '部件', '可靠性/部件', '电量', '风资源', '控制', '外部(电网)', '整体', '数据'} def validate_per_machine(findings): """③ 逐台字段可选契约 (在场则类型核): affected_machines(list[str]|str) / recommendation(str) / problem_nature(str∈枚举). 场级/健康 finding 合法缺省 (非每条都有问题机); 仅当显式提供时核类型 — 防 producer 写脏 (与 §9.1.2 同源: 字段也要结构化)。 """ issues = [] for idx, it in enumerate(findings or []): if not isinstance(it, dict): continue am = it.get('affected_machines') if am is not None and not isinstance(am, (list, str)): issues.append(f'finding[{idx}]: affected_machines 须 list/str (实为 {type(am).__name__})') rec = it.get('recommendation') if rec is not None and not isinstance(rec, str): issues.append(f'finding[{idx}]: recommendation 须 str') pn = it.get('problem_nature') if pn is not None and pn not in PROBLEM_NATURE: issues.append(f'finding[{idx}]: problem_nature "{pn}" ∉ {sorted(PROBLEM_NATURE)}') return issues # _OV_MACHINE_RE + validate_owner_view_consistency (业主版散文↔结构化一致性闸, RV-2 a) 已删 — 取消业主版 (2026-06-20), owner_view 已无消费者。 def validate_conservation(findings, n_total, window_days=None, slack_days=2): """9.1.4 守恒校验 (docs/SOP.md §9.1.4): 异常机数≤总机数 / 结论窗⊆数据窗 / (有分项)总=Σ分项. findings: findings.json 的 findings 列表; n_total: 指纹场站总机数; window_days: 数据窗跨度天. 防"算术不闭合"造假 — 与 §5 D8 窗对齐互补 (D8 防窗错配, 此条防数算不闭合). """ issues = [] for idx, it in enumerate(findings or []): if not isinstance(it, dict): continue tag = f'finding[{idx}]({str(it.get("title", ""))[:18]})' cov = it.get('coverage') or {} if not isinstance(cov, dict): # 2026-09-07 实逮: 旧条目 coverage 为字符串 → 此处 .get 抛异常, PB-1 9.1.4 整段 fail-closed 报"解析失败", 真问题反而看不见 issues.append(f'{tag}: coverage 非结构化 (str), 守恒无法核 (9.1.4)') continue nm = cov.get('n_machines') if isinstance(nm, (int, float)) and isinstance(n_total, (int, float)) and nm > n_total: issues.append(f'{tag}: 异常机数 {nm} > 场站总机 {n_total} → 守恒不闭合 (9.1.4)') wd = cov.get('window_days') if (isinstance(wd, (int, float)) and isinstance(window_days, (int, float)) and window_days > 0 and wd > window_days + slack_days): issues.append(f'{tag}: 结论窗 {wd}天 > 数据窗 {window_days}天 → 时间不⊆数据 (9.1.4)') comp, tot = it.get('components'), it.get('total') if isinstance(comp, list) and comp and isinstance(tot, (int, float)): s = sum(c.get('value', 0) for c in comp if isinstance(c, dict)) if abs(s - tot) > max(1e-6, abs(tot) * 0.01): issues.append(f'{tag}: 分项和 {s} ≠ 总 {tot} (9.1.4 守恒, 容差1%)') wf = it.get('waterfall') if wf is not None: issues.extend(validate_waterfall(wf, f'{tag}.waterfall')) return issues def validate_waterfall(wf, tag='waterfall'): """9.1.4 损失瀑布守恒 (docs/SOP.md §4.1 模块1.2b + §9.1.4): 让能量/¥账长牙. Σ分项 == total(容差1%) / 各项≥0(物理) / 恰一 actual(实发)项 / 残差打包必声明 basis / 若声明 economic_total: 必 == Σ(损失项 value × tariff) (防手填¥, 须 电量×电价 可复算). """ issues = [] if not isinstance(wf, dict): return [f'{tag}: 非 dict'] total, comps = wf.get('total'), wf.get('components') if not isinstance(total, (int, float)): issues.append(f'{tag}: 缺 total 数值') if not isinstance(comps, list) or not comps: return issues + [f'{tag}: 缺 components 非空列表'] s, actual_n = 0.0, 0 for c in comps: if not isinstance(c, dict) or not isinstance(c.get('value'), (int, float)): issues.append(f'{tag}: 分项 {c} 缺数值 value') continue v = c['value'] if v < 0: issues.append(f'{tag}: 分项 "{c.get("name")}" {v}<0 (实发/损失不可负, 物理)') s += v actual_n += 1 if c.get('kind') == 'actual' else 0 if c.get('kind') == 'residual' and not c.get('basis'): issues.append(f'{tag}: 残差项 "{c.get("name")}" 无 basis → 禁硬拆打包残差 (§4.1 1.2b)') if isinstance(total, (int, float)) and abs(s - total) > max(1e-6, abs(total) * 0.01): issues.append(f'{tag}: 分项和 {round(s, 3)} ≠ total {total} (差 {round(s - total, 3)}) → 守恒不闭合 (9.1.4)') if actual_n != 1: issues.append(f'{tag}: 须恰 1 个 kind=actual(实发)项, 实为 {actual_n} (瀑布=实发+Σ损失)') econ = wf.get('economic_total') if isinstance(econ, (int, float)): e, ok = 0.0, True for c in comps: if isinstance(c, dict) and c.get('kind') == 'loss': v, tar = c.get('value'), c.get('tariff') if isinstance(v, (int, float)) and isinstance(tar, (int, float)): e += v * tar else: ok = False if not ok: issues.append(f'{tag}: 声明 economic_total 但损失项缺 tariff → ¥不可复算 (须电量×电价, 禁手填)') elif abs(e - econ) > max(1e-6, abs(econ) * 0.01): issues.append(f'{tag}: economic_total {econ} ≠ Σ(损失×电价) {round(e, 2)} → ¥手填嫌疑 (9.1.4 ¥派生闭合)') return issues def validate_repro(findings, require_for=('定论',)): """9.1.5 可复现脚本声明 (docs/SOP.md §9.1.5): verdict∈require_for 的 finding 必带 repro:{script, expected}. 双模型评审铁律: 仅"脚本存在"= theater (可 echo 假值绕过), 故 expected 期望值必填 (供 verify-repro 重跑比对). 本函数只校验声明完整性 (纯函数); 脚本实存 + 重跑数值比对在 report_integrity.verify_repro (有副作用, 不在此). """ issues = [] for idx, it in enumerate(findings or []): if not isinstance(it, dict): continue if norm_verdict(it.get('verdict', '')) not in require_for: continue tag = f'finding[{idx}]({str(it.get("title", ""))[:18]})' repro = it.get('repro') if not isinstance(repro, dict) or not repro.get('script'): issues.append(f'{tag}: verdict=定论 但缺 repro.script (9.1.5 定论必附可复现脚本)') continue if repro.get('expected') is None: issues.append(f'{tag}: repro 缺 expected 期望值 (9.1.5 防空壳脚本: 仅脚本存在可 echo 假值绕过, 须可重跑比对)') return issues # §0.8 ON-0 分析前置锁定门: analysis_lock.yaml 自洽闸 (机器位 scripts/analysis_lock_check.py) LOCK_REQUIRED = ['farm', 'frozen', 'windows', 'rated_kw', 'turbines', 'capacity_mw', 'sources', 'metrics', 'cp_ceiling', 'gates'] _LOCK_TODO = {'todo', '待填', '待定', 'tbd', '?', ''} def validate_required_outputs(doc, produced): """§0.8 required_outputs 完整性自检 — 必算清单缺一报错 (解 "分析项时有时无" = V-019 数据层翻版). lock.required_outputs = 本次分析必产出项清单 (声明式, 写死在锁里)。produced = 实际产出项 (list/set/dict, 来自运行 manifest 或产物目录扫描的文件名)。清单上每一项必在 produced 出现 (精确或子串命中), 缺一 → issue, **不允许静默跳过**。这把 V-019 "声称做了实际没做" 的防造假机制延伸到分析项层: 不是事后发现少了, 是清单强制它不能少。纯函数, 返回 issue 列表 (空=PASS)。 """ if not isinstance(doc, dict): return ['analysis_lock: 顶层非 dict'] req = doc.get('required_outputs') if not req: return ['analysis_lock: 缺 required_outputs (必算清单) → 无法做完整性自检, ' '"分析项时有时无" 无机器拦截 (§0.8/V-019 数据层)'] if not isinstance(req, (list, tuple)): return [f'analysis_lock: required_outputs 须为列表, 实为 {type(req).__name__}'] if isinstance(produced, dict): produced = list(produced.keys()) prod = {str(p).strip() for p in (produced or [])} issues = [] for item in req: key = (item.get('id') if isinstance(item, dict) else str(item)).strip() if not key: continue if not (key in prod or any(key in p for p in prod)): issues.append(f'analysis_lock·required_outputs: 必算项 "{key}" 未产出 ' f'(缺一报错, 禁静默跳过; §0.8/V-019 数据层)') return issues def validate_analysis_lock(doc, cap_tol=0.005): """§0.8 ON-0 分析前置锁定门 — analysis_lock.yaml 自洽闸 (解 "同数据两次结果不一"). 根因实证: 利用小时 1330 vs 2389 / CF 27% vs 15% / 改造前后 +0.3 vs +23.6 — 窗/口径/源/容量/基准未锁 → 同数据两次结果不一。本函数机器化 §0.8 B 自洽闸: A 冻结六项+frozen+gates 在场 (未冻结=不得出数)。 ① Σ台=场: Σ rated_kw[turbines[t]] == capacity_mw.全场 (容差) — catches 27.5 vs 48.92 类装机口径错。 ② 口径完整: metric 标 须标口径:true 必给区分口径值/注 (L4 含状态过滤+风速来源)。 ③ INSUFFICIENT 一致: metrics/cp_ceiling 标 状态:INSUFFICIENT 的量必在 gates.INSUFFICIENT标记 列出 (防漂移)。 ④ Betz: cp_ceiling 数值 Cp ≥0.593 = 物理违背 (#32)。 纯函数, 返回 issue 列表 (空=PASS)。report↔lock 口径交叉在 analysis_lock_check.py (有副作用, 不在此)。 """ if not isinstance(doc, dict): return ['analysis_lock: 顶层非 dict'] issues = [] for k in LOCK_REQUIRED: v = doc.get(k) if k not in doc or v is None or (isinstance(v, (str, list, dict)) and len(v) == 0): issues.append(f'analysis_lock: 冻结项缺/空: {k} (§0.8 A 六项+frozen+gates)') elif isinstance(v, str) and v.strip().lower() in _LOCK_TODO: issues.append(f'analysis_lock: {k}="{v}" 占位未填 → 未冻结不得出数 (§0.5/§0.8)') # ★ freeze 须用户裁决 (§0.8 freeze 权威=用户非Claude自冻; 解"没找我确认"洞, 2026-06-30 加) # 防: Claude 自拟+自冻+盖哈希=把临场决定伪装成有权威的契约。frozen 有值则 frozen_by 必含"用户裁决"。 frz = doc.get('frozen') if frz is not None and not (isinstance(frz, str) and frz.strip().lower() in _LOCK_TODO): fb = str(doc.get('frozen_by') or '') if '用户裁决' not in fb: issues.append('analysis_lock: frozen 已设但 frozen_by 非"用户裁决" → 自冻锁(Claude/非用户裁决)' '不得当冻结锁出数 (§0.8 freeze 权威=用户; Claude自冻+哈希=临场决定伪装契约)') # ① Σ台=场 (§0.8 B①) — 装机口径错 → CF/利用小时全错 rated, turb, cap = doc.get('rated_kw'), doc.get('turbines'), doc.get('capacity_mw') if isinstance(rated, dict) and isinstance(turb, dict) and isinstance(cap, dict): unknown = sorted({str(m) for m in turb.values() if m not in rated}) if unknown: issues.append(f'analysis_lock①Σ台=场: 机型 {unknown} 在 turbines 用到但 rated_kw 未定义') elif isinstance(cap.get('全场'), (int, float)): sum_mw = sum(float(rated[m]) for m in turb.values()) / 1000.0 tot = float(cap['全场']) if abs(sum_mw - tot) > max(cap_tol, tot * cap_tol): issues.append(f'analysis_lock①Σ台=场不闭合: Σ台 {sum_mw:.3f}MW ≠ capacity_mw.全场 {tot}MW ' f'(差 {sum_mw - tot:+.3f}; §0.8 B① 装机口径错→CF/利用小时全错)') # ② 口径完整 (§0.8 L4) metrics = doc.get('metrics') if isinstance(doc.get('metrics'), dict) else {} for mn, mv in metrics.items(): if isinstance(mv, dict) and mv.get('须标口径') is True: extra = [x for x in mv if x not in ('须标口径', '定义', '注', '值')] if len(extra) < 2 and not mv.get('注'): issues.append(f'analysis_lock②口径: "{mn}" 标须标口径 却未给区分口径值/注 (L4 须含状态过滤+风速来源)') # ③ INSUFFICIENT 一致 (防 一处诚实留口、另一处被当定论引用) gates = doc.get('gates') if isinstance(doc.get('gates'), dict) else {} declared = json.dumps(gates.get('INSUFFICIENT标记') or [], ensure_ascii=False) cpc = doc.get('cp_ceiling') if isinstance(doc.get('cp_ceiling'), dict) else {} for src_name, src in (('metrics', metrics), ('cp_ceiling', cpc)): for mn, mv in (src.items() if isinstance(src, dict) else []): if isinstance(mv, dict) and str(mv.get('状态', '')).upper() == 'INSUFFICIENT' and str(mn) not in declared: issues.append(f'analysis_lock③: {src_name}."{mn}" 标 状态:INSUFFICIENT 但未在 gates.INSUFFICIENT标记 列出 → 口径漂移风险') # ④ Betz for k, v in (cpc.items() if isinstance(cpc, dict) else []): if isinstance(v, (int, float)) and 'cp' in str(k).lower() and v >= 0.593: issues.append(f'analysis_lock④Betz: cp_ceiling.{k}={v} ≥0.593 物理不可能 (#32 Cp>Betz 铁证)') return issues # ============================================================================ # 附录A Case 状态机 (docs/运维Agent转化评估_v0.1.md 附录A → 代码化; 运维 agent 柱1「智能工单」) # ---------------------------------------------------------------------------- # findings.json = 每批全量快照(无记性, B16 前兆每月被"重新发现"); case = 跨批次状态延续 + # 反向闭环(evidence→ticket→verify→CLOSED/RELAPSED→回灌 记分卡/基准库/误报台账)。 # 本节把附录A 10 态状态机 + 6 条机器不变量落成纯函数闸(风格同上: 返 issue 列表, 空=PASS)。 # 工单卡(TICKETED 载体)/回验卡(VERIFY 期)= case 生命周期的两个切面, 各自 schema 在此定义。 # 正交两轴不进 state (附录A.2): tier(T1/T2/T3 紧迫度) / verdict(§0.2 证据强度) = case 属性。 # ============================================================================ CASE_STATES = {'NORMAL', 'WATCH', 'CANDIDATE', 'CONFIRMED', 'TICKETED', 'IN_REPAIR', 'VERIFY', 'CLOSED', 'REFUTED', 'RELAPSED'} # 合法状态迁移边 (附录A.2; 非法跳转拒绝 = 不变量⑤) CASE_EDGES = { 'NORMAL': {'WATCH'}, 'WATCH': {'CANDIDATE', 'NORMAL'}, 'CANDIDATE': {'CONFIRMED', 'NORMAL', 'REFUTED'}, 'CONFIRMED': {'TICKETED', 'REFUTED'}, 'TICKETED': {'IN_REPAIR', 'VERIFY'}, 'IN_REPAIR': {'VERIFY'}, 'VERIFY': {'CLOSED', 'CONFIRMED'}, # verify_fail → 回 CONFIRMED (未根治, 变桨 SOP A014 反弹式) 'CLOSED': {'RELAPSED'}, 'REFUTED': set(), # 终态 (留档防再登顶) 'RELAPSED': {'CONFIRMED'}, } # 确诊过之后 own-data 证据须留存 (不变量①: 禁 fleet-z 直升终判) CASE_CONFIRMED_OR_BEYOND = {'CONFIRMED', 'TICKETED', 'IN_REPAIR', 'VERIFY', 'CLOSED', 'RELAPSED'} # 工单已签发之后 (不变量②: 人工裁决点①机器不可绕) CASE_TICKETED_OR_BEYOND = {'TICKETED', 'IN_REPAIR', 'VERIFY', 'CLOSED', 'RELAPSED'} CASE_TERMINAL = {'CLOSED', 'REFUTED'} # 非终态判据 (附录A.4⑥ 至多一个活跃档) CASE_REQUIRED = ['case_id', 'farm', 'turbine', 'system', 'state', 'opened_at', 'updated_at'] CASE_TIERS = {None, 'T1', 'T2', 'T3'} def validate_ticket_card(t, tag='ticket'): """工单卡 schema (附录A.3 ticket + P4 七段结构化): TICKETED 载体. 必: ticket_id / issued_by / actions(非空, = P4 §3 排查步骤). 可选: window_suggested(B1 排程建议) / parts / priority(P0/P1/P2). 无动作的工单 = 糊涂账 (P4 硬纪律)。""" if not isinstance(t, dict): return [f'{tag}: 非 dict'] issues = [] for k in ('ticket_id', 'issued_by'): if not t.get(k): issues.append(f'{tag}: 缺 {k}') acts = t.get('actions') if not isinstance(acts, list) or not acts: issues.append(f'{tag}: actions 须非空列表 (P4 §3 排查步骤; 无动作的工单=糊涂账)') pr = t.get('priority') if pr is not None and pr not in ('P0', 'P1', 'P2'): issues.append(f'{tag}: priority "{pr}" ∉ {{P0,P1,P2}} (P4 优先级)') parts = t.get('parts') if parts is not None and not isinstance(parts, list): issues.append(f'{tag}: parts 须 list') return issues def validate_verify_card(vf, tag='verify'): """回验卡 schema (附录A.3 verify): VERIFY 期量化闭环 = 反向闭环的一半价值 (P4 §7). 必: baseline(修前判据量) / target(回归带) / result∈{pass,fail,pending}. result=pass → readings 须非空 (无实测读数不得判 pass; 防"修完就是修好"糊涂账)。""" if not isinstance(vf, dict): return [f'{tag}: 非 dict'] issues = [] for k in ('baseline', 'target'): if vf.get(k) is None: issues.append(f'{tag}: 缺 {k} (修前基线/回归带; 无基线无法判回归)') res = vf.get('result') if res not in ('pass', 'fail', 'pending'): issues.append(f'{tag}: result "{res}" ∉ {{pass,fail,pending}}') rd = vf.get('readings') if res == 'pass' and (not isinstance(rd, list) or not rd): issues.append(f'{tag}: result=pass 但 readings 空 → 无实测读数不得判回归 (P4 §7 回验量化)') wd = vf.get('window_days') if wd is not None and not isinstance(wd, (int, float)): issues.append(f'{tag}: window_days 须数值') return issues def validate_analysis(an, tag='analysis'): """深挖层 schema (Phase-2 机制归因 + 独立审): case.analysis 可选扩展。 必: verdict(六枚举) + mechanism(机制, 无机制=空 verdict). 可选: review/field_needs/ref/date/by. ★纪律: analysis.verdict 是更深表征(可 > case.verdict, 如 D29 案候选→深挖准定论·预警), 但**不改 case.state** — 升 CONFIRMED 须 own-data(不变量①), 走 apply_human_action。""" if not isinstance(an, dict): return [f'{tag}: 非 dict'] issues = [] if norm_verdict(an.get('verdict', '')) not in VERDICTS: issues.append(f'{tag}: verdict "{an.get("verdict")}" ∉ 六枚举') if not an.get('mechanism'): issues.append(f'{tag}: 缺 mechanism (深挖必给机制归因, 否则=空 verdict)') return issues def validate_case(case, tag='case'): """附录A Case 状态机单档校验 (纯函数, 空=PASS). 6 条机器不变量 = 附录A.4. 注: 不变量⑤"追加式禁改写"的 append-only 语义无法从单份快照证 (需跨版本 diff, 在有副作用的 回灌器里查); 本函数证其可证部分 = 迁移边合法 + state↔审计链一致 + 结构完整。""" if not isinstance(case, dict): return [f'{tag}: 顶层非 dict'] cid = case.get('case_id', '?') tg = f'{tag}[{cid}]' issues = [] # 字段在场 (附录A.3) for k in CASE_REQUIRED: if not case.get(k): issues.append(f'{tg}: 缺 {k}') state = case.get('state') if state is not None and state not in CASE_STATES: issues.append(f'{tg}: state "{state}" ∉ 十态 {sorted(CASE_STATES)}') return issues # state 非法, 后续不变量无从判 # 正交轴类型 (附录A.2: 紧迫度/证据强度 ≠ 状态) if case.get('tier') not in CASE_TIERS: issues.append(f'{tg}: tier "{case.get("tier")}" ∉ {{None,T1,T2,T3}} (紧迫度≠状态, 正交轴)') v = case.get('verdict') if v is not None and norm_verdict(v) not in VERDICTS: issues.append(f'{tg}: verdict "{v}" ∉ 六枚举 (证据强度≠状态, 正交轴)') # evidence 追加流水 (附录A.3): discriminator_id 外键 + verdict 合法 ev = case.get('evidence') or [] if not isinstance(ev, list): issues.append(f'{tg}: evidence 须 list') ev = [] for j, e in enumerate(ev): if not isinstance(e, dict): issues.append(f'{tg}: evidence[{j}] 非 dict') continue if not e.get('discriminator_id'): issues.append(f'{tg}: evidence[{j}] 缺 discriminator_id (注册表外键, 判据溯源)') ev_v = e.get('verdict') if ev_v is not None and norm_verdict(ev_v) not in VERDICTS: issues.append(f'{tg}: evidence[{j}] verdict "{ev_v}" ∉ 六枚举') # history append-only 审计链: 每条迁移边合法 (不变量⑤可证部分) + by 枚举 hist = case.get('history') or [] if not isinstance(hist, list): issues.append(f'{tg}: history 须 list') hist = [] for j, h in enumerate(hist): if not isinstance(h, dict): issues.append(f'{tg}: history[{j}] 非 dict') continue fr, to, by = h.get('from_state'), h.get('to_state'), h.get('by') if by not in ('engine', 'human'): issues.append(f'{tg}: history[{j}] by "{by}" ∉ {{engine,human}}') if fr not in CASE_STATES or to not in CASE_STATES: issues.append(f'{tg}: history[{j}] from/to 非十态 ({fr}→{to})') elif to not in CASE_EDGES.get(fr, set()): issues.append(f'{tg}: history[{j}] 非法迁移 {fr}→{to} (附录A.2 未定义边; 不变量⑤)') if hist and isinstance(hist[-1], dict) and hist[-1].get('to_state') and state: if hist[-1]['to_state'] != state: issues.append(f'{tg}: state={state} ≠ history 末态 {hist[-1]["to_state"]} (审计链不一致)') # ── 6 条机器不变量 (附录A.4) ── # ① CONFIRMED 及以后 → evidence 含 ≥1 own_data 级 (禁 fleet-z 直升终判, 验证金字塔③) if state in CASE_CONFIRMED_OR_BEYOND and not any( isinstance(e, dict) and e.get('own_data') for e in ev): issues.append(f'{tg}①: state={state} 但 evidence 无 own_data 级判据 → 禁 fleet-z 直升 (金字塔③终判)') # ② TICKETED 及以后 → history 含 by=human 的 ticket_issued + 工单卡在场合法 (人工裁决点①机器不可绕) if state in CASE_TICKETED_OR_BEYOND: if not any(isinstance(h, dict) and h.get('by') == 'human' and h.get('event') == 'ticket_issued' for h in hist): issues.append(f'{tg}②: state={state} 但 history 无 by=human 的 ticket_issued (人工裁决点①机器不可绕)') issues += validate_ticket_card(case.get('ticket') or {}, f'{tg}.ticket') # ③ CLOSED → verify.result=pass (无回验不销案; 昌邑 A27 悬案按构造杜绝) if state == 'CLOSED': vfr = (case.get('verify') or {}).get('result') if vfr != 'pass': issues.append(f'{tg}③: state=CLOSED 但 verify.result≠pass ("{vfr}") → 无回验不销案 (附录A.4③)') if state in ('VERIFY', 'CLOSED'): issues += validate_verify_card(case.get('verify') or {}, f'{tg}.verify') # ④ REFUTED → refuted_basis 非空 (防同信号再登顶) if state == 'REFUTED' and not case.get('refuted_basis'): issues.append(f'{tg}④: state=REFUTED 但 refuted_basis 空 → 防同信号再登顶 (附录A.4④)') # 深挖层 (可选扩展): case.analysis 在场则校验 schema (Phase-2 机制 + 独立审) if case.get('analysis') is not None: issues += validate_analysis(case['analysis'], f'{tg}.analysis') # ⑥ RELAPSED → relapse_of 指原 case (D7 复发率数据源; 跨档"至多一活跃"在 validate_cases) if state == 'RELAPSED' and not case.get('relapse_of'): issues.append(f'{tg}⑥: state=RELAPSED 但缺 relapse_of (指原 case; D7 复发率数据源)') return issues def validate_cases(cases, tag='cases'): """整份 cases.json 跨档校验 (附录A.4⑥): 逐档 validate_case + case_id 唯一 + 同 (farm,turbine,system) 至多一个非终态 case (防重复开案).""" if not isinstance(cases, list): return [f'{tag}: 顶层须 list'] issues, seen_id, active = [], set(), {} for c in cases: issues += validate_case(c) if not isinstance(c, dict): continue cid = c.get('case_id') if cid in seen_id: issues.append(f'{tag}: case_id 重复 {cid}') seen_id.add(cid) if c.get('state') not in CASE_TERMINAL: key = (c.get('farm'), c.get('turbine'), c.get('system')) if key in active: issues.append(f'{tag}: {key} 同时 >1 非终态 case ({active[key]} + {cid}) → 防重复开案 (附录A.4⑥)') active[key] = cid return issues