schemas.py 41 KB

123456789101112131415161718192021222324252627282930313233343536373839404142434445464748495051525354555657585960616263646566676869707172737475767778798081828384858687888990919293949596979899100101102103104105106107108109110111112113114115116117118119120121122123124125126127128129130131132133134135136137138139140141142143144145146147148149150151152153154155156157158159160161162163164165166167168169170171172173174175176177178179180181182183184185186187188189190191192193194195196197198199200201202203204205206207208209210211212213214215216217218219220221222223224225226227228229230231232233234235236237238239240241242243244245246247248249250251252253254255256257258259260261262263264265266267268269270271272273274275276277278279280281282283284285286287288289290291292293294295296297298299300301302303304305306307308309310311312313314315316317318319320321322323324325326327328329330331332333334335336337338339340341342343344345346347348349350351352353354355356357358359360361362363364365366367368369370371372373374375376377378379380381382383384385386387388389390391392393394395396397398399400401402403404405406407408409410411412413414415416417418419420421422423424425426427428429430431432433434435436437438439440441442443444445446447448449450451452453454455456457458459460461462463464465466467468469470471472473474475476477478479480481482483484485486487488489490491492493494495496497498499500501502503504505506507508509510511512513514515516517518519520521522523524525526527528529530531532533534535536537538539540541542543544545546547548549550551552553554555556557558559560561562563564565566567568569570571572573574575576577578579580581582583584585586587588589590591592593594595596597598599600601602603604605606607608609610611612613614615616617618619620621622623624625626627628629630631632633634635636637638639640641642643644645646647648649650651652653654655656657658659660661662663664665666667668669670671672673674675676677678679680681682683684685686687688689690691692693694695696697698699700
  1. # -*- coding: utf-8 -*-
  2. """schemas.py — per 场 SOP artifact 链七 JSON schema 的单一权威源 (docs/SOP.md 附录B).
  3. 被 scripts/sop_check.py 与 (Batch2) 生成器共用; 改 schema 只改这里。
  4. 最小判据级 schema (非 jsonschema 库): 每 artifact 声明必填键 + 类型/枚举约束,
  5. validate(doc, name) 返回 issue 字符串列表 (空 = PASS)。
  6. """
  7. import re
  8. import json
  9. from fnmatch import fnmatch
  10. VERDICTS = {'定论', '准定论·预警', '候选', '参考', 'INSUFFICIENT', '撤回'}
  11. CLAIM_CLASSES = {'绝对性能', '相对排名', '趋势', '事件归因', '资源', '限电', '其他',
  12. '部件健康', '数据质量', '损失', '安全性'} # 后四类 2026-07-16 收编存量真实用法 (红队: 枚举定义了却无人校验)
  13. # artifact 名 → 顶层必填键 (dict 类 artifact)
  14. REQUIRED_KEYS = {
  15. 'survey.json': ['data_layers', 'window_matrix', 'n_machines_claimed', 'n_with_data',
  16. 'data_attribution', 'materials_read'],
  17. 'goem.json': ['fields'], # per 字段: 实证方法+结论+证据数字
  18. 'gate_record': ['section', 'verdict', 'assertions'], # gate_<section>.json
  19. 'clean_gate.json': ['clip_counts', 'sentinel_to_nan', 'dead_columns', 'rows_in_out'],
  20. 'mast_usability.json': ['shear_physical', 'direction_consistency', 'monthly_ratio', 'verdict'],
  21. 'findings_item': ['verdict', 'coverage', 'claim_class', 'basis', 'falsifiability'],
  22. 'review_item': ['reproduced', 'provenance'],
  23. 'input_fingerprint.json': ['farm', 'n_machines', 'window_min', 'window_max', 'layers'], # §9.1.1
  24. 'disk_manifest.json': ['source_roots', 'layers', 'by_ext'], # ON-1 对账门 onboard 端 (scripts/intake_scan.py)
  25. 'cleaned_fingerprint.json': ['section', 'parquet', 'sha256', 'rows', 'contract'], # §3 清洗快照留存(审计)
  26. }
  27. # §4 分析模块登记表 (configs/analysis_modules.yaml) — 每模块标准卡必填键 (单源, 改 schema 只改这里)
  28. MODULE_REQUIRED = ['id', 'name', 'domain', 'role', 'question', 'inputs', 'preconditions',
  29. 'method', 'threshold_basis', 'verdict_ceiling', 'falsifiability', 'impl', 'n', 'status']
  30. MODULE_DOMAINS = {'M0', 'M1', 'M2', 'M3', '横切'} # 机理域 (2026-07-08 D→M namespace; 交付九域=D1-D9, 审计门=AG0-AG9, 见 SOP §4 crosswalk)
  31. MODULE_ROLES = {'Gate', 'Diag', 'Integ', 'Act', 'Confirm'}
  32. MODULE_STATUS = {'稳定', '暂行', '待首跑', '条件性', 'N/A'}
  33. def norm_verdict(v):
  34. s = str(v).strip().strip('【】')
  35. # 判据函数筛查输出形态 ≡ 候选 (SOP §9.1.5 映射表 2026-07-16; 候选=内部流转禁入业主报告正文)
  36. return '候选' if s == 'CANDIDATE_RANKING' else s
  37. # 这些键 空 list/dict = 合法结果 (无死列 / 零 clip / 零哨兵 ≠ 漏算), 只查"在场"不要求非空
  38. ALLOW_EMPTY = {'dead_columns', 'clip_counts', 'sentinel_to_nan'}
  39. def validate_keys(doc, name):
  40. req = REQUIRED_KEYS.get(name, [])
  41. if not isinstance(doc, dict):
  42. return [f'{name}: 顶层非 dict']
  43. issues = []
  44. for k in req:
  45. if k in ALLOW_EMPTY:
  46. if k not in doc: # 空合法, 仅查键在场
  47. issues.append(f'{name}: 键缺: {k}')
  48. elif not doc.get(k):
  49. issues.append(f'{name}: 键缺/空: {k}')
  50. return issues
  51. def validate_module(rec, tag='module'):
  52. """单条分析模块卡校验: 必填键非空 + domain/role/status 枚举 + §0.3 准入 (n<3 须非稳定)."""
  53. if not isinstance(rec, dict):
  54. return [f'{tag}: 顶层非 dict']
  55. mid = rec.get('id', '?')
  56. issues = []
  57. for k in MODULE_REQUIRED:
  58. v = rec.get(k)
  59. if v is None or (isinstance(v, (str, list, dict)) and len(v) == 0):
  60. issues.append(f'module[{mid}]: 键缺/空: {k}')
  61. dom = rec.get('domain')
  62. if dom is not None and dom not in MODULE_DOMAINS:
  63. issues.append(f'module[{mid}]: domain 非法 {dom!r} (∈ {sorted(MODULE_DOMAINS)})')
  64. roles = rec.get('role') or []
  65. if isinstance(roles, list):
  66. bad = [r for r in roles if r not in MODULE_ROLES]
  67. if bad:
  68. issues.append(f'module[{mid}]: role 非法 {bad} (∈ {sorted(MODULE_ROLES)})')
  69. st = rec.get('status')
  70. if st is not None and st not in MODULE_STATUS:
  71. issues.append(f'module[{mid}]: status 非法 {st!r} (∈ {sorted(MODULE_STATUS)})')
  72. n = rec.get('n')
  73. if isinstance(n, int) and n < 3 and st == '稳定': # §0.3 准入闸: n<3 必标 [暂行]/条件性
  74. issues.append(f'module[{mid}]: n={n}<3 但 status=稳定 (§0.3 须标 暂行/条件性/待首跑)')
  75. return issues
  76. def validate_modules_doc(doc):
  77. """整份 analysis_modules.yaml: doc['modules'] 列表逐条 + id 唯一. 返回 issue 列表 (空=PASS)."""
  78. if not isinstance(doc, dict) or not isinstance(doc.get('modules'), list):
  79. return ['analysis_modules: 顶层须 dict 且含 modules 列表']
  80. issues, seen = [], set()
  81. for rec in doc['modules']:
  82. issues += validate_module(rec)
  83. mid = rec.get('id')
  84. if mid in seen:
  85. issues.append(f'module[{mid}]: id 重复')
  86. seen.add(mid)
  87. return issues
  88. def _name_tokens(name):
  89. """从文件/层名抽可匹配 token: 英数字 run(≥3) + CJK 连续子串(≥3). 鲁棒于缩写/空格/标点差异
  90. (disk 'FD116A-…-风力发电机组产品说明书' vs survey 'FD116A产品说明书' → 共享 '产品说明书'/'fd116a' 即认领)."""
  91. s = re.sub(r'\.[a-z0-9]{1,5}$', '', str(name).lower()) # 去扩展名: 防 'pdf'/'csv' 等通用 ext 假认领
  92. toks = set(re.findall(r'[a-z0-9]{3,}', s))
  93. for run in re.findall(r'[一-鿿]{3,}', s):
  94. run = run[:24]
  95. for L in range(3, len(run) + 1):
  96. for i in range(len(run) - L + 1):
  97. toks.add(run[i:i + L])
  98. return toks
  99. def reconcile_survey(dman, survey):
  100. """ON-1 对账 (docs/SOP.md §1.1): disk_manifest 枚举的每个数据层/资料/压缩包都必须在 survey.json
  101. 被认领 (claim-text 含其名 token), 否则 = 漏读 → FAIL. 唯一豁免 = survey.skipped:[{path,reason}] (reason 必填).
  102. 纯函数 (不碰文件系统; 枚举副作用在 scripts/intake_scan.py). 与 report 端 §9.1.1 指纹门对称:
  103. 一道证"进来的原始全枚举", 一道证"出去的清洗后可复算" → 两端闭环防漏读。
  104. """
  105. issues = []
  106. if not isinstance(dman, dict) or not isinstance(survey, dict):
  107. return ['reconcile: disk_manifest/survey 非 dict']
  108. skip_globs = []
  109. for s in (survey.get('skipped') or []):
  110. if not isinstance(s, dict) or not s.get('reason') or not (s.get('path') or s.get('glob')):
  111. issues.append(f'reconcile: skipped 条目缺 path/reason → 豁免必显式带理由 ({s})')
  112. else:
  113. skip_globs.append(s.get('path') or s.get('glob'))
  114. survey_wo_skip = {k: v for k, v in survey.items() if k != 'skipped'}
  115. claim = json.dumps(survey_wo_skip, ensure_ascii=False).lower()
  116. def basename(p):
  117. return str(p).replace('\\', '/').rstrip('/').rsplit('/', 1)[-1]
  118. def accounted(name, path):
  119. if any(t in claim for t in _name_tokens(name)):
  120. return True
  121. # 短名盲区补丁 (2026-07-14 古城 "6.zip"/海装 "总览.xls" 实证): stem<3字符/CJK<3 无 token 可抽
  122. # → 永 FAIL 无法认领。补精确文件名子串分支, 边界字符防 "6.zip" 误配 "16.zip"。
  123. bn = basename(name or path).lower()
  124. if bn and (f'/{bn}' in claim or f'"{bn}' in claim):
  125. return True
  126. return any(fnmatch(str(path), g) or fnmatch(str(name), g) or g in str(path) for g in skip_globs)
  127. for ly in dman.get('layers', []):
  128. pth = ly.get('path', '')
  129. nm = ly.get('name') or basename(pth)
  130. if not accounted(nm, pth):
  131. issues.append(f'reconcile: 数据层 "{pth}" ({ly.get("n_data_files")}文件) 未在 survey 认领 '
  132. f'→ 漏读 (§1.1①递归列全数据层); 在 survey 认领或加 skipped:{{path,reason}}')
  133. for mt in dman.get('materials', []):
  134. pth = mt.get('path', '')
  135. nm = mt.get('name') or basename(pth)
  136. if not accounted(nm, pth):
  137. issues.append(f'reconcile: 资料 "{pth}" 未在 survey.materials_read 认领 '
  138. f'→ 漏读 (§1.1②读全非数据资料); 在 survey 认领或加 skipped:{{path,reason}}')
  139. for ar in dman.get('archives', []):
  140. pth = ar.get('path', '')
  141. nm = ar.get('name') or basename(pth)
  142. if not accounted(nm, pth):
  143. issues.append(f'reconcile: 压缩包 "{pth}" 未在 survey 认领 → 漏读 (压缩易藏整层); '
  144. f'在 survey 认领或加 skipped:{{path,reason}}')
  145. return issues
  146. def validate_finding(it, tag='finding'):
  147. """AN-1 单条 finding 判据 (docs/SOP.md §4.0 + §0.2 降级表)."""
  148. issues = []
  149. v = norm_verdict(it.get('verdict', ''))
  150. if v not in VERDICTS:
  151. return [f'{tag}: verdict "{it.get("verdict")}" ∉ 六枚举 {sorted(VERDICTS)}']
  152. # 旧格式 findings (如白山/淄川/诺木洪 CMS 线) 把 coverage/basis 写成 str, 新 schema 的
  153. # 降级门读 cov.get(...)/basis.get(...) → 对旧数据 AttributeError 崩。checker 对旧数据崩
  154. # = checker 健壮性 bug, 不逼旧 findings 迁移 → 非 dict 时以空 dict 跑 (dict-key 门天然跳过),
  155. # 各留一条软提示防静默吞掉 (provenance: 不改别线 committed 结论, 只让 checker 不崩)。
  156. cov = it.get('coverage') or {}
  157. if not isinstance(cov, dict):
  158. issues.append(f'{tag}: coverage 为 {type(cov).__name__} 非 dict (旧格式) → n_machines/window_days 等无法机检, 建议迁 dict (软提示)')
  159. cov = {}
  160. for k in ('n_machines', 'window_days'):
  161. if not cov.get(k):
  162. issues.append(f'{tag}: coverage.{k} 缺/空')
  163. if v not in ('INSUFFICIENT', '撤回') and not cov.get('per_day_evidence'):
  164. issues.append(f'{tag}: 非INSUFFICIENT结论缺 per_day_evidence (单日快照骗3次, #18)')
  165. basis = it.get('basis') or {}
  166. if not isinstance(basis, dict):
  167. issues.append(f'{tag}: basis 为 {type(basis).__name__} 非 dict (旧格式) → 降级门(mast/curtail/window等)无法机检, 建议迁 dict (软提示)')
  168. basis = {}
  169. cc = str(it.get('claim_class', ''))
  170. # claim_class = §0.2 降级表的键; 自声明无校验 = 改标签即关掉整表 (2026-07-16 红队 live 复现后闸)
  171. if v not in ('INSUFFICIENT', '撤回'):
  172. if not cc:
  173. issues.append(f'{tag}: 缺 claim_class (§0.2 降级表交叉断言的键, 缺省=绕过降级)')
  174. elif cc not in CLAIM_CLASSES:
  175. issues.append(f'{tag}: claim_class "{cc}" ∉ 枚举 {sorted(CLAIM_CLASSES)} (自由文本=降级表逃逸)')
  176. # 2026-08-10: 补 '撤回' 豁免 — 本文件 line197 与 line204 两处同类规则均豁免 '撤回',
  177. # 唯此处漏, 导致**已撤回的 finding 仍被当生效结论校验**(如东 findings[22] 长期 FAIL)。
  178. # 撤回 = 该 claim 已作废, 不应再受内容规则约束; 三处规则的豁免集合就此对齐。
  179. if cc == '绝对性能' and str(basis.get('mast_usability', '')).upper() != 'PASS' and v not in ('INSUFFICIENT', '撤回'):
  180. issues.append(f'{tag}: 绝对性能∧mast非PASS → 必 INSUFFICIENT (实为 {v})')
  181. if cc in ('绝对性能', '损失', '相对排名') and not basis.get('curtail_strip') and v not in ('INSUFFICIENT', '撤回'):
  182. issues.append(f'{tag}: {cc}类缺 basis.curtail_strip(剥限电证据) → 限电=性能/损失头号混杂(¥620万教训); '
  183. f'补 {{method,n_removed}} 或显式 N/A (软警告, 审计 D4-4)')
  184. wm = basis.get('window_months')
  185. if cc == '资源' and isinstance(wm, (int, float)) and wm < 12 and v in ('定论', '准定论·预警', '候选'):
  186. issues.append(f'{tag}: 资源类单季窗({wm}月) → ≤参考 (实为 {v})')
  187. cp = it.get('cp_max', it.get('cp'))
  188. if isinstance(cp, (int, float)) and cp > 0.593 and v != 'INSUFFICIENT':
  189. issues.append(f'{tag}: Cp={cp}>0.593 Betz → INSUFFICIENT 物理铁证 (#32)')
  190. if not it.get('falsifiability'):
  191. issues.append(f'{tag}: 缺 falsifiability ("若X则假")')
  192. # 2026-07-16 Codex 对账三件: '其他'防错分类逃逸 + self_ref 降级行机器化 + [暂行] 跨场引用闸 (§0.3 待补兑现)
  193. if cc == '其他' and v not in ('INSUFFICIENT', '撤回') and not basis.get('claim_class_note'):
  194. issues.append(f'{tag}: claim_class=其他 须 basis.claim_class_note 注明为何不属既有类 (错分类=关降级表)')
  195. if basis.get('self_ref') and cc in ('相对排名', '趋势') and v in ('定论', '准定论·预警', '候选'):
  196. issues.append(f'{tag}: self_ref 风速上做{cc} → ≤参考 (§0.2 降级表, 实为 {v})')
  197. for i, ref in enumerate(it.get('cross_farm_refs') or []):
  198. if isinstance(ref, dict) and ref.get('tabu') and v != 'INSUFFICIENT':
  199. issues.append(f'{tag}: cross_farm_refs[{i}] 引 [暂行] 规则作 verdict 支撑 → 上限 INSUFFICIENT (§0.3 跨场引用闸, 实为 {v})')
  200. # §0.2 降级表机器覆盖补齐两行 (2026-07-17 红队 P1-6 附带; 字段自愿填=半机器化, 生产侧渐进接)
  201. if basis.get('reactive_fleet_relative') and not basis.get('q_setpoint_log') and v not in ('参考', 'INSUFFICIENT', '撤回'):
  202. issues.append(f'{tag}: 无功 fleet-relative 离群无园区 Q-setpoint 日志 → 归因 ≤【参考】 (§0.2 无功行, 实为 {v})')
  203. cwm = basis.get('calib_window_months')
  204. if isinstance(cwm, (int, float)) and cwm < 3 and not basis.get('calib_season_matched') and v in ('定论', '准定论·预警', '候选'):
  205. issues.append(f'{tag}: 变化型筛查标定窗 {cwm} 月 <1 季节周期且未同季匹配 → ≤【参考】 (§0.2 mset/NBM 行, 实为 {v})')
  206. # §4.6 温度域可选字段软校验 (2026-07-02 用户裁决清债; 缺省不报, 给了就须合法)
  207. avc = it.get('acute_vs_chronic')
  208. if avc is not None and avc not in ('急性', '慢性'):
  209. issues.append(f'{tag}: acute_vs_chronic "{avc}" ∉ {{急性,慢性}} (超限跳机码=急性, 无码持续偏离=慢性)')
  210. for bkey in ('oem_threshold_crossed', 'temp_clean_mech_not_cleared'):
  211. bv = it.get(bkey)
  212. if bv is not None and not isinstance(bv, bool):
  213. issues.append(f'{tag}: {bkey} 须 bool (OEM绝对阈交叉 / 温度干净≠机械已清→转CMS)')
  214. return issues
  215. def validate_review(it, tag='review'):
  216. """RV-2 单条评审判据 (docs/SOP.md §6.2)."""
  217. issues = []
  218. rep = it.get('reproduced')
  219. if not rep or not all(isinstance(r, dict) and r.get('claim') and r.get('cmd') for r in rep):
  220. issues.append(f'{tag}: reproduced 缺或条目缺 claim/cmd (subagent 数字必复现, #23/#30)')
  221. elif not all(r.get('output_digest') for r in rep):
  222. issues.append(f'{tag}: reproduced 条目缺 output_digest (无输出摘要=复现不可核对, §6.2; 2026-07-16 Codex 外审补)')
  223. prov = it.get('provenance')
  224. if not prov or not all(isinstance(r, dict) and r.get('src_doc') and r.get('origin') for r in prov):
  225. issues.append(f'{tag}: provenance 缺或条目缺 src_doc/origin (归属=audit单点故障, zyx)')
  226. elif not any(r.get('independence') or r.get('独立性') or r.get('独立性声明') for r in prov):
  227. issues.append(f'{tag}: provenance 无任何独立性声明字段 (independence/独立性; "多源印证"隐含独立性 claim 必显式, §6.2; 2026-07-16 Codex 外审补)')
  228. return issues
  229. # §7.1 逐台结论总表 (压轴章) 字段契约 (2026-06-20 ③ 解锁): finding 可选 per-机字段单源.
  230. PROBLEM_NATURE = {'性能', '安全性', '可靠性', '部件', '可靠性/部件', '电量', '风资源',
  231. '控制', '外部(电网)', '整体', '数据'}
  232. def validate_per_machine(findings):
  233. """③ 逐台字段可选契约 (在场则类型核): affected_machines(list[str]|str) / recommendation(str) / problem_nature(str∈枚举).
  234. 场级/健康 finding 合法缺省 (非每条都有问题机); 仅当显式提供时核类型 — 防 producer 写脏 (与 §9.1.2 同源: 字段也要结构化)。
  235. """
  236. issues = []
  237. for idx, it in enumerate(findings or []):
  238. if not isinstance(it, dict):
  239. continue
  240. am = it.get('affected_machines')
  241. if am is not None and not isinstance(am, (list, str)):
  242. issues.append(f'finding[{idx}]: affected_machines 须 list/str (实为 {type(am).__name__})')
  243. rec = it.get('recommendation')
  244. if rec is not None and not isinstance(rec, str):
  245. issues.append(f'finding[{idx}]: recommendation 须 str')
  246. pn = it.get('problem_nature')
  247. if pn is not None and pn not in PROBLEM_NATURE:
  248. issues.append(f'finding[{idx}]: problem_nature "{pn}" ∉ {sorted(PROBLEM_NATURE)}')
  249. return issues
  250. # _OV_MACHINE_RE + validate_owner_view_consistency (业主版散文↔结构化一致性闸, RV-2 a) 已删 — 取消业主版 (2026-06-20), owner_view 已无消费者。
  251. def validate_conservation(findings, n_total, window_days=None, slack_days=2):
  252. """9.1.4 守恒校验 (docs/SOP.md §9.1.4): 异常机数≤总机数 / 结论窗⊆数据窗 / (有分项)总=Σ分项.
  253. findings: findings.json 的 findings 列表; n_total: 指纹场站总机数; window_days: 数据窗跨度天.
  254. 防"算术不闭合"造假 — 与 §5 D8 窗对齐互补 (D8 防窗错配, 此条防数算不闭合).
  255. """
  256. issues = []
  257. for idx, it in enumerate(findings or []):
  258. if not isinstance(it, dict):
  259. continue
  260. tag = f'finding[{idx}]({str(it.get("title", ""))[:18]})'
  261. cov = it.get('coverage') or {}
  262. if not isinstance(cov, dict): # 2026-09-07 实逮: 旧条目 coverage 为字符串 → 此处 .get 抛异常, PB-1 9.1.4 整段 fail-closed 报"解析失败", 真问题反而看不见
  263. issues.append(f'{tag}: coverage 非结构化 (str), 守恒无法核 (9.1.4)')
  264. continue
  265. nm = cov.get('n_machines')
  266. if isinstance(nm, (int, float)) and isinstance(n_total, (int, float)) and nm > n_total:
  267. issues.append(f'{tag}: 异常机数 {nm} > 场站总机 {n_total} → 守恒不闭合 (9.1.4)')
  268. wd = cov.get('window_days')
  269. if (isinstance(wd, (int, float)) and isinstance(window_days, (int, float))
  270. and window_days > 0 and wd > window_days + slack_days):
  271. issues.append(f'{tag}: 结论窗 {wd}天 > 数据窗 {window_days}天 → 时间不⊆数据 (9.1.4)')
  272. comp, tot = it.get('components'), it.get('total')
  273. if isinstance(comp, list) and comp and isinstance(tot, (int, float)):
  274. s = sum(c.get('value', 0) for c in comp if isinstance(c, dict))
  275. if abs(s - tot) > max(1e-6, abs(tot) * 0.01):
  276. issues.append(f'{tag}: 分项和 {s} ≠ 总 {tot} (9.1.4 守恒, 容差1%)')
  277. wf = it.get('waterfall')
  278. if wf is not None:
  279. issues.extend(validate_waterfall(wf, f'{tag}.waterfall'))
  280. return issues
  281. def validate_waterfall(wf, tag='waterfall'):
  282. """9.1.4 损失瀑布守恒 (docs/SOP.md §4.1 模块1.2b + §9.1.4): 让能量/¥账长牙.
  283. Σ分项 == total(容差1%) / 各项≥0(物理) / 恰一 actual(实发)项 / 残差打包必声明 basis /
  284. 若声明 economic_total: 必 == Σ(损失项 value × tariff) (防手填¥, 须 电量×电价 可复算).
  285. """
  286. issues = []
  287. if not isinstance(wf, dict):
  288. return [f'{tag}: 非 dict']
  289. total, comps = wf.get('total'), wf.get('components')
  290. if not isinstance(total, (int, float)):
  291. issues.append(f'{tag}: 缺 total 数值')
  292. if not isinstance(comps, list) or not comps:
  293. return issues + [f'{tag}: 缺 components 非空列表']
  294. s, actual_n = 0.0, 0
  295. for c in comps:
  296. if not isinstance(c, dict) or not isinstance(c.get('value'), (int, float)):
  297. issues.append(f'{tag}: 分项 {c} 缺数值 value')
  298. continue
  299. v = c['value']
  300. if v < 0:
  301. issues.append(f'{tag}: 分项 "{c.get("name")}" {v}<0 (实发/损失不可负, 物理)')
  302. s += v
  303. actual_n += 1 if c.get('kind') == 'actual' else 0
  304. if c.get('kind') == 'residual' and not c.get('basis'):
  305. issues.append(f'{tag}: 残差项 "{c.get("name")}" 无 basis → 禁硬拆打包残差 (§4.1 1.2b)')
  306. if isinstance(total, (int, float)) and abs(s - total) > max(1e-6, abs(total) * 0.01):
  307. issues.append(f'{tag}: 分项和 {round(s, 3)} ≠ total {total} (差 {round(s - total, 3)}) → 守恒不闭合 (9.1.4)')
  308. if actual_n != 1:
  309. issues.append(f'{tag}: 须恰 1 个 kind=actual(实发)项, 实为 {actual_n} (瀑布=实发+Σ损失)')
  310. econ = wf.get('economic_total')
  311. if isinstance(econ, (int, float)):
  312. e, ok = 0.0, True
  313. for c in comps:
  314. if isinstance(c, dict) and c.get('kind') == 'loss':
  315. v, tar = c.get('value'), c.get('tariff')
  316. if isinstance(v, (int, float)) and isinstance(tar, (int, float)):
  317. e += v * tar
  318. else:
  319. ok = False
  320. if not ok:
  321. issues.append(f'{tag}: 声明 economic_total 但损失项缺 tariff → ¥不可复算 (须电量×电价, 禁手填)')
  322. elif abs(e - econ) > max(1e-6, abs(econ) * 0.01):
  323. issues.append(f'{tag}: economic_total {econ} ≠ Σ(损失×电价) {round(e, 2)} → ¥手填嫌疑 (9.1.4 ¥派生闭合)')
  324. return issues
  325. def validate_repro(findings, require_for=('定论',)):
  326. """9.1.5 可复现脚本声明 (docs/SOP.md §9.1.5): verdict∈require_for 的 finding 必带 repro:{script, expected}.
  327. 双模型评审铁律: 仅"脚本存在"= theater (可 echo 假值绕过), 故 expected 期望值必填 (供 verify-repro 重跑比对).
  328. 本函数只校验声明完整性 (纯函数); 脚本实存 + 重跑数值比对在 report_integrity.verify_repro (有副作用, 不在此).
  329. """
  330. issues = []
  331. for idx, it in enumerate(findings or []):
  332. if not isinstance(it, dict):
  333. continue
  334. if norm_verdict(it.get('verdict', '')) not in require_for:
  335. continue
  336. tag = f'finding[{idx}]({str(it.get("title", ""))[:18]})'
  337. repro = it.get('repro')
  338. if not isinstance(repro, dict) or not repro.get('script'):
  339. issues.append(f'{tag}: verdict=定论 但缺 repro.script (9.1.5 定论必附可复现脚本)')
  340. continue
  341. if repro.get('expected') is None:
  342. issues.append(f'{tag}: repro 缺 expected 期望值 (9.1.5 防空壳脚本: 仅脚本存在可 echo 假值绕过, 须可重跑比对)')
  343. return issues
  344. # §0.8 ON-0 分析前置锁定门: analysis_lock.yaml 自洽闸 (机器位 scripts/analysis_lock_check.py)
  345. LOCK_REQUIRED = ['farm', 'frozen', 'windows', 'rated_kw', 'turbines', 'capacity_mw',
  346. 'sources', 'metrics', 'cp_ceiling', 'gates']
  347. _LOCK_TODO = {'todo', '待填', '待定', 'tbd', '?', ''}
  348. def validate_required_outputs(doc, produced):
  349. """§0.8 required_outputs 完整性自检 — 必算清单缺一报错 (解 "分析项时有时无" = V-019 数据层翻版).
  350. lock.required_outputs = 本次分析必产出项清单 (声明式, 写死在锁里)。produced = 实际产出项
  351. (list/set/dict, 来自运行 manifest 或产物目录扫描的文件名)。清单上每一项必在 produced 出现 (精确或子串命中),
  352. 缺一 → issue, **不允许静默跳过**。这把 V-019 "声称做了实际没做" 的防造假机制延伸到分析项层:
  353. 不是事后发现少了, 是清单强制它不能少。纯函数, 返回 issue 列表 (空=PASS)。
  354. """
  355. if not isinstance(doc, dict):
  356. return ['analysis_lock: 顶层非 dict']
  357. req = doc.get('required_outputs')
  358. if not req:
  359. return ['analysis_lock: 缺 required_outputs (必算清单) → 无法做完整性自检, '
  360. '"分析项时有时无" 无机器拦截 (§0.8/V-019 数据层)']
  361. if not isinstance(req, (list, tuple)):
  362. return [f'analysis_lock: required_outputs 须为列表, 实为 {type(req).__name__}']
  363. if isinstance(produced, dict):
  364. produced = list(produced.keys())
  365. prod = {str(p).strip() for p in (produced or [])}
  366. issues = []
  367. for item in req:
  368. key = (item.get('id') if isinstance(item, dict) else str(item)).strip()
  369. if not key:
  370. continue
  371. if not (key in prod or any(key in p for p in prod)):
  372. issues.append(f'analysis_lock·required_outputs: 必算项 "{key}" 未产出 '
  373. f'(缺一报错, 禁静默跳过; §0.8/V-019 数据层)')
  374. return issues
  375. def validate_analysis_lock(doc, cap_tol=0.005):
  376. """§0.8 ON-0 分析前置锁定门 — analysis_lock.yaml 自洽闸 (解 "同数据两次结果不一").
  377. 根因实证: 利用小时 1330 vs 2389 / CF 27% vs 15% / 改造前后 +0.3 vs +23.6 —
  378. 窗/口径/源/容量/基准未锁 → 同数据两次结果不一。本函数机器化 §0.8 B 自洽闸:
  379. A 冻结六项+frozen+gates 在场 (未冻结=不得出数)。
  380. ① Σ台=场: Σ rated_kw[turbines[t]] == capacity_mw.全场 (容差) — catches 27.5 vs 48.92 类装机口径错。
  381. ② 口径完整: metric 标 须标口径:true 必给区分口径值/注 (L4 含状态过滤+风速来源)。
  382. ③ INSUFFICIENT 一致: metrics/cp_ceiling 标 状态:INSUFFICIENT 的量必在 gates.INSUFFICIENT标记 列出 (防漂移)。
  383. ④ Betz: cp_ceiling 数值 Cp ≥0.593 = 物理违背 (#32)。
  384. 纯函数, 返回 issue 列表 (空=PASS)。report↔lock 口径交叉在 analysis_lock_check.py (有副作用, 不在此)。
  385. """
  386. if not isinstance(doc, dict):
  387. return ['analysis_lock: 顶层非 dict']
  388. issues = []
  389. for k in LOCK_REQUIRED:
  390. v = doc.get(k)
  391. if k not in doc or v is None or (isinstance(v, (str, list, dict)) and len(v) == 0):
  392. issues.append(f'analysis_lock: 冻结项缺/空: {k} (§0.8 A 六项+frozen+gates)')
  393. elif isinstance(v, str) and v.strip().lower() in _LOCK_TODO:
  394. issues.append(f'analysis_lock: {k}="{v}" 占位未填 → 未冻结不得出数 (§0.5/§0.8)')
  395. # ★ freeze 须用户裁决 (§0.8 freeze 权威=用户非Claude自冻; 解"没找我确认"洞, 2026-06-30 加)
  396. # 防: Claude 自拟+自冻+盖哈希=把临场决定伪装成有权威的契约。frozen 有值则 frozen_by 必含"用户裁决"。
  397. frz = doc.get('frozen')
  398. if frz is not None and not (isinstance(frz, str) and frz.strip().lower() in _LOCK_TODO):
  399. fb = str(doc.get('frozen_by') or '')
  400. if '用户裁决' not in fb:
  401. issues.append('analysis_lock: frozen 已设但 frozen_by 非"用户裁决" → 自冻锁(Claude/非用户裁决)'
  402. '不得当冻结锁出数 (§0.8 freeze 权威=用户; Claude自冻+哈希=临场决定伪装契约)')
  403. # ① Σ台=场 (§0.8 B①) — 装机口径错 → CF/利用小时全错
  404. rated, turb, cap = doc.get('rated_kw'), doc.get('turbines'), doc.get('capacity_mw')
  405. if isinstance(rated, dict) and isinstance(turb, dict) and isinstance(cap, dict):
  406. unknown = sorted({str(m) for m in turb.values() if m not in rated})
  407. if unknown:
  408. issues.append(f'analysis_lock①Σ台=场: 机型 {unknown} 在 turbines 用到但 rated_kw 未定义')
  409. elif isinstance(cap.get('全场'), (int, float)):
  410. sum_mw = sum(float(rated[m]) for m in turb.values()) / 1000.0
  411. tot = float(cap['全场'])
  412. if abs(sum_mw - tot) > max(cap_tol, tot * cap_tol):
  413. issues.append(f'analysis_lock①Σ台=场不闭合: Σ台 {sum_mw:.3f}MW ≠ capacity_mw.全场 {tot}MW '
  414. f'(差 {sum_mw - tot:+.3f}; §0.8 B① 装机口径错→CF/利用小时全错)')
  415. # ② 口径完整 (§0.8 L4)
  416. metrics = doc.get('metrics') if isinstance(doc.get('metrics'), dict) else {}
  417. for mn, mv in metrics.items():
  418. if isinstance(mv, dict) and mv.get('须标口径') is True:
  419. extra = [x for x in mv if x not in ('须标口径', '定义', '注', '值')]
  420. if len(extra) < 2 and not mv.get('注'):
  421. issues.append(f'analysis_lock②口径: "{mn}" 标须标口径 却未给区分口径值/注 (L4 须含状态过滤+风速来源)')
  422. # ③ INSUFFICIENT 一致 (防 一处诚实留口、另一处被当定论引用)
  423. gates = doc.get('gates') if isinstance(doc.get('gates'), dict) else {}
  424. declared = json.dumps(gates.get('INSUFFICIENT标记') or [], ensure_ascii=False)
  425. cpc = doc.get('cp_ceiling') if isinstance(doc.get('cp_ceiling'), dict) else {}
  426. for src_name, src in (('metrics', metrics), ('cp_ceiling', cpc)):
  427. for mn, mv in (src.items() if isinstance(src, dict) else []):
  428. if isinstance(mv, dict) and str(mv.get('状态', '')).upper() == 'INSUFFICIENT' and str(mn) not in declared:
  429. issues.append(f'analysis_lock③: {src_name}."{mn}" 标 状态:INSUFFICIENT 但未在 gates.INSUFFICIENT标记 列出 → 口径漂移风险')
  430. # ④ Betz
  431. for k, v in (cpc.items() if isinstance(cpc, dict) else []):
  432. if isinstance(v, (int, float)) and 'cp' in str(k).lower() and v >= 0.593:
  433. issues.append(f'analysis_lock④Betz: cp_ceiling.{k}={v} ≥0.593 物理不可能 (#32 Cp>Betz 铁证)')
  434. return issues
  435. # ============================================================================
  436. # 附录A Case 状态机 (docs/运维Agent转化评估_v0.1.md 附录A → 代码化; 运维 agent 柱1「智能工单」)
  437. # ----------------------------------------------------------------------------
  438. # findings.json = 每批全量快照(无记性, B16 前兆每月被"重新发现"); case = 跨批次状态延续 +
  439. # 反向闭环(evidence→ticket→verify→CLOSED/RELAPSED→回灌 记分卡/基准库/误报台账)。
  440. # 本节把附录A 10 态状态机 + 6 条机器不变量落成纯函数闸(风格同上: 返 issue 列表, 空=PASS)。
  441. # 工单卡(TICKETED 载体)/回验卡(VERIFY 期)= case 生命周期的两个切面, 各自 schema 在此定义。
  442. # 正交两轴不进 state (附录A.2): tier(T1/T2/T3 紧迫度) / verdict(§0.2 证据强度) = case 属性。
  443. # ============================================================================
  444. CASE_STATES = {'NORMAL', 'WATCH', 'CANDIDATE', 'CONFIRMED', 'TICKETED',
  445. 'IN_REPAIR', 'VERIFY', 'CLOSED', 'REFUTED', 'RELAPSED'}
  446. # 合法状态迁移边 (附录A.2; 非法跳转拒绝 = 不变量⑤)
  447. CASE_EDGES = {
  448. 'NORMAL': {'WATCH'},
  449. 'WATCH': {'CANDIDATE', 'NORMAL'},
  450. 'CANDIDATE': {'CONFIRMED', 'NORMAL', 'REFUTED'},
  451. 'CONFIRMED': {'TICKETED', 'REFUTED'},
  452. 'TICKETED': {'IN_REPAIR', 'VERIFY'},
  453. 'IN_REPAIR': {'VERIFY'},
  454. 'VERIFY': {'CLOSED', 'CONFIRMED'}, # verify_fail → 回 CONFIRMED (未根治, 变桨 SOP A014 反弹式)
  455. 'CLOSED': {'RELAPSED'},
  456. 'REFUTED': set(), # 终态 (留档防再登顶)
  457. 'RELAPSED': {'CONFIRMED'},
  458. }
  459. # 确诊过之后 own-data 证据须留存 (不变量①: 禁 fleet-z 直升终判)
  460. CASE_CONFIRMED_OR_BEYOND = {'CONFIRMED', 'TICKETED', 'IN_REPAIR', 'VERIFY', 'CLOSED', 'RELAPSED'}
  461. # 工单已签发之后 (不变量②: 人工裁决点①机器不可绕)
  462. CASE_TICKETED_OR_BEYOND = {'TICKETED', 'IN_REPAIR', 'VERIFY', 'CLOSED', 'RELAPSED'}
  463. CASE_TERMINAL = {'CLOSED', 'REFUTED'} # 非终态判据 (附录A.4⑥ 至多一个活跃档)
  464. CASE_REQUIRED = ['case_id', 'farm', 'turbine', 'system', 'state', 'opened_at', 'updated_at']
  465. CASE_TIERS = {None, 'T1', 'T2', 'T3'}
  466. def validate_ticket_card(t, tag='ticket'):
  467. """工单卡 schema (附录A.3 ticket + P4 七段结构化): TICKETED 载体.
  468. 必: ticket_id / issued_by / actions(非空, = P4 §3 排查步骤). 可选: window_suggested(B1 排程建议) /
  469. parts / priority(P0/P1/P2). 无动作的工单 = 糊涂账 (P4 硬纪律)。"""
  470. if not isinstance(t, dict):
  471. return [f'{tag}: 非 dict']
  472. issues = []
  473. for k in ('ticket_id', 'issued_by'):
  474. if not t.get(k):
  475. issues.append(f'{tag}: 缺 {k}')
  476. acts = t.get('actions')
  477. if not isinstance(acts, list) or not acts:
  478. issues.append(f'{tag}: actions 须非空列表 (P4 §3 排查步骤; 无动作的工单=糊涂账)')
  479. pr = t.get('priority')
  480. if pr is not None and pr not in ('P0', 'P1', 'P2'):
  481. issues.append(f'{tag}: priority "{pr}" ∉ {{P0,P1,P2}} (P4 优先级)')
  482. parts = t.get('parts')
  483. if parts is not None and not isinstance(parts, list):
  484. issues.append(f'{tag}: parts 须 list')
  485. return issues
  486. def validate_verify_card(vf, tag='verify'):
  487. """回验卡 schema (附录A.3 verify): VERIFY 期量化闭环 = 反向闭环的一半价值 (P4 §7).
  488. 必: baseline(修前判据量) / target(回归带) / result∈{pass,fail,pending}.
  489. result=pass → readings 须非空 (无实测读数不得判 pass; 防"修完就是修好"糊涂账)。"""
  490. if not isinstance(vf, dict):
  491. return [f'{tag}: 非 dict']
  492. issues = []
  493. for k in ('baseline', 'target'):
  494. if vf.get(k) is None:
  495. issues.append(f'{tag}: 缺 {k} (修前基线/回归带; 无基线无法判回归)')
  496. res = vf.get('result')
  497. if res not in ('pass', 'fail', 'pending'):
  498. issues.append(f'{tag}: result "{res}" ∉ {{pass,fail,pending}}')
  499. rd = vf.get('readings')
  500. if res == 'pass' and (not isinstance(rd, list) or not rd):
  501. issues.append(f'{tag}: result=pass 但 readings 空 → 无实测读数不得判回归 (P4 §7 回验量化)')
  502. wd = vf.get('window_days')
  503. if wd is not None and not isinstance(wd, (int, float)):
  504. issues.append(f'{tag}: window_days 须数值')
  505. return issues
  506. def validate_analysis(an, tag='analysis'):
  507. """深挖层 schema (Phase-2 机制归因 + 独立审): case.analysis 可选扩展。
  508. 必: verdict(六枚举) + mechanism(机制, 无机制=空 verdict). 可选: review/field_needs/ref/date/by.
  509. ★纪律: analysis.verdict 是更深表征(可 > case.verdict, 如 D29 案候选→深挖准定论·预警),
  510. 但**不改 case.state** — 升 CONFIRMED 须 own-data(不变量①), 走 apply_human_action。"""
  511. if not isinstance(an, dict):
  512. return [f'{tag}: 非 dict']
  513. issues = []
  514. if norm_verdict(an.get('verdict', '')) not in VERDICTS:
  515. issues.append(f'{tag}: verdict "{an.get("verdict")}" ∉ 六枚举')
  516. if not an.get('mechanism'):
  517. issues.append(f'{tag}: 缺 mechanism (深挖必给机制归因, 否则=空 verdict)')
  518. return issues
  519. def validate_case(case, tag='case'):
  520. """附录A Case 状态机单档校验 (纯函数, 空=PASS). 6 条机器不变量 = 附录A.4.
  521. 注: 不变量⑤"追加式禁改写"的 append-only 语义无法从单份快照证 (需跨版本 diff, 在有副作用的
  522. 回灌器里查); 本函数证其可证部分 = 迁移边合法 + state↔审计链一致 + 结构完整。"""
  523. if not isinstance(case, dict):
  524. return [f'{tag}: 顶层非 dict']
  525. cid = case.get('case_id', '?')
  526. tg = f'{tag}[{cid}]'
  527. issues = []
  528. # 字段在场 (附录A.3)
  529. for k in CASE_REQUIRED:
  530. if not case.get(k):
  531. issues.append(f'{tg}: 缺 {k}')
  532. state = case.get('state')
  533. if state is not None and state not in CASE_STATES:
  534. issues.append(f'{tg}: state "{state}" ∉ 十态 {sorted(CASE_STATES)}')
  535. return issues # state 非法, 后续不变量无从判
  536. # 正交轴类型 (附录A.2: 紧迫度/证据强度 ≠ 状态)
  537. if case.get('tier') not in CASE_TIERS:
  538. issues.append(f'{tg}: tier "{case.get("tier")}" ∉ {{None,T1,T2,T3}} (紧迫度≠状态, 正交轴)')
  539. v = case.get('verdict')
  540. if v is not None and norm_verdict(v) not in VERDICTS:
  541. issues.append(f'{tg}: verdict "{v}" ∉ 六枚举 (证据强度≠状态, 正交轴)')
  542. # evidence 追加流水 (附录A.3): discriminator_id 外键 + verdict 合法
  543. ev = case.get('evidence') or []
  544. if not isinstance(ev, list):
  545. issues.append(f'{tg}: evidence 须 list')
  546. ev = []
  547. for j, e in enumerate(ev):
  548. if not isinstance(e, dict):
  549. issues.append(f'{tg}: evidence[{j}] 非 dict')
  550. continue
  551. if not e.get('discriminator_id'):
  552. issues.append(f'{tg}: evidence[{j}] 缺 discriminator_id (注册表外键, 判据溯源)')
  553. ev_v = e.get('verdict')
  554. if ev_v is not None and norm_verdict(ev_v) not in VERDICTS:
  555. issues.append(f'{tg}: evidence[{j}] verdict "{ev_v}" ∉ 六枚举')
  556. # history append-only 审计链: 每条迁移边合法 (不变量⑤可证部分) + by 枚举
  557. hist = case.get('history') or []
  558. if not isinstance(hist, list):
  559. issues.append(f'{tg}: history 须 list')
  560. hist = []
  561. for j, h in enumerate(hist):
  562. if not isinstance(h, dict):
  563. issues.append(f'{tg}: history[{j}] 非 dict')
  564. continue
  565. fr, to, by = h.get('from_state'), h.get('to_state'), h.get('by')
  566. if by not in ('engine', 'human'):
  567. issues.append(f'{tg}: history[{j}] by "{by}" ∉ {{engine,human}}')
  568. if fr not in CASE_STATES or to not in CASE_STATES:
  569. issues.append(f'{tg}: history[{j}] from/to 非十态 ({fr}→{to})')
  570. elif to not in CASE_EDGES.get(fr, set()):
  571. issues.append(f'{tg}: history[{j}] 非法迁移 {fr}→{to} (附录A.2 未定义边; 不变量⑤)')
  572. if hist and isinstance(hist[-1], dict) and hist[-1].get('to_state') and state:
  573. if hist[-1]['to_state'] != state:
  574. issues.append(f'{tg}: state={state} ≠ history 末态 {hist[-1]["to_state"]} (审计链不一致)')
  575. # ── 6 条机器不变量 (附录A.4) ──
  576. # ① CONFIRMED 及以后 → evidence 含 ≥1 own_data 级 (禁 fleet-z 直升终判, 验证金字塔③)
  577. if state in CASE_CONFIRMED_OR_BEYOND and not any(
  578. isinstance(e, dict) and e.get('own_data') for e in ev):
  579. issues.append(f'{tg}①: state={state} 但 evidence 无 own_data 级判据 → 禁 fleet-z 直升 (金字塔③终判)')
  580. # ② TICKETED 及以后 → history 含 by=human 的 ticket_issued + 工单卡在场合法 (人工裁决点①机器不可绕)
  581. if state in CASE_TICKETED_OR_BEYOND:
  582. if not any(isinstance(h, dict) and h.get('by') == 'human'
  583. and h.get('event') == 'ticket_issued' for h in hist):
  584. issues.append(f'{tg}②: state={state} 但 history 无 by=human 的 ticket_issued (人工裁决点①机器不可绕)')
  585. issues += validate_ticket_card(case.get('ticket') or {}, f'{tg}.ticket')
  586. # ③ CLOSED → verify.result=pass (无回验不销案; 昌邑 A27 悬案按构造杜绝)
  587. if state == 'CLOSED':
  588. vfr = (case.get('verify') or {}).get('result')
  589. if vfr != 'pass':
  590. issues.append(f'{tg}③: state=CLOSED 但 verify.result≠pass ("{vfr}") → 无回验不销案 (附录A.4③)')
  591. if state in ('VERIFY', 'CLOSED'):
  592. issues += validate_verify_card(case.get('verify') or {}, f'{tg}.verify')
  593. # ④ REFUTED → refuted_basis 非空 (防同信号再登顶)
  594. if state == 'REFUTED' and not case.get('refuted_basis'):
  595. issues.append(f'{tg}④: state=REFUTED 但 refuted_basis 空 → 防同信号再登顶 (附录A.4④)')
  596. # 深挖层 (可选扩展): case.analysis 在场则校验 schema (Phase-2 机制 + 独立审)
  597. if case.get('analysis') is not None:
  598. issues += validate_analysis(case['analysis'], f'{tg}.analysis')
  599. # ⑥ RELAPSED → relapse_of 指原 case (D7 复发率数据源; 跨档"至多一活跃"在 validate_cases)
  600. if state == 'RELAPSED' and not case.get('relapse_of'):
  601. issues.append(f'{tg}⑥: state=RELAPSED 但缺 relapse_of (指原 case; D7 复发率数据源)')
  602. return issues
  603. def validate_cases(cases, tag='cases'):
  604. """整份 cases.json 跨档校验 (附录A.4⑥): 逐档 validate_case + case_id 唯一
  605. + 同 (farm,turbine,system) 至多一个非终态 case (防重复开案)."""
  606. if not isinstance(cases, list):
  607. return [f'{tag}: 顶层须 list']
  608. issues, seen_id, active = [], set(), {}
  609. for c in cases:
  610. issues += validate_case(c)
  611. if not isinstance(c, dict):
  612. continue
  613. cid = c.get('case_id')
  614. if cid in seen_id:
  615. issues.append(f'{tag}: case_id 重复 {cid}')
  616. seen_id.add(cid)
  617. if c.get('state') not in CASE_TERMINAL:
  618. key = (c.get('farm'), c.get('turbine'), c.get('system'))
  619. if key in active:
  620. issues.append(f'{tag}: {key} 同时 >1 非终态 case ({active[key]} + {cid}) → 防重复开案 (附录A.4⑥)')
  621. active[key] = cid
  622. return issues