audit.py 2.5 KB

123456789101112131415161718192021222324252627282930313233343536373839404142434445464748495051525354
  1. # -*- coding: utf-8 -*-
  2. """windscada 审级件 (统一技术栈 L3; 借 windcms cross_review 模式, 独立拷贝防运行时耦合).
  3. 引文回抓闸: 答案/审核输出里的台号与数字必须在权威源里有出处 — 治'引用失实'(14b/32b 各1例实逮的病)."""
  4. import re
  5. def _turbs(text):
  6. out = set()
  7. for a, b, c in re.findall(r'WTG-?(\d{1,2})|(\d{1,2})#|(\d{1,2})号机?', str(text)):
  8. out.add(int(a or b or c))
  9. return out
  10. def _nums(text):
  11. return set(re.findall(r'\d+\.?\d*', str(text)))
  12. def _strip_labels(text):
  13. """剔非断言型数字: 窗名 wMMDD / ISO 日期 / 版本号 — 它们是标签不是数值断言.
  14. (2026-08-25 实逮: 30b 答案里 w0127/w0707/w0724/w0805 被判'数字无出处' 8 处全是窗名假阳)"""
  15. s = str(text)
  16. s = re.sub(r'[wW]\d{4}\b', ' ', s) # 窗名 w0707
  17. s = re.sub(r'\b\d{4}-\d{2}(-\d{2})?\b', ' ', s) # ISO 日期
  18. s = re.sub(r'\bv\d+(\.\d+)*\b', ' ', s) # 版本号 v5 / v2.0
  19. return s
  20. def citation_gate(claim, source, num_min_digits=2):
  21. bad = []
  22. # ★先去千分位逗号 (2026-08-28 实测假阳性): 答案里的 "301,512.0" 会被数字提取切成
  23. # "301" 与 "512.0" 两个数, 而证据里写的是 "301512.0" ⇒ 两个碎片都判无出处。
  24. # 模型按中文习惯给数字加千分位是正常排版, 不该因此被判编造。
  25. import re as _re
  26. _decomma = lambda t: _re.sub(r'(?<=\d),(?=\d{3}\b)', '', t or '')
  27. claim, source = _decomma(claim), _decomma(source)
  28. claim_s, source_s = _strip_labels(claim), _strip_labels(source)
  29. st, sn = _turbs(source), _nums(source_s)
  30. norm = lambda x: x.rstrip('0').rstrip('.') if '.' in x else x
  31. sn_n = sn | {norm(x) for x in sn} | {x.split('.')[0] for x in sn}
  32. for t in _turbs(claim):
  33. if t not in st:
  34. bad.append(f'台号 {t}# 无出处')
  35. for n in _nums(claim_s):
  36. if len(n.replace('.', '')) < num_min_digits or n in sn_n:
  37. continue
  38. if norm(n) in sn_n or n.split('.')[0] in sn_n:
  39. continue
  40. bad.append(f'数字 {n} 无出处')
  41. return bad
  42. def grounding(answer, facts):
  43. """MCP claim_check 用: (ok, {turbines, numbers}) 形式包装 citation_gate 同一判据."""
  44. bad = citation_gate(answer, facts)
  45. bad_t = sorted({int(x.split()[1].rstrip('#')) for x in bad if x.startswith('台号')})
  46. bad_n = [x.split()[1] for x in bad if x.startswith('数字')][:12]
  47. return (not bad), {'turbines': bad_t, 'numbers': bad_n}