# -*- coding: utf-8 -*- """windscada 审级件 (统一技术栈 L3; 借 windcms cross_review 模式, 独立拷贝防运行时耦合). 引文回抓闸: 答案/审核输出里的台号与数字必须在权威源里有出处 — 治'引用失实'(14b/32b 各1例实逮的病).""" import re def _turbs(text): out = set() for a, b, c in re.findall(r'WTG-?(\d{1,2})|(\d{1,2})#|(\d{1,2})号机?', str(text)): out.add(int(a or b or c)) return out def _nums(text): return set(re.findall(r'\d+\.?\d*', str(text))) def _strip_labels(text): """剔非断言型数字: 窗名 wMMDD / ISO 日期 / 版本号 — 它们是标签不是数值断言. (2026-08-25 实逮: 30b 答案里 w0127/w0707/w0724/w0805 被判'数字无出处' 8 处全是窗名假阳)""" s = str(text) s = re.sub(r'[wW]\d{4}\b', ' ', s) # 窗名 w0707 s = re.sub(r'\b\d{4}-\d{2}(-\d{2})?\b', ' ', s) # ISO 日期 s = re.sub(r'\bv\d+(\.\d+)*\b', ' ', s) # 版本号 v5 / v2.0 return s def citation_gate(claim, source, num_min_digits=2): bad = [] # ★先去千分位逗号 (2026-08-28 实测假阳性): 答案里的 "301,512.0" 会被数字提取切成 # "301" 与 "512.0" 两个数, 而证据里写的是 "301512.0" ⇒ 两个碎片都判无出处。 # 模型按中文习惯给数字加千分位是正常排版, 不该因此被判编造。 import re as _re _decomma = lambda t: _re.sub(r'(?<=\d),(?=\d{3}\b)', '', t or '') claim, source = _decomma(claim), _decomma(source) claim_s, source_s = _strip_labels(claim), _strip_labels(source) st, sn = _turbs(source), _nums(source_s) norm = lambda x: x.rstrip('0').rstrip('.') if '.' in x else x sn_n = sn | {norm(x) for x in sn} | {x.split('.')[0] for x in sn} for t in _turbs(claim): if t not in st: bad.append(f'台号 {t}# 无出处') for n in _nums(claim_s): if len(n.replace('.', '')) < num_min_digits or n in sn_n: continue if norm(n) in sn_n or n.split('.')[0] in sn_n: continue bad.append(f'数字 {n} 无出处') return bad def grounding(answer, facts): """MCP claim_check 用: (ok, {turbines, numbers}) 形式包装 citation_gate 同一判据.""" bad = citation_gate(answer, facts) bad_t = sorted({int(x.split()[1].rstrip('#')) for x in bad if x.startswith('台号')}) bad_n = [x.split()[1] for x in bad if x.startswith('数字')][:12] return (not bad), {'turbines': bad_t, 'numbers': bad_n}