# -*- coding: utf-8 -*- """DeepSeek 审千问: 本地跨模型审核 v1 (2026-08-24 用户令 "建立对千问的审核机制"). 分层: ① citation_gate 确定性引文回抓闸 — 治本地审核模型的"引用失实"病 (14b/32b 各 1 例实逮); ② r1 审核员只出质疑清单, 不出裁决 (推理无界、结论受闸); ③ 审核员的每条质疑自身也过 ① — 摘录在被审文本中找不到 → 该条自动废弃并标记. 能力边界 (gold 实测 2/8): 本地审 = 快筛层; 审级定谳只认云端双模审.""" import json, re, time, urllib.request URL = 'http://localhost:11434' REVIEWER_PREF = ['deepseek-r1:14b', 'deepseek-r1:32b'] RULES = ('①答案中的台号/关键数字必须在【事实】里有出处 ②等级词(优秀/良好/报警/危险/不可判)须与事实一致, 无擅自升降级 ' '③结论方向须与证据方向一致 ④【事实】未提的部件/时间/机制不得出现断言') PROMPT = """你是独立审核员, 审【答案】是否忠实于【事实】。规则: {rules} 只输出 JSON, 不要任何其他文字: {{"issues": [{{"quote": "答案原文摘录(逐字)", "problem": "一句话问题", "rule": "①-④"}}], "verdict": "PASS 或 N条质疑"}} 没有问题就输出 {{"issues": [], "verdict": "PASS"}}。 【问题】{q} 【事实(工具产出, 权威)】{facts} 【答案(被审)】{answer} """ def _turbs(text): out = set() for a, b, c in re.findall(r'WTG-?(\d{1,2})|(\d{1,2})#|(\d{1,2})号机?', str(text)): out.add(int(a or b or c)) return out def _nums(text): return set(re.findall(r'\d+\.?\d*', str(text))) def _strip_labels(text): """剔非断言型数字: 窗名 wMMDD / ISO 日期 / 版本号 — 它们是标签不是数值断言. (2026-08-25 实逮: 30b 答案里 w0127/w0707/w0724/w0805 被判'数字无出处' 8 处全是窗名假阳)""" s = str(text) s = re.sub(r'[wW]\d{4}\b', ' ', s) # 窗名 w0707 s = re.sub(r'\b\d{4}-\d{2}(-\d{2})?\b', ' ', s) # ISO 日期 s = re.sub(r'\bv\d+(\.\d+)*\b', ' ', s) # 版本号 v5 / v2.0 return s def citation_gate(claim, source, num_min_digits=2): """claim 里引用的台号/数字必须在 source 有出处; 返回失实清单 (确定性, 零模型).""" bad = [] claim_s, source_s = _strip_labels(claim), _strip_labels(source) st, sn = _turbs(source), _nums(source_s) norm = lambda x: x.rstrip('0').rstrip('.') if '.' in x else x sn_n = sn | {norm(x) for x in sn} | {x.split('.')[0] for x in sn} for t in _turbs(claim): if t not in st: bad.append(f'台号 {t}# 无出处') for n in _nums(claim_s): if len(n.replace('.', '')) < num_min_digits or n in sn_n: continue if norm(n) in sn_n or n.split('.')[0] in sn_n: continue bad.append(f'数字 {n} 无出处') return bad def _chat(prompt, model, npred=800, timeout=180): body = {'model': model, 'messages': [{'role': 'user', 'content': prompt}], 'stream': False, 'think': True, 'keep_alive': '30m', 'options': {'num_predict': npred, 'num_ctx': 16384}} req = urllib.request.Request(URL + '/api/chat', data=json.dumps(body).encode('utf-8'), headers={'Content-Type': 'application/json'}) d = json.loads(urllib.request.urlopen(req, timeout=timeout).read()) return (d.get('message', {}) or {}).get('content', '') def pick_reviewer(): try: req = urllib.request.Request(URL + '/api/tags') tags = json.loads(urllib.request.urlopen(req, timeout=4).read()) names = {m['name'] for m in tags.get('models', [])} for p in REVIEWER_PREF: if p in names: return p except Exception: pass return None def review_answer(question, answer, facts, model=None): """跨模型审核: 机器闸 + r1 质疑清单 (质疑自过闸). 返回 dict, 不抛异常.""" t0 = time.time() out = dict(machine_flags=citation_gate(answer, facts), issues=[], discarded=[], verdict='', model=None, seconds=0) model = model or pick_reviewer() if not model: out['verdict'] = ('机器闸: ' + '; '.join(out['machine_flags'])) if out['machine_flags'] else 'PASS(仅机器闸, 审核模型不在线)' return out out['model'] = model try: txt = _chat(PROMPT.format(rules=RULES, q=str(question)[:500], facts=str(facts)[:6000], answer=str(answer)[:4000]), model) m = re.search(r'\{.*\}', txt, flags=re.S) r = json.loads(m.group(0)) if m else {} norm = lambda s: re.sub(r'\s+', '', str(s)) na = norm(answer) for it in (r.get('issues') or []): q_ = norm(it.get('quote', '')) # ③ 审核员摘录必须真在答案里 + 质疑正文自身过引文闸 if q_ and q_[:40] in na and not citation_gate(it.get('problem', ''), str(answer) + str(facts)): out['issues'].append(it) else: out['discarded'].append(dict(it, reason='引用失实(摘录不在答案中或质疑引了不存在的台号/数字)')) n = len(out['issues']) + len(out['machine_flags']) out['verdict'] = 'PASS' if n == 0 else f'{n} 条质疑' except Exception as e: out['verdict'] = f'审核失败: {e}' out['seconds'] = round(time.time() - t0, 1) return out