cross_review.py 5.3 KB

123456789101112131415161718192021222324252627282930313233343536373839404142434445464748495051525354555657585960616263646566676869707172737475767778798081828384858687888990919293949596979899100101102103104105106107108109110111112
  1. # -*- coding: utf-8 -*-
  2. """DeepSeek 审千问: 本地跨模型审核 v1 (2026-08-24 用户令 "建立对千问的审核机制").
  3. 分层: ① citation_gate 确定性引文回抓闸 — 治本地审核模型的"引用失实"病 (14b/32b 各 1 例实逮);
  4. ② r1 审核员只出质疑清单, 不出裁决 (推理无界、结论受闸);
  5. ③ 审核员的每条质疑自身也过 ① — 摘录在被审文本中找不到 → 该条自动废弃并标记.
  6. 能力边界 (gold 实测 2/8): 本地审 = 快筛层; 审级定谳只认云端双模审."""
  7. import json, re, time, urllib.request
  8. URL = 'http://localhost:11434'
  9. REVIEWER_PREF = ['deepseek-r1:14b', 'deepseek-r1:32b']
  10. RULES = ('①答案中的台号/关键数字必须在【事实】里有出处 ②等级词(优秀/良好/报警/危险/不可判)须与事实一致, 无擅自升降级 '
  11. '③结论方向须与证据方向一致 ④【事实】未提的部件/时间/机制不得出现断言')
  12. PROMPT = """你是独立审核员, 审【答案】是否忠实于【事实】。规则: {rules}
  13. 只输出 JSON, 不要任何其他文字: {{"issues": [{{"quote": "答案原文摘录(逐字)", "problem": "一句话问题", "rule": "①-④"}}], "verdict": "PASS 或 N条质疑"}}
  14. 没有问题就输出 {{"issues": [], "verdict": "PASS"}}。
  15. 【问题】{q}
  16. 【事实(工具产出, 权威)】{facts}
  17. 【答案(被审)】{answer}
  18. """
  19. def _turbs(text):
  20. out = set()
  21. for a, b, c in re.findall(r'WTG-?(\d{1,2})|(\d{1,2})#|(\d{1,2})号机?', str(text)):
  22. out.add(int(a or b or c))
  23. return out
  24. def _nums(text):
  25. return set(re.findall(r'\d+\.?\d*', str(text)))
  26. def _strip_labels(text):
  27. """剔非断言型数字: 窗名 wMMDD / ISO 日期 / 版本号 — 它们是标签不是数值断言.
  28. (2026-08-25 实逮: 30b 答案里 w0127/w0707/w0724/w0805 被判'数字无出处' 8 处全是窗名假阳)"""
  29. s = str(text)
  30. s = re.sub(r'[wW]\d{4}\b', ' ', s) # 窗名 w0707
  31. s = re.sub(r'\b\d{4}-\d{2}(-\d{2})?\b', ' ', s) # ISO 日期
  32. s = re.sub(r'\bv\d+(\.\d+)*\b', ' ', s) # 版本号 v5 / v2.0
  33. return s
  34. def citation_gate(claim, source, num_min_digits=2):
  35. """claim 里引用的台号/数字必须在 source 有出处; 返回失实清单 (确定性, 零模型)."""
  36. bad = []
  37. claim_s, source_s = _strip_labels(claim), _strip_labels(source)
  38. st, sn = _turbs(source), _nums(source_s)
  39. norm = lambda x: x.rstrip('0').rstrip('.') if '.' in x else x
  40. sn_n = sn | {norm(x) for x in sn} | {x.split('.')[0] for x in sn}
  41. for t in _turbs(claim):
  42. if t not in st:
  43. bad.append(f'台号 {t}# 无出处')
  44. for n in _nums(claim_s):
  45. if len(n.replace('.', '')) < num_min_digits or n in sn_n:
  46. continue
  47. if norm(n) in sn_n or n.split('.')[0] in sn_n:
  48. continue
  49. bad.append(f'数字 {n} 无出处')
  50. return bad
  51. def _chat(prompt, model, npred=800, timeout=180):
  52. body = {'model': model, 'messages': [{'role': 'user', 'content': prompt}], 'stream': False,
  53. 'think': True, 'keep_alive': '30m', 'options': {'num_predict': npred, 'num_ctx': 16384}}
  54. req = urllib.request.Request(URL + '/api/chat', data=json.dumps(body).encode('utf-8'), headers={'Content-Type': 'application/json'})
  55. d = json.loads(urllib.request.urlopen(req, timeout=timeout).read())
  56. return (d.get('message', {}) or {}).get('content', '')
  57. def pick_reviewer():
  58. try:
  59. req = urllib.request.Request(URL + '/api/tags')
  60. tags = json.loads(urllib.request.urlopen(req, timeout=4).read())
  61. names = {m['name'] for m in tags.get('models', [])}
  62. for p in REVIEWER_PREF:
  63. if p in names:
  64. return p
  65. except Exception:
  66. pass
  67. return None
  68. def review_answer(question, answer, facts, model=None):
  69. """跨模型审核: 机器闸 + r1 质疑清单 (质疑自过闸). 返回 dict, 不抛异常."""
  70. t0 = time.time()
  71. out = dict(machine_flags=citation_gate(answer, facts), issues=[], discarded=[], verdict='', model=None, seconds=0)
  72. model = model or pick_reviewer()
  73. if not model:
  74. out['verdict'] = ('机器闸: ' + '; '.join(out['machine_flags'])) if out['machine_flags'] else 'PASS(仅机器闸, 审核模型不在线)'
  75. return out
  76. out['model'] = model
  77. try:
  78. txt = _chat(PROMPT.format(rules=RULES, q=str(question)[:500], facts=str(facts)[:6000], answer=str(answer)[:4000]), model)
  79. m = re.search(r'\{.*\}', txt, flags=re.S)
  80. r = json.loads(m.group(0)) if m else {}
  81. norm = lambda s: re.sub(r'\s+', '', str(s))
  82. na = norm(answer)
  83. for it in (r.get('issues') or []):
  84. q_ = norm(it.get('quote', ''))
  85. # ③ 审核员摘录必须真在答案里 + 质疑正文自身过引文闸
  86. if q_ and q_[:40] in na and not citation_gate(it.get('problem', ''), str(answer) + str(facts)):
  87. out['issues'].append(it)
  88. else:
  89. out['discarded'].append(dict(it, reason='引用失实(摘录不在答案中或质疑引了不存在的台号/数字)'))
  90. n = len(out['issues']) + len(out['machine_flags'])
  91. out['verdict'] = 'PASS' if n == 0 else f'{n} 条质疑'
  92. except Exception as e:
  93. out['verdict'] = f'审核失败: {e}'
  94. out['seconds'] = round(time.time() - t0, 1)
  95. return out