llm.py 4.8 KB

123456789101112131415161718192021222324252627282930313233343536373839404142434445464748495051525354555657585960616263646566676869707172737475767778798081828384858687888990919293949596979899100101
  1. # -*- coding: utf-8 -*-
  2. """离线大模型客户端 (ollama, 模型按名可换: qwen3:32b / deepseek-r1:32b ...) + 接地闸 (台号 + 数字必须来自事实上下文)."""
  3. import json, re, urllib.request
  4. URL = 'http://localhost:11434'
  5. PREFER = ['qwen3:32b', 'qwen3:30b-a3b', 'qwen3:14b', 'qwen3:8b', 'qwen2.5:14b', 'qwen2.5:7b', 'deepseek-r1:32b', 'deepseek-r1:14b']
  6. # 交互路径优先小模型 (2026-08-24 A/B: 8b 对比题 4s/诊断题 27s vs 32b 8s/36s, 两题结论与数字全对; 用户批"模型可以降级");
  7. # 报告润色/审核仍走 PREFER (32b 优先, 批处理质量优先)
  8. PREFER_FAST = ['qwen3:8b', 'qwen3:14b', 'qwen3:30b-a3b', 'qwen3:32b', 'qwen2.5:7b']
  9. # think:false 失效模型 (恒思考版, 如 Qwen3-30B-A3B Thinking-2507 系): think:false 只关 ollama 的
  10. # thinking/content 分离、关不掉模型思考 → 思考文本整段灌 content (2026-08-26 实测 chat/generate 两端点同病,
  11. # /no_think 软开关同无效; 对照 qwen3:8b 两端点均正常). 修法 = 强制 think:true 拿分离, 取 content 丢 thinking;
  12. # 调用侧对被强制场景放大 num_predict (thinking 先吃预算, 实测 400 tok 全被吃光 content 为空).
  13. THINK_LOCKED = {'qwen3:30b-a3b'}
  14. def effective_think(model, want_think):
  15. """实际应传给 ollama 的 think 值: THINK_LOCKED 模型恒 True (关不掉, 只能要分离), 其余透传."""
  16. return True if model in THINK_LOCKED else bool(want_think)
  17. def models():
  18. try:
  19. with urllib.request.urlopen(URL + '/api/tags', timeout=3) as r:
  20. return [m['name'] for m in json.loads(r.read()).get('models', [])]
  21. except Exception:
  22. return []
  23. def pick(model=None, fast=False):
  24. av = models()
  25. if model and model in av:
  26. return model
  27. for p in (PREFER_FAST if fast else PREFER):
  28. if p in av:
  29. return p
  30. return av[0] if av else None
  31. def generate(prompt, model=None, system=None, temperature=0.2, num_predict=900, timeout=240, think=False):
  32. """返回 (text|None, model|None). 任何失败 → (None, model) 触发结构化回退."""
  33. m = pick(model)
  34. if not m:
  35. return None, None
  36. tf = effective_think(m, think)
  37. npd = num_predict + 1200 if (tf and not think) else num_predict # 被强制思考: 补 thinking 预算, 保正文空间
  38. body = {'model': m, 'prompt': prompt, 'stream': False, 'keep_alive': '30m', 'think': tf, 'options': {'temperature': temperature, 'num_predict': npd}}
  39. if system:
  40. body['system'] = system
  41. try:
  42. req = urllib.request.Request(URL + '/api/generate', data=json.dumps(body).encode('utf-8'), headers={'Content-Type': 'application/json'})
  43. with urllib.request.urlopen(req, timeout=timeout) as r:
  44. d = json.loads(r.read().decode('utf-8'))
  45. txt = re.sub(r'<think>.*?</think>', '', d.get('response', ''), flags=re.S).strip()
  46. return txt, m
  47. except Exception:
  48. return None, m
  49. _TID = re.compile(r'WTG\s?\d{1,2}|\d{1,2}\s?#|\d{1,2}\s?号')
  50. _NUM = re.compile(r'\d+(?:\.\d+)?')
  51. def _norm_tid(t):
  52. d = re.findall(r'\d+', t)
  53. return f'{int(d[0]):02d}' if d else t
  54. def _num_ok(a_str, ctx_vals):
  55. """答案数字 a 可接受: 上下文里有同串; 或有 c 使 round(c, a 的小数位数)==a; 或相对差 <1%; 或 a 是 ≤60 的小整数 (条数/序号/窗数)."""
  56. try:
  57. a = float(a_str)
  58. except ValueError:
  59. return True
  60. if '.' not in a_str and a <= 60:
  61. return True
  62. dec = len(a_str.split('.')[1]) if '.' in a_str else 0
  63. for c in ctx_vals:
  64. if round(c, dec) == a or (c != 0 and abs(c - a) / abs(c) < 0.01):
  65. return True
  66. return False
  67. def grounding(answer, context_text):
  68. """接地闸: 回答里的台号 ⊆ 上下文台号; 回答里的数字 ⊆ 上下文数字集 (允许舍入变体/1% 内/小整数; 年份放行). 返回 (ok, 违规项)."""
  69. ctx_t = {_norm_tid(t) for t in _TID.findall(context_text)}
  70. ans_t = {_norm_tid(t) for t in _TID.findall(answer)}
  71. bad_t = ans_t - ctx_t
  72. ctx_raw = set(_NUM.findall(context_text))
  73. ctx_vals = []
  74. for s in ctx_raw:
  75. try:
  76. ctx_vals.append(float(s))
  77. except ValueError:
  78. pass
  79. bad_n = [n for n in _NUM.findall(answer) if n not in ctx_raw and not re.fullmatch(r'20\d\d', n) and not _num_ok(n, ctx_vals)]
  80. # 级别升格闸 (2026-08-24): "准定论·预警"被写成"定论·预警"= 掉字升级; 答案出现独立"定论"而上下文没有 → 拦
  81. lvl_up = ((bool(re.search(r'(?<!准)定论', answer)) and not re.search(r'(?<!准)定论', context_text))
  82. or ('危险' in answer and '危险' not in context_text)) # 无依据写危险 = 升格, 同拦
  83. return (not bad_t and not bad_n and not lvl_up), dict(turbines=sorted(bad_t), numbers=sorted(set(bad_n))[:10], **({'level_upgrade': '答案含"定论"而事实只有"准定论·预警"级'} if lvl_up else {}))