# -*- coding: utf-8 -*- """离线大模型客户端 (ollama, 模型按名可换: qwen3:32b / deepseek-r1:32b ...) + 接地闸 (台号 + 数字必须来自事实上下文).""" import json, re, urllib.request URL = 'http://localhost:11434' PREFER = ['qwen3:32b', 'qwen3:30b-a3b', 'qwen3:14b', 'qwen3:8b', 'qwen2.5:14b', 'qwen2.5:7b', 'deepseek-r1:32b', 'deepseek-r1:14b'] # 交互路径优先小模型 (2026-08-24 A/B: 8b 对比题 4s/诊断题 27s vs 32b 8s/36s, 两题结论与数字全对; 用户批"模型可以降级"); # 报告润色/审核仍走 PREFER (32b 优先, 批处理质量优先) PREFER_FAST = ['qwen3:8b', 'qwen3:14b', 'qwen3:30b-a3b', 'qwen3:32b', 'qwen2.5:7b'] # think:false 失效模型 (恒思考版, 如 Qwen3-30B-A3B Thinking-2507 系): think:false 只关 ollama 的 # thinking/content 分离、关不掉模型思考 → 思考文本整段灌 content (2026-08-26 实测 chat/generate 两端点同病, # /no_think 软开关同无效; 对照 qwen3:8b 两端点均正常). 修法 = 强制 think:true 拿分离, 取 content 丢 thinking; # 调用侧对被强制场景放大 num_predict (thinking 先吃预算, 实测 400 tok 全被吃光 content 为空). THINK_LOCKED = {'qwen3:30b-a3b'} def effective_think(model, want_think): """实际应传给 ollama 的 think 值: THINK_LOCKED 模型恒 True (关不掉, 只能要分离), 其余透传.""" return True if model in THINK_LOCKED else bool(want_think) def models(): try: with urllib.request.urlopen(URL + '/api/tags', timeout=3) as r: return [m['name'] for m in json.loads(r.read()).get('models', [])] except Exception: return [] def pick(model=None, fast=False): av = models() if model and model in av: return model for p in (PREFER_FAST if fast else PREFER): if p in av: return p return av[0] if av else None def generate(prompt, model=None, system=None, temperature=0.2, num_predict=900, timeout=240, think=False): """返回 (text|None, model|None). 任何失败 → (None, model) 触发结构化回退.""" m = pick(model) if not m: return None, None tf = effective_think(m, think) npd = num_predict + 1200 if (tf and not think) else num_predict # 被强制思考: 补 thinking 预算, 保正文空间 body = {'model': m, 'prompt': prompt, 'stream': False, 'keep_alive': '30m', 'think': tf, 'options': {'temperature': temperature, 'num_predict': npd}} if system: body['system'] = system try: req = urllib.request.Request(URL + '/api/generate', data=json.dumps(body).encode('utf-8'), headers={'Content-Type': 'application/json'}) with urllib.request.urlopen(req, timeout=timeout) as r: d = json.loads(r.read().decode('utf-8')) txt = re.sub(r'.*?', '', d.get('response', ''), flags=re.S).strip() return txt, m except Exception: return None, m _TID = re.compile(r'WTG\s?\d{1,2}|\d{1,2}\s?#|\d{1,2}\s?号') _NUM = re.compile(r'\d+(?:\.\d+)?') def _norm_tid(t): d = re.findall(r'\d+', t) return f'{int(d[0]):02d}' if d else t def _num_ok(a_str, ctx_vals): """答案数字 a 可接受: 上下文里有同串; 或有 c 使 round(c, a 的小数位数)==a; 或相对差 <1%; 或 a 是 ≤60 的小整数 (条数/序号/窗数).""" try: a = float(a_str) except ValueError: return True if '.' not in a_str and a <= 60: return True dec = len(a_str.split('.')[1]) if '.' in a_str else 0 for c in ctx_vals: if round(c, dec) == a or (c != 0 and abs(c - a) / abs(c) < 0.01): return True return False def grounding(answer, context_text): """接地闸: 回答里的台号 ⊆ 上下文台号; 回答里的数字 ⊆ 上下文数字集 (允许舍入变体/1% 内/小整数; 年份放行). 返回 (ok, 违规项).""" ctx_t = {_norm_tid(t) for t in _TID.findall(context_text)} ans_t = {_norm_tid(t) for t in _TID.findall(answer)} bad_t = ans_t - ctx_t ctx_raw = set(_NUM.findall(context_text)) ctx_vals = [] for s in ctx_raw: try: ctx_vals.append(float(s)) except ValueError: pass bad_n = [n for n in _NUM.findall(answer) if n not in ctx_raw and not re.fullmatch(r'20\d\d', n) and not _num_ok(n, ctx_vals)] # 级别升格闸 (2026-08-24): "准定论·预警"被写成"定论·预警"= 掉字升级; 答案出现独立"定论"而上下文没有 → 拦 lvl_up = ((bool(re.search(r'(?