models.json 1.2 KB

12345678910111213141516
  1. {
  2. "_note": "离线模型档策略 (2026-09-07). 判断在代码, 模型只问答/转述/初筛; 换模型改这里 + ollama pull, 不改结论.",
  3. "runtime": "ollama", "endpoint": "http://127.0.0.1:11434",
  4. "tiers": {
  5. "default_qa": {"model": "qwen3:8b", "vram_gb": 6, "note": "默认问答, think=false; 校闸不过自动升档一次"},
  6. "escalate_qa": {"model": "qwen3:32b", "vram_gb": 22, "note": "复杂题升档(★2026-09-22 用户人工测试 7 实逮: 原写 qwen3.8:27b, 本机没这个 tag ⇒ 升档必失败、ask 最终 error; 已改为本机实装的 qwen3:32b)"},
  7. "review": {"model": "deepseek-r1:14b", "vram_gb": 10, "note": "交叉审核初筛票 (非独立审级; 确诊/¥/上云仍需云端跨厂商双审)"},
  8. "embed": {"model": "bge-m3", "vram_gb": 2, "note": "检索向量"}
  9. },
  10. "profiles": {
  11. "gpu_24gb_plus": ["default_qa", "escalate_qa", "review", "embed"],
  12. "gpu_12_16gb": ["default_qa", "review", "embed"],
  13. "cpu_only": ["default_qa", "embed"]
  14. },
  15. "pull_commands": ["ollama pull qwen3:8b", "ollama pull qwen3:32b", "ollama pull deepseek-r1:14b", "ollama pull bge-m3"]
  16. }