models.json 1.0 KB

12345678910111213141516
  1. {
  2. "_note": "离线模型档策略 (2026-09-07). 判断在代码, 模型只问答/转述/初筛; 换模型改这里 + ollama pull, 不改结论.",
  3. "runtime": "ollama", "endpoint": "http://127.0.0.1:11434",
  4. "tiers": {
  5. "default_qa": {"model": "qwen3:8b", "vram_gb": 6, "note": "默认问答, think=false; 校闸不过自动升档一次"},
  6. "escalate_qa": {"model": "qwen3.8:27b", "vram_gb": 20, "note": "复杂题升档 (开发机实测六问 6/6, 复杂题最长 14.5 min)"},
  7. "review": {"model": "deepseek-r1:14b", "vram_gb": 10, "note": "交叉审核初筛票 (非独立审级; 确诊/¥/上云仍需云端跨厂商双审)"},
  8. "embed": {"model": "bge-m3", "vram_gb": 2, "note": "检索向量"}
  9. },
  10. "profiles": {
  11. "gpu_24gb_plus": ["default_qa", "escalate_qa", "review", "embed"],
  12. "gpu_12_16gb": ["default_qa", "review", "embed"],
  13. "cpu_only": ["default_qa", "embed"]
  14. },
  15. "pull_commands": ["ollama pull qwen3:8b", "ollama pull qwen3.8:27b", "ollama pull deepseek-r1:14b", "ollama pull bge-m3"]
  16. }