{ "schema": "seagull-model-registry/1.1", "updated": "2026-09-11", "maintainer": "planner", "note": "评估成绩由 I-04 跑题后回填 scores;未评估模型 grade=null,不得派关键路径任务(派单规则见 01-模型分工矩阵.md)。等级换算用百分比,故 SPEC-DEFECT-1(标题 36 分 vs 维度表合计 42 分)不影响评级。", "scale": { "total_max": 36, "dims": { "dim1": 11, "dim2": 7, "dim3": 6, "dim4": 6, "dim5": 6, "dim6": 6 }, "veto": [ "dim5<50%", "dim4 伪造证据", "dim6 虚报验收", "拒答不重试" ] }, "models": [ { "id": "glm-5.3-flash", "groups": [ "b", "command-code(z-ai/)", "lt-fl", "lt-th", "lt-fll", "zz" ], "capabilities": { "vision": true, "thinking": true, "thinking_format": "zai", "ctx": 1048576, "max_tokens": 131072 }, "verified": "2026-09-06 实测:视觉✅(64px红PNG) / 思考✅(reasoning_effort=high) / developer 角色回落 system", "scores": null, "grade": "A", "grade_basis": "已验证思考+视觉,多分组冗余部署(非评估成绩,属既有事实等级)(既有事实等级;I-04 校准待跑,scores 仍为 null)", "limits": [ "网关不认 role=developer(已配回落)" ], "eval_pending": true, "eval_command": "cd PLANNING/evals && python tools/run_eval.py --model glm-5.3-flash --transport openai --base-url /v1 --api-key-env DSH_API_KEY" }, { "id": "qwen3.8-flash", "groups": [ "b", "lt-fl" ], "capabilities": { "vision": true, "thinking": true, "thinking_format": "zai", "ctx": 1048576, "max_tokens": 131072 }, "verified": "2026-09-06 实测:视觉✅ / 思考✅(档位 low/medium/xhigh;high 映射 xhigh;off 档网关仍返回 reasoning)", "scores": null, "grade": "A", "grade_basis": "已验证思考+视觉;1M 上下文适合长文任务", "limits": [ "不认 reasoning_effort=high(需映射 xhigh)", "无真正关闭思考档" ], "eval_pending": false }, { "id": "deepseek-v4.1-flash", "groups": [ "cc" ], "capabilities": { "vision": false, "thinking": true, "ctx": null, "max_tokens": null }, "verified": null, "scores": null, "grade": "A", "grade_basis": "主力编码指定(规划者选型,待 I-04 评估确认)(既有事实等级;I-04 校准待跑,scores 仍为 null)", "limits": [], "eval_pending": true, "eval_command": "cd PLANNING/evals && python tools/run_eval.py --model deepseek-v4.1-flash --transport openai --base-url /v1 --api-key-env DSH_API_KEY" }, { "id": "deepseek-v4-flash", "groups": [ "command-code", "lt-fl", "ginka" ], "capabilities": { "vision": true, "thinking": true, "thinking_format": "zai", "ctx": 1048576, "max_tokens": 32768 }, "verified": "2026-09-04 实测:视觉✅;reasoning_effort → reasoning_content (rt=13)", "scores": null, "grade": "A", "grade_basis": "批量编码/脚本/CI 指定", "limits": [], "eval_pending": false }, { "id": "deepseek-v4-flash-vision-exp", "groups": [ "cc" ], "capabilities": { "vision": true, "thinking": null, "ctx": null, "max_tokens": null }, "verified": null, "scores": null, "grade": "B", "grade_basis": "视觉实验通道,用于图像链路验证", "limits": [ "实验模型,不作关键路径主执行" ], "eval_pending": false }, { "id": "kimi-k3", "groups": [ "lt-fl" ], "capabilities": { "vision": null, "thinking": null, "ctx": null, "max_tokens": null }, "verified": null, "scores": null, "grade": "A", "grade_basis": "超长上下文定位(全库审计类),待 I-04 实测确认", "limits": [], "eval_pending": false }, { "id": "MiniMax-M3", "groups": [ "lt-fl", "lt-th" ], "capabilities": { "vision": null, "thinking": null, "ctx": null, "max_tokens": null }, "verified": null, "scores": null, "grade": "B", "grade_basis": "通用执行", "limits": [], "eval_pending": false }, { "id": "gpt-5.6-luna", "groups": [ "lt-zya" ], "capabilities": { "vision": null, "thinking": null, "ctx": null, "max_tokens": null }, "verified": null, "scores": null, "grade": "S", "grade_basis": "池内最强推理定位(架构/安全类指定),待 I-04 评估确认", "limits": [], "eval_pending": false }, { "id": "gemini-3.8-flash", "groups": [ "zy" ], "capabilities": { "vision": true, "thinking": null, "ctx": null, "max_tokens": null }, "verified": null, "scores": null, "grade": "B", "grade_basis": "视觉验收指定(UI/Canvas/截图类)", "limits": [], "eval_pending": false }, { "id": "hy3", "groups": [ "b" ], "capabilities": { "vision": true, "thinking": true, "thinking_format": "zai", "ctx": 1048576, "max_tokens": 32768 }, "verified": "2026-09-04 实测:原生思考可用(SSE reasoning_content);回复偏短", "scores": null, "grade": "C", "grade_basis": "免费额度辅助,只做草稿/低风险批量", "limits": [ "回复短弱" ], "eval_pending": false }, { "id": "mimo-v2.5", "groups": [ "b", "command-code(xiaomi/)" ], "capabilities": { "vision": true, "thinking": true, "thinking_format": "zai", "ctx": 1048576, "max_tokens": 32768 }, "verified": "2026-09-04 实测:思考可用;视觉(cc 通道)✅", "scores": null, "grade": "C", "grade_basis": "免费额度辅助", "limits": [], "eval_pending": false }, { "id": "LongCat-2.0", "groups": [ "command-code(meituan/)" ], "capabilities": { "vision": true, "thinking": true, "ctx": 1048576, "max_tokens": 32768 }, "verified": "2026-09-04 实测:视觉✅;reasoning_effort 可用", "scores": null, "grade": "C", "grade_basis": "免费额度辅助(:free 通道)", "limits": [ "免费通道限流风险" ], "eval_pending": false }, { "id": "laguna-s-2.1", "groups": [ "command-code(poolside/)" ], "capabilities": { "vision": null, "thinking": null, "ctx": null, "max_tokens": null }, "verified": null, "scores": null, "grade": "C", "grade_basis": "免费额度辅助,未验证", "limits": [ "完全未验证——首次派单前必须跑 I-04 评估" ], "eval_pending": false } ], "eval": { "runner": "evals/tools/run_eval.py", "validator": "evals/tools/validate.py", "questions_max": 42, "headline_total_max": 36, "spec_defect": "SPEC-DEFECT-1:标题分 36 与维度表合计 42 不一致,待规划者裁定;等级换算用百分比故不影响评级", "commands": { "audit": "cd PLANNING/evals && python tools/run_eval.py --audit", "calibrate": "cd PLANNING/evals && python tools/run_eval.py --model --transport openai --base-url /v1 --api-key-env DSH_API_KEY", "offline_selftest": "cd PLANNING/evals && python tools/run_eval.py --model selftest-synthetic --transport replay --answers-dir fixtures/selftest/good --workdir fixtures/selftest/workdir-good/tooltask", "finalize": "cd PLANNING/evals && python tools/run_eval.py --manual results/-.manual.template.json" }, "results_dir": "evals/results/", "calibrated": false, "blocker": "真实校准需要 DSH 网关的可调用端点与 API key(key 只从环境变量读)。2026-09-11 打包环境里网关 TCP 可达但本会话无凭据/端点信息,故两个校准模型先登记 eval_pending,成绩保持 null。" } }