Files
vscode-workbench/PLANNING/model-registry.json
T

306 lines
8.9 KiB
JSON
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
{
"schema": "seagull-model-registry/1.1",
"updated": "2026-09-11",
"maintainer": "planner",
"note": "评估成绩由 I-04 跑题后回填 scores;未评估模型 grade=null,不得派关键路径任务(派单规则见 01-模型分工矩阵.md)。等级换算用百分比,故 SPEC-DEFECT-1(标题 36 分 vs 维度表合计 42 分)不影响评级。",
"scale": {
"total_max": 36,
"dims": {
"dim1": 11,
"dim2": 7,
"dim3": 6,
"dim4": 6,
"dim5": 6,
"dim6": 6
},
"veto": [
"dim5<50%",
"dim4 伪造证据",
"dim6 虚报验收",
"拒答不重试"
]
},
"models": [
{
"id": "glm-5.3-flash",
"groups": [
"b",
"command-code(z-ai/)",
"lt-fl",
"lt-th",
"lt-fll",
"zz"
],
"capabilities": {
"vision": true,
"thinking": true,
"thinking_format": "zai",
"ctx": 1048576,
"max_tokens": 131072
},
"verified": "2026-09-06 实测:视觉✅(64px红PNG) / 思考✅(reasoning_effort=high) / developer 角色回落 system",
"scores": null,
"grade": "A",
"grade_basis": "已验证思考+视觉,多分组冗余部署(非评估成绩,属既有事实等级)(既有事实等级;I-04 校准待跑,scores 仍为 null)",
"limits": [
"网关不认 role=developer(已配回落)"
],
"eval_pending": true,
"eval_command": "cd PLANNING/evals && python tools/run_eval.py --model glm-5.3-flash --transport openai --base-url <GATEWAY>/v1 --api-key-env DSH_API_KEY"
},
{
"id": "qwen3.8-flash",
"groups": [
"b",
"lt-fl"
],
"capabilities": {
"vision": true,
"thinking": true,
"thinking_format": "zai",
"ctx": 1048576,
"max_tokens": 131072
},
"verified": "2026-09-06 实测:视觉✅ / 思考✅(档位 low/medium/xhigh;high 映射 xhigh;off 档网关仍返回 reasoning)",
"scores": null,
"grade": "A",
"grade_basis": "已验证思考+视觉;1M 上下文适合长文任务",
"limits": [
"不认 reasoning_effort=high(需映射 xhigh)",
"无真正关闭思考档"
],
"eval_pending": false
},
{
"id": "deepseek-v4.1-flash",
"groups": [
"cc"
],
"capabilities": {
"vision": false,
"thinking": true,
"ctx": null,
"max_tokens": null
},
"verified": null,
"scores": null,
"grade": "A",
"grade_basis": "主力编码指定(规划者选型,待 I-04 评估确认)(既有事实等级;I-04 校准待跑,scores 仍为 null)",
"limits": [],
"eval_pending": true,
"eval_command": "cd PLANNING/evals && python tools/run_eval.py --model deepseek-v4.1-flash --transport openai --base-url <GATEWAY>/v1 --api-key-env DSH_API_KEY"
},
{
"id": "deepseek-v4-flash",
"groups": [
"command-code",
"lt-fl",
"ginka"
],
"capabilities": {
"vision": true,
"thinking": true,
"thinking_format": "zai",
"ctx": 1048576,
"max_tokens": 32768
},
"verified": "2026-09-04 实测:视觉✅;reasoning_effort → reasoning_content (rt=13)",
"scores": null,
"grade": "A",
"grade_basis": "批量编码/脚本/CI 指定",
"limits": [],
"eval_pending": false
},
{
"id": "deepseek-v4-flash-vision-exp",
"groups": [
"cc"
],
"capabilities": {
"vision": true,
"thinking": null,
"ctx": null,
"max_tokens": null
},
"verified": null,
"scores": null,
"grade": "B",
"grade_basis": "视觉实验通道,用于图像链路验证",
"limits": [
"实验模型,不作关键路径主执行"
],
"eval_pending": false
},
{
"id": "kimi-k3",
"groups": [
"lt-fl"
],
"capabilities": {
"vision": null,
"thinking": null,
"ctx": null,
"max_tokens": null
},
"verified": null,
"scores": null,
"grade": "A",
"grade_basis": "超长上下文定位(全库审计类),待 I-04 实测确认",
"limits": [],
"eval_pending": false
},
{
"id": "MiniMax-M3",
"groups": [
"lt-fl",
"lt-th"
],
"capabilities": {
"vision": null,
"thinking": null,
"ctx": null,
"max_tokens": null
},
"verified": null,
"scores": null,
"grade": "B",
"grade_basis": "通用执行",
"limits": [],
"eval_pending": false
},
{
"id": "gpt-5.6-luna",
"groups": [
"lt-zya"
],
"capabilities": {
"vision": null,
"thinking": null,
"ctx": null,
"max_tokens": null
},
"verified": null,
"scores": null,
"grade": "S",
"grade_basis": "池内最强推理定位(架构/安全类指定),待 I-04 评估确认",
"limits": [],
"eval_pending": false
},
{
"id": "gemini-3.8-flash",
"groups": [
"zy"
],
"capabilities": {
"vision": true,
"thinking": null,
"ctx": null,
"max_tokens": null
},
"verified": null,
"scores": null,
"grade": "B",
"grade_basis": "视觉验收指定(UI/Canvas/截图类)",
"limits": [],
"eval_pending": false
},
{
"id": "hy3",
"groups": [
"b"
],
"capabilities": {
"vision": true,
"thinking": true,
"thinking_format": "zai",
"ctx": 1048576,
"max_tokens": 32768
},
"verified": "2026-09-04 实测:原生思考可用(SSE reasoning_content);回复偏短",
"scores": null,
"grade": "C",
"grade_basis": "免费额度辅助,只做草稿/低风险批量",
"limits": [
"回复短弱"
],
"eval_pending": false
},
{
"id": "mimo-v2.5",
"groups": [
"b",
"command-code(xiaomi/)"
],
"capabilities": {
"vision": true,
"thinking": true,
"thinking_format": "zai",
"ctx": 1048576,
"max_tokens": 32768
},
"verified": "2026-09-04 实测:思考可用;视觉(cc 通道)✅",
"scores": null,
"grade": "C",
"grade_basis": "免费额度辅助",
"limits": [],
"eval_pending": false
},
{
"id": "LongCat-2.0",
"groups": [
"command-code(meituan/)"
],
"capabilities": {
"vision": true,
"thinking": true,
"ctx": 1048576,
"max_tokens": 32768
},
"verified": "2026-09-04 实测:视觉✅;reasoning_effort 可用",
"scores": null,
"grade": "C",
"grade_basis": "免费额度辅助(:free 通道)",
"limits": [
"免费通道限流风险"
],
"eval_pending": false
},
{
"id": "laguna-s-2.1",
"groups": [
"command-code(poolside/)"
],
"capabilities": {
"vision": null,
"thinking": null,
"ctx": null,
"max_tokens": null
},
"verified": null,
"scores": null,
"grade": "C",
"grade_basis": "免费额度辅助,未验证",
"limits": [
"完全未验证——首次派单前必须跑 I-04 评估"
],
"eval_pending": false
}
],
"eval": {
"runner": "evals/tools/run_eval.py",
"validator": "evals/tools/validate.py",
"questions_max": 42,
"headline_total_max": 36,
"spec_defect": "SPEC-DEFECT-1:标题分 36 与维度表合计 42 不一致,待规划者裁定;等级换算用百分比故不影响评级",
"commands": {
"audit": "cd PLANNING/evals && python tools/run_eval.py --audit",
"calibrate": "cd PLANNING/evals && python tools/run_eval.py --model <MODEL> --transport openai --base-url <GATEWAY>/v1 --api-key-env DSH_API_KEY",
"offline_selftest": "cd PLANNING/evals && python tools/run_eval.py --model selftest-synthetic --transport replay --answers-dir fixtures/selftest/good --workdir fixtures/selftest/workdir-good/tooltask",
"finalize": "cd PLANNING/evals && python tools/run_eval.py --manual results/<MODEL>-<DATE>.manual.template.json"
},
"results_dir": "evals/results/",
"calibrated": false,
"blocker": "真实校准需要 DSH 网关的可调用端点与 API key(key 只从环境变量读)。2026-09-11 打包环境里网关 TCP 可达但本会话无凭据/端点信息,故两个校准模型先登记 eval_pending,成绩保持 null。"
}
}