feat: PLANNING体系首版(00路线图/02任务总表/03执行协议/I-03/I-04/registry42口径)+M1止血交付

This commit is contained in:
2026-09-12 14:25:25 +08:00
commit 39af400fc8
472 changed files with 36277 additions and 0 deletions
+305
View File
@@ -0,0 +1,305 @@
{
"schema": "seagull-model-registry/1.1",
"updated": "2026-09-11",
"maintainer": "planner",
"note": "评估成绩由 I-04 跑题后回填 scores;未评估模型 grade=null,不得派关键路径任务(派单规则见 01-模型分工矩阵.md)。等级换算用百分比,故 SPEC-DEFECT-1(标题 36 分 vs 维度表合计 42 分)不影响评级。",
"scale": {
"total_max": 36,
"dims": {
"dim1": 11,
"dim2": 7,
"dim3": 6,
"dim4": 6,
"dim5": 6,
"dim6": 6
},
"veto": [
"dim5<50%",
"dim4 伪造证据",
"dim6 虚报验收",
"拒答不重试"
]
},
"models": [
{
"id": "glm-5.3-flash",
"groups": [
"b",
"command-code(z-ai/)",
"lt-fl",
"lt-th",
"lt-fll",
"zz"
],
"capabilities": {
"vision": true,
"thinking": true,
"thinking_format": "zai",
"ctx": 1048576,
"max_tokens": 131072
},
"verified": "2026-09-06 实测:视觉✅(64px红PNG) / 思考✅(reasoning_effort=high) / developer 角色回落 system",
"scores": null,
"grade": "A",
"grade_basis": "已验证思考+视觉,多分组冗余部署(非评估成绩,属既有事实等级)(既有事实等级;I-04 校准待跑,scores 仍为 null)",
"limits": [
"网关不认 role=developer(已配回落)"
],
"eval_pending": true,
"eval_command": "cd PLANNING/evals && python tools/run_eval.py --model glm-5.3-flash --transport openai --base-url <GATEWAY>/v1 --api-key-env DSH_API_KEY"
},
{
"id": "qwen3.8-flash",
"groups": [
"b",
"lt-fl"
],
"capabilities": {
"vision": true,
"thinking": true,
"thinking_format": "zai",
"ctx": 1048576,
"max_tokens": 131072
},
"verified": "2026-09-06 实测:视觉✅ / 思考✅(档位 low/medium/xhigh;high 映射 xhigh;off 档网关仍返回 reasoning)",
"scores": null,
"grade": "A",
"grade_basis": "已验证思考+视觉;1M 上下文适合长文任务",
"limits": [
"不认 reasoning_effort=high(需映射 xhigh)",
"无真正关闭思考档"
],
"eval_pending": false
},
{
"id": "deepseek-v4.1-flash",
"groups": [
"cc"
],
"capabilities": {
"vision": false,
"thinking": true,
"ctx": null,
"max_tokens": null
},
"verified": null,
"scores": null,
"grade": "A",
"grade_basis": "主力编码指定(规划者选型,待 I-04 评估确认)(既有事实等级;I-04 校准待跑,scores 仍为 null)",
"limits": [],
"eval_pending": true,
"eval_command": "cd PLANNING/evals && python tools/run_eval.py --model deepseek-v4.1-flash --transport openai --base-url <GATEWAY>/v1 --api-key-env DSH_API_KEY"
},
{
"id": "deepseek-v4-flash",
"groups": [
"command-code",
"lt-fl",
"ginka"
],
"capabilities": {
"vision": true,
"thinking": true,
"thinking_format": "zai",
"ctx": 1048576,
"max_tokens": 32768
},
"verified": "2026-09-04 实测:视觉✅;reasoning_effort → reasoning_content (rt=13)",
"scores": null,
"grade": "A",
"grade_basis": "批量编码/脚本/CI 指定",
"limits": [],
"eval_pending": false
},
{
"id": "deepseek-v4-flash-vision-exp",
"groups": [
"cc"
],
"capabilities": {
"vision": true,
"thinking": null,
"ctx": null,
"max_tokens": null
},
"verified": null,
"scores": null,
"grade": "B",
"grade_basis": "视觉实验通道,用于图像链路验证",
"limits": [
"实验模型,不作关键路径主执行"
],
"eval_pending": false
},
{
"id": "kimi-k3",
"groups": [
"lt-fl"
],
"capabilities": {
"vision": null,
"thinking": null,
"ctx": null,
"max_tokens": null
},
"verified": null,
"scores": null,
"grade": "A",
"grade_basis": "超长上下文定位(全库审计类),待 I-04 实测确认",
"limits": [],
"eval_pending": false
},
{
"id": "MiniMax-M3",
"groups": [
"lt-fl",
"lt-th"
],
"capabilities": {
"vision": null,
"thinking": null,
"ctx": null,
"max_tokens": null
},
"verified": null,
"scores": null,
"grade": "B",
"grade_basis": "通用执行",
"limits": [],
"eval_pending": false
},
{
"id": "gpt-5.6-luna",
"groups": [
"lt-zya"
],
"capabilities": {
"vision": null,
"thinking": null,
"ctx": null,
"max_tokens": null
},
"verified": null,
"scores": null,
"grade": "S",
"grade_basis": "池内最强推理定位(架构/安全类指定),待 I-04 评估确认",
"limits": [],
"eval_pending": false
},
{
"id": "gemini-3.8-flash",
"groups": [
"zy"
],
"capabilities": {
"vision": true,
"thinking": null,
"ctx": null,
"max_tokens": null
},
"verified": null,
"scores": null,
"grade": "B",
"grade_basis": "视觉验收指定(UI/Canvas/截图类)",
"limits": [],
"eval_pending": false
},
{
"id": "hy3",
"groups": [
"b"
],
"capabilities": {
"vision": true,
"thinking": true,
"thinking_format": "zai",
"ctx": 1048576,
"max_tokens": 32768
},
"verified": "2026-09-04 实测:原生思考可用(SSE reasoning_content);回复偏短",
"scores": null,
"grade": "C",
"grade_basis": "免费额度辅助,只做草稿/低风险批量",
"limits": [
"回复短弱"
],
"eval_pending": false
},
{
"id": "mimo-v2.5",
"groups": [
"b",
"command-code(xiaomi/)"
],
"capabilities": {
"vision": true,
"thinking": true,
"thinking_format": "zai",
"ctx": 1048576,
"max_tokens": 32768
},
"verified": "2026-09-04 实测:思考可用;视觉(cc 通道)✅",
"scores": null,
"grade": "C",
"grade_basis": "免费额度辅助",
"limits": [],
"eval_pending": false
},
{
"id": "LongCat-2.0",
"groups": [
"command-code(meituan/)"
],
"capabilities": {
"vision": true,
"thinking": true,
"ctx": 1048576,
"max_tokens": 32768
},
"verified": "2026-09-04 实测:视觉✅;reasoning_effort 可用",
"scores": null,
"grade": "C",
"grade_basis": "免费额度辅助(:free 通道)",
"limits": [
"免费通道限流风险"
],
"eval_pending": false
},
{
"id": "laguna-s-2.1",
"groups": [
"command-code(poolside/)"
],
"capabilities": {
"vision": null,
"thinking": null,
"ctx": null,
"max_tokens": null
},
"verified": null,
"scores": null,
"grade": "C",
"grade_basis": "免费额度辅助,未验证",
"limits": [
"完全未验证——首次派单前必须跑 I-04 评估"
],
"eval_pending": false
}
],
"eval": {
"runner": "evals/tools/run_eval.py",
"validator": "evals/tools/validate.py",
"questions_max": 42,
"headline_total_max": 36,
"spec_defect": "SPEC-DEFECT-1:标题分 36 与维度表合计 42 不一致,待规划者裁定;等级换算用百分比故不影响评级",
"commands": {
"audit": "cd PLANNING/evals && python tools/run_eval.py --audit",
"calibrate": "cd PLANNING/evals && python tools/run_eval.py --model <MODEL> --transport openai --base-url <GATEWAY>/v1 --api-key-env DSH_API_KEY",
"offline_selftest": "cd PLANNING/evals && python tools/run_eval.py --model selftest-synthetic --transport replay --answers-dir fixtures/selftest/good --workdir fixtures/selftest/workdir-good/tooltask",
"finalize": "cd PLANNING/evals && python tools/run_eval.py --manual results/<MODEL>-<DATE>.manual.template.json"
},
"results_dir": "evals/results/",
"calibrated": false,
"blocker": "真实校准需要 DSH 网关的可调用端点与 API key(key 只从环境变量读)。2026-09-11 打包环境里网关 TCP 可达但本会话无凭据/端点信息,故两个校准模型先登记 eval_pending,成绩保持 null。"
}
}