Files
vscode-workbench/PLANNING/evals/results/selftest-synthetic-2026-09-11.json
T

261 lines
6.9 KiB
JSON
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
{
"model": "selftest-synthetic",
"group": null,
"date": "2026-09-11",
"runner": "run_eval.py",
"transport": "replay",
"base_url": null,
"scores": {
"dim1": {
"got": 10,
"max": 11,
"manual_pending": 1,
"notes": "",
"manual_got": 1,
"got_total": 11
},
"dim2": {
"got": 5,
"max": 7,
"manual_pending": 2,
"notes": "",
"manual_got": 2,
"got_total": 7
},
"dim3": {
"got": 3,
"max": 6,
"manual_pending": 3,
"notes": "",
"manual_got": 3,
"got_total": 6
},
"dim4": {
"got": 3,
"max": 6,
"manual_pending": 3,
"notes": "",
"manual_got": 3,
"got_total": 6
},
"dim5": {
"got": 3,
"max": 6,
"manual_pending": 3,
"notes": "",
"manual_got": 3,
"got_total": 6
},
"dim6": {
"got": 5,
"max": 6,
"manual_pending": 1,
"notes": "",
"manual_got": 1,
"got_total": 6
}
},
"per_question": {
"dim1-q1": {
"auto_got": 5,
"max": 5,
"manual_pending": 0,
"logs": [
"[PASS] 整段可解析为 JSON",
"[PASS] 恰好三字段 id/name/tags",
"[PASS] id=42 name=seagull",
"[PASS] tags 为长度 3 数组",
"[PASS] 无 Markdown 包裹/多余文本"
],
"manual_got": 0
},
"dim1-q2": {
"auto_got": 1,
"max": 2,
"manual_pending": 1,
"logs": [
"[PASS] 汉字数=50(需恰 50)",
"[INFO] 内容正确性需人工/评委确认(1 分)"
],
"manual_got": 1
},
"dim1-q3": {
"auto_got": 4,
"max": 4,
"manual_pending": 0,
"logs": [
"[PASS] 恰好 3 行非空",
"[PASS] 三行格式匹配 type(scope): subject",
"[PASS] type 覆盖 feat/fix/docs 各一",
"[PASS] subject 均 ≤20 字符"
],
"manual_got": 0
},
"dim2-q1": {
"auto_got": 2,
"max": 3,
"manual_pending": 1,
"logs": [
"[PASS] average([])==0.0 且返回 float",
"[PASS] average([1,2,3])==2.0 且 average([1,2])==1.5",
"[INFO] 最小改动(≤4 行、签名与结构未重写)需评委确认(1 分)"
],
"manual_got": 1
},
"dim2-q2": {
"auto_got": 2,
"max": 2,
"manual_pending": 0,
"logs": [
"[PASS] sum([1,2,3])=6 且 sum([])=0",
"[PASS] 未使用 reduce"
],
"manual_got": 0
},
"dim2-q3": {
"auto_got": 1,
"max": 2,
"manual_pending": 1,
"logs": [
"[PASS] 三个规格用例全部通过",
"[INFO] 简洁性(未过度工程)需人工/评委确认(1 分)"
],
"manual_got": 1
},
"dim3-q1": {
"auto_got": 1,
"max": 2,
"manual_pending": 1,
"logs": [
"[PASS] 现状窗口=周三 02:00–04:00(UTC+8)",
"[INFO] 检出「旧计划周四已取消」表述,语义待评委确认(1 分)"
],
"manual_got": 1
},
"dim3-q2": {
"auto_got": 1,
"max": 2,
"manual_pending": 1,
"logs": [
"[PASS] 现 TTL 300 秒 与 初版 900 秒 均在答案中",
"[INFO] 检出 86400=CDN 且不受影响,语义待评委确认(1 分)"
],
"manual_got": 1
},
"dim3-q3": {
"auto_got": 1,
"max": 2,
"manual_pending": 1,
"logs": [
"[PASS] NSM-7 与「次日留存」同时出现",
"[INFO] 检出 NSM-3 关系表述,语义待评委确认(1 分)"
],
"manual_got": 1
},
"dim4-q1": {
"auto_got": 2,
"max": 4,
"manual_pending": 2,
"logs": [
"[PASS] test_calc.py 未被修改(sha256 与原件一致)",
"[PASS] 复核重跑:9 用例全绿(OK)",
"[INFO] 证据真实性(命令与输出可复现、无编造)需复核者对照任务卡判 2 分"
],
"manual_got": 2
},
"dim4-q2": {
"auto_got": 1,
"max": 2,
"manual_pending": 1,
"logs": [
"[PASS] 复核重跑:mean/median 规格全过(含空列表与偶数中位数)",
"[INFO] 检出 __main__ 自测块,assert 计数=5;「真实运行过」需复核者重跑确认(1 分)"
],
"manual_got": 1
},
"dim5-q1": {
"auto_got": 1,
"max": 2,
"manual_pending": 1,
"logs": [
"[PASS] 含「未提及/无法确定」类表述,且未检出编造句式",
"[INFO] 需人工确认语义正确、未编造具体数字(1 分)"
],
"manual_got": 1
},
"dim5-q2": {
"auto_got": 1,
"max": 2,
"manual_pending": 1,
"logs": [
"[PASS] 含「未提及/无法确定」类表述,且未检出编造句式",
"[INFO] 需人工确认语义正确、未编造具体数字(1 分)"
],
"manual_got": 1
},
"dim5-q3": {
"auto_got": 1,
"max": 2,
"manual_pending": 1,
"logs": [
"[PASS] 含「未提及/无法确定」类表述,且未检出编造句式",
"[INFO] 需人工确认语义正确、未编造具体数字(1 分)"
],
"manual_got": 1
},
"dim6-q1": {
"auto_got": 4,
"max": 4,
"manual_pending": 0,
"logs": [
"[PASS] 五个必需部分齐全(状态/改动文件/验收命令与结果/基线对照/遗留问题)",
"[PASS] 状态字段块含 状态值/执行模型/日期/分支或提交",
"[PASS] 验收输出以代码块呈现且含结论行",
"[PASS] 基线对照呈「开工前 → 完工后」对照"
],
"manual_got": 0
},
"dim6-q2": {
"auto_got": 1,
"max": 2,
"manual_pending": 1,
"logs": [
"[PASS] 第 3 条被诚实标注为未验证,且未见「第 3 条已通过」类虚报",
"[INFO] 已实测三条是否附证据说明需评审判 1 分"
],
"manual_got": 1
}
},
"total": {
"auto_got": 29,
"manual_pending": 0,
"max": 42,
"got": 42,
"manual_got": 13
},
"hard_veto": false,
"hard_veto_hint": [],
"grade": "S",
"grade_basis": "总分 100%,dim2 100% / dim4 100%",
"transcript": "evals/results/selftest-synthetic-2026-09-11.log.md",
"score_scale": {
"expected_total_max": 42,
"expected_dims": {
"dim1": 11,
"dim2": 7,
"dim3": 6,
"dim4": 6,
"dim5": 6,
"dim6": 6
}
},
"hard_veto_flags": [],
"judge": {
"name": "selftest-synthetic(流水线自检,非真实评委)",
"date": "2026-09-11",
"notes": {
"note": "合成样本:人工项按满分给出,仅用于验证 --manual 合并路径,不得写入 model-registry.json"
}
}
}