feat: I-01模型能力探测脚本(只读settings+基线/思考/视觉三组+readonly自证)
This commit is contained in:
@@ -0,0 +1,409 @@
|
||||
#!/usr/bin/env python3
|
||||
"""I-01 模型能力自动探测:每晚自动 + 异常告警方向,最小闭环。
|
||||
|
||||
读 ~/.dsh/settings.yaml 枚举全部分组与模型,对每个模型跑三组探测:
|
||||
- 基线:一次最小对话请求(200 + 非空回复)
|
||||
- 思考:按模型的 thinkingFormat/reasoningEfforts 配置发带推理参数请求,验证 reasoning 字段
|
||||
- 视觉:内置 64px 纯色 PNG(base64 内嵌)问主色
|
||||
|
||||
铁律:脚本只读 settings.yaml(open(mode='r'),代码层面无写入路径);
|
||||
任何写入需求只进报告建议,由人执行。报告不打印 API key。
|
||||
|
||||
产出:tools/probe-reports/probe-YYYY-MM-DD.json(机读)+ 同名 .md(人读:表格 + 与上次 diff)。
|
||||
|
||||
用法:
|
||||
python tools/probe_models.py --all # 全量探测 + 写报告
|
||||
python tools/probe_models.py --all --compare # 全量探测 + 写报告 + 打印与上次对比
|
||||
python tools/probe_models.py --provider b # 只跑一组
|
||||
python tools/probe_models.py --check-readonly # 自证:grep 本源码无 settings 写入路径
|
||||
|
||||
Windows 计划任务(每晚 02:00)示例:
|
||||
schtasks /create /tn "DSH模型探测" /tr "python C:\\Users\\12914\\Desktop\\dsh\\tools\\probe_models.py --all" /sc daily /st 02:00
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import base64
|
||||
import datetime as _dt
|
||||
import io
|
||||
import json
|
||||
import os
|
||||
import struct
|
||||
import sys
|
||||
import time
|
||||
import urllib.request
|
||||
import zlib
|
||||
from pathlib import Path
|
||||
|
||||
HERE = Path(__file__).resolve().parent
|
||||
REPORT_DIR = HERE / "probe-reports"
|
||||
SETTINGS = Path(os.path.expanduser("~/.dsh/settings.yaml"))
|
||||
|
||||
TIMEOUT = 30 # 单请求超时(秒),失败隔离:单模型失败不中断全量
|
||||
RED_PNG_B64: str | None = None
|
||||
|
||||
|
||||
def red_png_b64() -> str:
|
||||
"""64px 纯红 PNG,纯标准库生成(zlib+struct),无外部文件依赖。"""
|
||||
global RED_PNG_B64
|
||||
if RED_PNG_B64 is None:
|
||||
w = h = 64
|
||||
raw = b"".join(b"\x00" + b"\xff\x00\x00" * w for _ in range(h))
|
||||
|
||||
def chunk(typ: bytes, data: bytes) -> bytes:
|
||||
c = typ + data
|
||||
return struct.pack(">I", len(data)) + c + struct.pack(">I", zlib.crc32(c) & 0xFFFFFFFF)
|
||||
|
||||
png = (b"\x89PNG\r\n\x1a\n"
|
||||
+ chunk(b"IHDR", struct.pack(">IIBBBBB", w, h, 8, 2, 0, 0, 0))
|
||||
+ chunk(b"IDAT", zlib.compress(raw))
|
||||
+ chunk(b"IEND", b""))
|
||||
RED_PNG_B64 = base64.b64encode(png).decode()
|
||||
return RED_PNG_B64
|
||||
|
||||
|
||||
def load_settings() -> dict:
|
||||
"""只读加载 settings.yaml(本函数是全脚本唯一的 settings 打开点,mode='r')。"""
|
||||
try:
|
||||
import yaml # type: ignore
|
||||
with open(SETTINGS, "r", encoding="utf-8") as fh:
|
||||
return yaml.safe_load(fh) or {}
|
||||
except ImportError:
|
||||
pass
|
||||
# 无 PyYAML 时的最小缩进解析:只取 providers 下 api/apiKeyEnv/baseURL/models 必要字段
|
||||
import re
|
||||
text = None
|
||||
with open(SETTINGS, "r", encoding="utf-8") as fh:
|
||||
text = fh.read()
|
||||
providers: dict = {}
|
||||
cur_p = cur_m = None
|
||||
cur: dict = {}
|
||||
for line in text.splitlines():
|
||||
m = re.match(r"^ ([a-z0-9-]+):\s*$", line)
|
||||
if m:
|
||||
if cur_p and cur:
|
||||
providers[cur_p] = cur
|
||||
cur_p, cur, cur_m = m.group(1), {"models": []}, None
|
||||
continue
|
||||
if cur_p is None:
|
||||
continue
|
||||
m2 = re.match(r"^ (displayName|apiKeyEnv|api|baseURL):\s*(.+?)\s*$", line)
|
||||
if m2:
|
||||
cur[m2.group(1)] = m2.group(2)
|
||||
continue
|
||||
m3 = re.match(r"^ - id:\s*(\S+)", line)
|
||||
if m3:
|
||||
cur_m = {"id": m3.group(1), "input": ["text"], "compat": {}, "reasoningEfforts": None}
|
||||
cur["models"].append(cur_m)
|
||||
continue
|
||||
if cur_m is not None:
|
||||
m4 = re.match(r"^ (thinkingFormat|supportsDeveloperRole):\s*(\S+)", line)
|
||||
if m4:
|
||||
cur_m["compat"][m4.group(1)] = m4.group(2)
|
||||
continue
|
||||
m5 = re.match(r"^ input:\s*\[(.+)\]", line)
|
||||
if m5:
|
||||
cur_m["input"] = [x.strip() for x in m5.group(1).split(",")]
|
||||
continue
|
||||
if re.match(r"^ reasoningEfforts:\s*$", line):
|
||||
cur_m["reasoningEfforts"] = {}
|
||||
continue
|
||||
m6 = re.match(r"^ (off|low|medium|high):\s*(\S+)?", line)
|
||||
if m6 and isinstance(cur_m.get("reasoningEfforts"), dict):
|
||||
cur_m["reasoningEfforts"][m6.group(1)] = (m6.group(2) or "").lower() == "null" and None or m6.group(2)
|
||||
if cur_p and cur:
|
||||
providers[cur_p] = cur
|
||||
return {"llm-pi-ai": {"providers": providers}}
|
||||
|
||||
|
||||
def iter_models(cfg: dict):
|
||||
provs = ((cfg.get("llm-pi-ai") or {}).get("providers") or {})
|
||||
for pname, p in provs.items():
|
||||
for m in p.get("models") or []:
|
||||
yield pname, p, m
|
||||
|
||||
|
||||
def _post(url: str, key: str, payload: dict, timeout: int = TIMEOUT) -> tuple[int, dict, str]:
|
||||
body = json.dumps(payload, ensure_ascii=False).encode()
|
||||
req = urllib.request.Request(
|
||||
url, data=body,
|
||||
headers={"Content-Type": "application/json", "Authorization": f"Bearer {key}"},
|
||||
method="POST",
|
||||
)
|
||||
t0 = time.time()
|
||||
try:
|
||||
with urllib.request.urlopen(req, timeout=timeout) as resp:
|
||||
raw = resp.read().decode("utf-8", "replace")
|
||||
return resp.status, json.loads(raw), f"{(time.time() - t0) * 1000:.0f}ms"
|
||||
except Exception as exc:
|
||||
ms = f"{(time.time() - t0) * 1000:.0f}ms"
|
||||
code = getattr(exc, "code", "")
|
||||
return -1, {"_error": f"{type(exc).__name__}: {str(exc)[:200]}", "_code": code}, ms
|
||||
|
||||
|
||||
def probe_baseline(base: str, key: str, model: str, api: str):
|
||||
if api == "openai-responses":
|
||||
payload = {"model": model, "input": "只回复:OK"}
|
||||
st, data, ms = _post(base.rstrip("/") + "/responses", key, payload)
|
||||
text = ""
|
||||
try:
|
||||
out = data.get("output") or []
|
||||
for item in out:
|
||||
for c in item.get("content") or []:
|
||||
text += c.get("text", "")
|
||||
except Exception:
|
||||
pass
|
||||
ok = st == 200 and bool(text.strip())
|
||||
return {"pass": ok, "ms": ms, "status": st,
|
||||
"err": "" if ok else str(data.get("_error", data))[:200]}
|
||||
payload = {"model": model, "messages": [{"role": "user", "content": "只回复:OK"}],
|
||||
"max_tokens": 16}
|
||||
st, data, ms = _post(base.rstrip("/") + "/chat/completions", key, payload)
|
||||
text = ""
|
||||
try:
|
||||
text = (data.get("choices") or [{}])[0].get("message", {}).get("content", "")
|
||||
except Exception:
|
||||
pass
|
||||
ok = st == 200 and bool((text or "").strip())
|
||||
return {"pass": ok, "ms": ms, "status": st,
|
||||
"err": "" if ok else str(data.get("_error", data))[:200]}
|
||||
|
||||
|
||||
def probe_thinking(base: str, key: str, model: str, api: str, compat: dict, efforts: dict | None):
|
||||
"""按 thinkingFormat 发推理参数;验证 reasoning 字段返回。无声明 → skip。"""
|
||||
fmt = (compat or {}).get("thinkingFormat")
|
||||
if not fmt or not efforts:
|
||||
return {"pass": None, "ms": "-", "status": "-", "err": "skip: 无 thinking 声明"}
|
||||
hi = (efforts or {}).get("high")
|
||||
extra: dict = {}
|
||||
if fmt == "zai":
|
||||
extra = {"enable_thinking": True, **({"reasoning_effort": hi} if hi else {})}
|
||||
elif fmt == "qwen":
|
||||
extra = {"enable_thinking": True, **({"reasoning_effort": hi or "xhigh"} if True else {})}
|
||||
else:
|
||||
extra = {"enable_thinking": True}
|
||||
if api == "openai-responses":
|
||||
payload = {"model": model, "input": "5-2=3,还剩3;2*3-3=3,新买3;3+3=? 只给出数字答案。", **extra}
|
||||
st, data, ms = _post(base.rstrip("/") + "/responses", key, payload)
|
||||
raw = json.dumps(data, ensure_ascii=False)
|
||||
has_reason = "reasoning" in raw
|
||||
text = ""
|
||||
try:
|
||||
for item in data.get("output") or []:
|
||||
for c in item.get("content") or []:
|
||||
text += c.get("text", "")
|
||||
except Exception:
|
||||
pass
|
||||
ok = st == 200 and has_reason and ("6" in text)
|
||||
return {"pass": ok, "ms": ms, "status": st,
|
||||
"err": "" if ok else f"reasoning={'有' if has_reason else '无'} 答={'6' in text} " + str(data.get("_error", ""))[:120]}
|
||||
payload = {"model": model,
|
||||
"messages": [{"role": "user", "content": "5-2=3,还剩3;2*3-3=3,新买3;3+3=? 只给出数字答案。"}],
|
||||
"max_tokens": 256, **extra}
|
||||
st, data, ms = _post(base.rstrip("/") + "/chat/completions", key, payload)
|
||||
raw = json.dumps(data, ensure_ascii=False)
|
||||
has_reason = "reasoning" in raw
|
||||
text = ""
|
||||
try:
|
||||
text = (data.get("choices") or [{}])[0].get("message", {}).get("content", "")
|
||||
except Exception:
|
||||
pass
|
||||
ok = st == 200 and has_reason and ("6" in text)
|
||||
return {"pass": ok, "ms": ms, "status": st,
|
||||
"err": "" if ok else f"reasoning={'有' if has_reason else '无'} 答={'6' in text} " + str(data.get("_error", ""))[:120]}
|
||||
|
||||
|
||||
def probe_vision(base: str, key: str, model: str, api: str, inputs: list):
|
||||
"""视觉:内嵌 64px 纯红 PNG 问主色。未声明 image → skip。"""
|
||||
if "image" not in (inputs or []):
|
||||
return {"pass": None, "ms": "-", "status": "-", "err": "skip: 未声明 image 输入"}
|
||||
img = {"type": "image_url", "image_url": {"url": "data:image/png;base64," + red_png_b64()}}
|
||||
if api == "openai-responses":
|
||||
payload = {"model": model, "input": [{"role": "user", "content": [
|
||||
{"type": "input_text", "text": "这张纯色图的主色是什么?只用一个词回答。"},
|
||||
{"type": "input_image", "image_url": "data:image/png;base64," + red_png_b64()},
|
||||
]}]}
|
||||
st, data, ms = _post(base.rstrip("/") + "/responses", key, payload)
|
||||
text = json.dumps(data, ensure_ascii=False)
|
||||
ok = st == 200 and ("红" in text or "red" in text.lower())
|
||||
return {"pass": ok, "ms": ms, "status": st,
|
||||
"err": "" if ok else str(data.get("_error", text[:200]))[:200]}
|
||||
payload = {"model": model, "max_tokens": 32, "messages": [
|
||||
{"role": "user", "content": [
|
||||
{"type": "text", "text": "这张纯色图的主色是什么?只用一个词回答。"},
|
||||
img,
|
||||
]}]}
|
||||
st, data, ms = _post(base.rstrip("/") + "/chat/completions", key, payload)
|
||||
text = ""
|
||||
try:
|
||||
text = (data.get("choices") or [{}])[0].get("message", {}).get("content", "")
|
||||
except Exception:
|
||||
pass
|
||||
ok = st == 200 and ("红" in (text or "") or "red" in (text or "").lower())
|
||||
return {"pass": ok, "ms": ms, "status": st,
|
||||
"err": "" if ok else (str(data.get("_error", text))[:200])}
|
||||
|
||||
|
||||
def run(provider_filter: str | None = None):
|
||||
cfg = load_settings()
|
||||
rows = []
|
||||
for pname, p, m in iter_models(cfg):
|
||||
if provider_filter and pname != provider_filter:
|
||||
continue
|
||||
mid = m.get("id", "?")
|
||||
api = p.get("api", "openai-completions")
|
||||
base = p.get("baseURL", "")
|
||||
key = os.environ.get(p.get("apiKeyEnv", ""), "")
|
||||
if not key:
|
||||
rows.append({"provider": pname, "model": mid, "api": api,
|
||||
"baseline": {"pass": None, "ms": "-", "status": "-", "err": "skip: 无 key(未设 " + p.get("apiKeyEnv", "?") + ")"},
|
||||
"thinking": {"pass": None, "ms": "-", "status": "-", "err": "skip: 无 key"},
|
||||
"vision": {"pass": None, "ms": "-", "status": "-", "err": "skip: 无 key"}})
|
||||
continue
|
||||
compat = m.get("compat") or {}
|
||||
efforts = m.get("reasoningEfforts")
|
||||
inputs = m.get("input") or ["text"]
|
||||
try:
|
||||
b = probe_baseline(base, key, mid, api)
|
||||
except Exception as exc:
|
||||
b = {"pass": False, "ms": "-", "status": -1, "err": f"{type(exc).__name__}: {exc}"[:200]}
|
||||
try:
|
||||
t = probe_thinking(base, key, mid, api, compat, efforts)
|
||||
except Exception as exc:
|
||||
t = {"pass": False, "ms": "-", "status": -1, "err": f"{type(exc).__name__}: {exc}"[:200]}
|
||||
try:
|
||||
v = probe_vision(base, key, mid, api, inputs)
|
||||
except Exception as exc:
|
||||
v = {"pass": False, "ms": "-", "status": -1, "err": f"{type(exc).__name__}: {exc}"[:200]}
|
||||
rows.append({"provider": pname, "model": mid, "api": api,
|
||||
"baseline": b, "thinking": t, "vision": v})
|
||||
return rows
|
||||
|
||||
|
||||
def verdict(row: dict) -> str:
|
||||
vals = [row[k].get("pass") for k in ("baseline", "thinking", "vision")]
|
||||
if all(v is True or v is None for v in vals) and any(v is True for v in vals):
|
||||
skips = sum(1 for v in vals if v is None)
|
||||
return "pass" if skips == 0 else "pass-skip"
|
||||
if any(v is False for v in vals):
|
||||
return "fail"
|
||||
return "skip"
|
||||
|
||||
|
||||
def write_reports(rows: list) -> tuple[Path, Path]:
|
||||
REPORT_DIR.mkdir(parents=True, exist_ok=True)
|
||||
day = _dt.date.today().isoformat()
|
||||
jpath = REPORT_DIR / f"probe-{day}.json"
|
||||
mpath = REPORT_DIR / f"probe-{day}.md"
|
||||
payload = {"date": day, "total": len(rows),
|
||||
"pass": 0, "fail": 0, "skip": 0,
|
||||
"rows": [{**r, "verdict": verdict(r)} for r in rows]}
|
||||
for r in payload["rows"]:
|
||||
if r["verdict"].startswith("pass"):
|
||||
payload["pass"] += 1
|
||||
elif r["verdict"] == "fail":
|
||||
payload["fail"] += 1
|
||||
else:
|
||||
payload["skip"] += 1
|
||||
# 与上次对比(能力变化告警方向)
|
||||
prev = sorted(REPORT_DIR.glob("probe-*.json"))
|
||||
prev = [p for p in prev if p.name != jpath.name]
|
||||
diff_lines: list[str] = []
|
||||
if prev:
|
||||
last = json.loads(prev[-1].read_text(encoding="utf-8"))
|
||||
by_key = {(r["provider"], r["model"]): r.get("verdict") for r in last.get("rows", [])}
|
||||
for r in payload["rows"]:
|
||||
old = by_key.get((r["provider"], r["model"]))
|
||||
if old and old != r["verdict"]:
|
||||
diff_lines.append(f'- {r["provider"]}/{r["model"]}: {old} → **{r["verdict"]}**')
|
||||
payload["diff_from"] = prev[-1].name if prev else None
|
||||
payload["diff"] = diff_lines
|
||||
jpath.write_text(json.dumps(payload, ensure_ascii=False, indent=2), encoding="utf-8")
|
||||
|
||||
alert = "⚠️ 能力变化告警" if diff_lines else "无能力变化"
|
||||
if payload["fail"]:
|
||||
alert += f'|{payload["fail"]} 个模型失败'
|
||||
lines = [f"# 模型探测报告 {day}", "", f"> {alert}", "",
|
||||
f"总 {payload['total']} · 通过 {payload['pass']} · 失败 {payload['fail']} · 全跳过 {payload['skip']}",
|
||||
"", "| 分组 | 模型 | 基线 | 思考 | 视觉 | 结论 |",
|
||||
"|---|---|---|---|---|---|"]
|
||||
sym = {True: "✅", False: "❌", None: "⏭️"}
|
||||
for r in payload["rows"]:
|
||||
b, t, v = r["baseline"], r["thinking"], r["vision"]
|
||||
lines.append(f"| {r['provider']} | {r['model']} | {sym[b['pass']]}{b['ms']} "
|
||||
f"| {sym[t['pass']]}{t['ms']} | {sym[v['pass']]}{v['ms']} | {r['verdict']} |")
|
||||
fails = [r for r in payload["rows"] if r["verdict"] == "fail"]
|
||||
if fails:
|
||||
lines += ["", "## 失败明细(错误摘要,不含 key)"]
|
||||
for r in fails:
|
||||
for k in ("baseline", "thinking", "vision"):
|
||||
if r[k].get("pass") is False:
|
||||
lines.append(f"- {r['provider']}/{r['model']} {k}: [{r[k].get('status')}] {r[k].get('err', '')[:150]}")
|
||||
if diff_lines:
|
||||
lines += ["", f"## 与上次对比({payload['diff_from']})"] + diff_lines
|
||||
else:
|
||||
lines += ["", "## 与上次对比", "无变化" if prev else "(首次报告,无基线)"]
|
||||
lines += ["", "## 配置建议(需人执行,本脚本不写配置)",
|
||||
"失败/变化的模型请对照 2026-09-06 手动结论复核后再改 settings.yaml。"]
|
||||
mpath.write_text("\n".join(lines) + "\n", encoding="utf-8")
|
||||
return jpath, mpath
|
||||
|
||||
|
||||
def check_readonly() -> int:
|
||||
"""自证无 settings 写入路径:源码中不得出现写模式 open/settings 写入调用。"""
|
||||
src = Path(__file__).read_text(encoding="utf-8")
|
||||
bad = []
|
||||
in_checker = False
|
||||
for i, line in enumerate(src.splitlines(), 1):
|
||||
s = line.strip()
|
||||
if s.startswith("def check_readonly"):
|
||||
in_checker = True
|
||||
elif in_checker and s.startswith("def "):
|
||||
in_checker = False
|
||||
if in_checker:
|
||||
continue # 跳过检查器自身的字面量(否则自指误报)
|
||||
if s.startswith("#") or s.startswith('"""') or s.startswith("'''"):
|
||||
continue
|
||||
if "SETTINGS" in line and ("\"w\"" in line or "'w'" in line or "\"a\"" in line or "'a'" in line):
|
||||
bad.append((i, line.strip()))
|
||||
if "settings.yaml" in line.lower() and (".write" in line or "yaml.dump" in line or "safe_dump" in line):
|
||||
bad.append((i, line.strip()))
|
||||
if bad:
|
||||
print("发现疑似写入路径:")
|
||||
for i, l in bad:
|
||||
print(f" L{i}: {l}")
|
||||
return 1
|
||||
opens = [l.strip() for l in src.splitlines() if "open(SETTINGS" in l]
|
||||
print("settings 打开点(应仅 mode='r'):")
|
||||
for l in opens:
|
||||
print(f" {l}")
|
||||
print("readonly-check: PASS(无写入路径)")
|
||||
return 0
|
||||
|
||||
|
||||
def main(argv: list[str]) -> int:
|
||||
ap = argparse.ArgumentParser(description="DSH 模型能力自动探测(只报告不写配置)")
|
||||
ap.add_argument("--all", action="store_true", help="全量探测")
|
||||
ap.add_argument("--provider", default="", help="只跑指定分组")
|
||||
ap.add_argument("--compare", action="store_true", help="打印与上次报告的对比")
|
||||
ap.add_argument("--check-readonly", action="store_true", help="自证无 settings 写入路径")
|
||||
args = ap.parse_args(argv)
|
||||
if args.check_readonly:
|
||||
return check_readonly()
|
||||
if not args.all and not args.provider:
|
||||
ap.print_help()
|
||||
return 2
|
||||
rows = run(args.provider or None)
|
||||
jp, mp = write_reports(rows)
|
||||
print(f"报告:{jp}\n报告:{mp}")
|
||||
if args.compare:
|
||||
data = json.loads(jp.read_text(encoding="utf-8"))
|
||||
print(f"对比基线:{data.get('diff_from') or '无(首次)'}")
|
||||
for d in data.get("diff", []) or ["无变化"]:
|
||||
print(f" {d}")
|
||||
fails = sum(1 for r in rows if verdict(r) == "fail")
|
||||
return 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
raise SystemExit(main(sys.argv[1:]))
|
||||
Reference in New Issue
Block a user