zcbot/scripts/diag_glm_empty_probe.py

135 lines
5.1 KiB
Python

"""定层 glm.pro52 空响应(失败面板 kind=empty_response 主角,11/15/近7天)。
背景:assistant 轮既无 tool_calls 又无正文,被 run loop 当正常收尾静默 done。
与 opus48 的「tool_use 漏成正文」(narrated,见 diag_narrated_toolcall_2a1bc25d)
是不同现象——要判 glm 到底是①真吐空(tc=0 且 content=0)、②漏成正文(narrated)、
还是③正常(能回结构化 tool_calls,线上那次只是瞬态)。
做法:build_agent(resume=True) 复原真实 system prompt + 全套工具 schema + **完整多轮
上下文**(空响应发生在深层,单条 user 消息复现不了 —— 这是与 2a1bc25d 探针的关键差别),
直接打 glm.pro52,stream / non-stream 各打一发。只读 DB,不跑 loop 不写库。
输出 ASCII(Windows GBK 安全),完整正文写文件。
用法(生产机跑,本机 import litellm 卡 ~20min):
.venv/Scripts/python.exe scripts/diag_glm_empty_probe.py [task_id] [user_id] [model_profile]
默认 = 35744bea(近7天 empty 5 次、最可复现)。model_profile 缺省用 task 自身档(glm.pro52),
传第三参可换档做对照(如 deepseek_v4.pro 看是否 glm 独有)。
"""
import os
import sys
from pathlib import Path
from uuid import UUID
ROOT = Path(__file__).resolve().parent.parent
env = ROOT / ".env"
if env.exists():
for line in env.read_text(encoding="utf-8").splitlines():
line = line.strip()
if line and not line.startswith("#") and "=" in line:
k, v = line.split("=", 1)
os.environ.setdefault(k.strip(), v.strip().strip('"').strip("'"))
os.environ.setdefault("LITELLM_LOCAL_MODEL_COST_MAP", "True")
sys.path.insert(0, str(ROOT))
import litellm # noqa: E402
from core.agent_builder import build_agent # noqa: E402
# 默认锁定 35744bea(近7天 empty 5 次、最可复现);可 argv 覆盖。
TASK_ID = sys.argv[1] if len(sys.argv) > 1 else "35744bea-49c0-41c8-86be-8d0c5c05bd32"
USER_ID = UUID(sys.argv[2]) if len(sys.argv) > 2 else UUID("3cc07a22-f667-47f3-9617-e7d8a1a60cf0")
MODEL_OVERRIDE = sys.argv[3] if len(sys.argv) > 3 else None
_tag = (MODEL_OVERRIDE or "self").replace(".", "_").replace("/", "_")
OUT = open(ROOT / "scripts" / f"_glm_empty_probe_{TASK_ID[:8]}_{_tag}.txt", "w", encoding="utf-8")
def w(*a):
print(*a, file=OUT)
agent, session, sid, tstate, wd = build_agent(
user_id=USER_ID, session_id=TASK_ID, resume=True,
model_name=MODEL_OVERRIDE,
)
schemas = agent.executor.schemas()
caps = agent.caps
llm = agent.llm
# 复原 loop 下一轮真会送的完整消息(system + 全部历史)—— 空响应就发生在这个上下文上。
messages = list(session.messages)
sys_txt = next((m["content"] for m in messages if m.get("role") == "system"), "")
roles = [m.get("role") for m in messages]
w("[SETUP]")
w(f" task={TASK_ID} model_id={caps.model_id} profile={caps.family}.{caps.variant}")
w(f" api_base={llm.api_base} temp={caps.optimal_temperature} "
f"parallel_tools={caps.parallel_tools}")
w(f" system_prompt chars={len(sys_txt)} tools={len(schemas)}")
w(f" messages count={len(messages)} last_role={roles[-1] if roles else None}")
w(f" role tail={roles[-6:]}")
w("")
kwargs_base = dict(
model=caps.model_id,
api_base=llm.api_base,
api_key=llm.api_key,
messages=messages,
tools=schemas,
temperature=caps.optimal_temperature,
timeout=300,
)
if caps.parallel_tools:
kwargs_base["parallel_tool_calls"] = True
stdout_lines = []
def probe(tag, stream):
w(f"========== {tag} ==========")
try:
if stream:
chunks = list(litellm.completion(
**kwargs_base, stream=True,
stream_options={"include_usage": True}))
resp = litellm.stream_chunk_builder(chunks)
else:
resp = litellm.completion(**kwargs_base)
except Exception as e:
w(f"[FAIL] {type(e).__name__}: {str(e)[:300]}")
stdout_lines.append(f"{tag}: FAIL {type(e).__name__}")
return
msg = resp.choices[0].message
tc = msg.tool_calls or []
content = msg.content or ""
fr = resp.choices[0].finish_reason
u = resp.usage
w(f" finish_reason={fr}")
w(f" usage prompt={u.prompt_tokens} completion={u.completion_tokens}")
w(f" tool_calls count={len(tc)}")
for i, t in enumerate(tc):
w(f" [{i}] {t.function.name}({(t.function.arguments or '')[:200]})")
w(f" content chars={len(content)}")
w(" content:")
w(content[:2000])
w("")
# 三态判定:EMPTY(真吐空)/ NARRATED(工具意图漏成正文)/ STRUCTURED / plain-text
narrated = (not tc) and ("`" in content) and ("\n- " in content or "No result received" in content)
if tc:
verdict = "STRUCTURED tool_calls"
elif not content.strip():
verdict = "EMPTY (tc=0 content=0) <-- 复现空响应"
elif narrated:
verdict = "NARRATED-as-text"
else:
verdict = "plain-text-no-tools"
stdout_lines.append(f"{tag}: {verdict} (finish={fr}, tc={len(tc)}, content={len(content)}c)")
probe("NON-STREAM", stream=False)
probe("STREAM", stream=True)
OUT.close()
for ln in stdout_lines:
print(ln)
print(f"full -> scripts/_glm_empty_probe_{TASK_ID[:8]}_{_tag}.txt")