/* Generated by evaluation-site/build_research_data.py. */ window.K1412_EVIDENCE={"source":"research/completion-verification/evidence-ledger.md","source_sha256":"5658a539bbafc25e9ef946c5254c67a897e9eedc8f51cf5868891a5d8ec1c445","counts":{"pool":40,"audited":37,"core":26,"support":11,"context":3},"records":[{"id":"2606.09863","title":"From Confident Closing to Silent Failure","url":"https://arxiv.org/abs/2606.09863","depth":"core","theme":"verification","evidence":"tau2 共 9,876 条轨迹;airline/retail 失败轨迹中的 false success 为 45%/48%,telecom 为 3%;AppWorld 的 1,879 条明确完成声明中有 1,425 条 false success;任一 LLM judge 配置 AUROC 不超过约 0.65/0.54","supports":"Agent 自述不能作为完成证据;假完成需要独立状态检查","boundary":"tau2 域间控制方式混杂,telecom 只有 15 个 false-success 样本;AppWorld 只含会明确自评的架构"},{"id":"2406.12045","title":"tau-bench","url":"https://arxiv.org/abs/2406.12045","depth":"core","theme":"verification","evidence":"最终 reward 结合数据库/动作状态和输出信息;论文明确指出最终状态对 policy compliance 必要但不充分;`pass^k` 随重复快速下降","supports":"结果状态、策略遵守和跨次可靠性应分开","boundary":"客服域、模拟用户和手写策略不代表所有 Agent 工作流"},{"id":"2407.18901","title":"AppWorld","url":"https://arxiv.org/abs/2407.18901","depth":"core","theme":"verification","evidence":"750 个任务、9 个应用、457 个 API;任务测试要求预期状态变化成立,且预期/允许集合之外无额外变化","supports":"完成应同时检查目标效果和 collateral damage,并允许不同实现路径","boundary":"任务测试人工编写;不能覆盖所有潜在副作用和 UI/多 Agent 场景"},{"id":"2404.07972","title":"OSWorld","url":"https://arxiv.org/abs/2404.07972","depth":"core","theme":"verification","evidence":"369 个任务、302 个初始状态、134 个 evaluator;作者报告约 1,800 人时构建成本,并承认 evaluator 不能发现所有 latent side effects","supports":"高质量状态 verifier 昂贵且天然不完备","boundary":"桌面任务的成本不能直接外推到 API、代码或开放研究任务"},{"id":"2408.04682","title":"ToolSandbox","url":"https://arxiv.org/abs/2408.04682","depth":"core","theme":"verification","evidence":"1,032 个场景、34 个工具;用 milestone DAG 表示必要进展,用 minefield 表示禁止事件,允许部分进度和弹性顺序","supports":"评估可以路径有弹性,同时保留必要事件和禁止动作","boundary":"milestone/minefield 仍需领域专家反复编写;模拟工具会产生自己的错误"},{"id":"2504.08942","title":"AgentRewardBench","url":"https://arxiv.org/abs/2504.08942","depth":"core","theme":"verification","evidence":"1,302 条 web 轨迹、5 个 benchmark、4 个 Agent;官方规则 evaluator precision/recall/F1 为 83.8/55.9/67.1;最佳 LLM judge precision 低于 70%","supports":"规则判分偏低召回,LLM 判分偏低精度;不能把任一方当万能 oracle","boundary":"专家标签自身在抽样上约 89.3% 一致;结果限于 web 轨迹"},{"id":"2604.18240","title":"AJ-Bench","url":"https://arxiv.org/abs/2604.18240","depth":"core","theme":"verification","evidence":"155 个任务、516 条标注轨迹,覆盖 search/data/GUI;judge 使用工具后 Avg@3 明显提高,但不同域 FPR 仍为 8.77%-56.60%","supports":"environment-aware judge 比纯文本 judge 更好,但仍不能独立作为发布门禁","boundary":"judge、任务域和标注规模有限;更多 reasoning 不稳定单调增益"},{"id":"2606.22737","title":"GroundEval","url":"https://arxiv.org/abs/2606.22737","depth":"support","theme":"verification","evidence":"定义 evidence path、temporal、access、causal 和 absence contract;96 个合成企业问题中,合理回答可获 LLM judge 高分而确定性证据为 0","supports":"可用 typed evidence contract 取代“像答案”的文本判断","boundary":"合成数据、单模型和少量 case study;未做充分 judge 对照"},{"id":"2607.20531","title":"DynamicMCPBench","url":"https://arxiv.org/abs/2607.20531","depth":"core","theme":"verification","evidence":"1,845 个任务、121 个 MCP server;Tier-1 按效果确定性判分;750 个任务人工复核 grader agreement 74%;参考答案完全正确仅 79%;126 次 live 重放中 36% 相同、33% 漂移、32% 损坏","supports":"应按效果而非 gold path 判分;live 环境、参考答案和 grader 本身都需验证","boundary":"Tier-1 是保守下界,仍会漏掉等价路径;MCP 服务样本不能代表全部生产系统"},{"id":"2607.05391","title":"LLM-as-a-Verifier","url":"https://arxiv.org/abs/2607.05391","depth":"core","theme":"verification","evidence":"连续 logit 分数和重复采样提升 pairwise 与 best-of-N 选择;但失败轨迹的 verifier-progress 相关仍达 0.769","supports":"learned verifier 适合候选选择和进度信号,不足以单独证明完成;重复可减方差但不去偏","boundary":"依赖 logit 访问;主要是候选排序/RL 设置,不是生产发布验证"},{"id":"2603.03116","title":"Beyond Task Completion / PAE","url":"https://arxiv.org/abs/2603.03116","depth":"core","theme":"verification","evidence":"tau-bench retail/airline 中加入程序 gate 后,不同模型成功率由 40%-79% 降到 9%-58%;作者人工审计 131 条 airline corrupt success,judge precision 约 93.8%-95.2%","supports":"状态成功会掩盖程序违规;“完成”可因必要过程被破坏而无效","boundary":"单 benchmark,程序 gate 和部分判定依赖 GPT-5 judge;所有违规被同等二元处理"},{"id":"2607.02599","title":"AgentLTL","url":"https://arxiv.org/abs/2607.02599","depth":"core","theme":"verification","evidence":"FO-LTL 独立检查轨迹;block/warn 在 7 个模型中改善 5 个、恶化 2 个;soft-block 在 4/7 中最差","supports":"正确性和合规性是不同轴;门禁响应策略会制造拒绝或强制终止","boundary":"主要是合成算术工具和一个 repo-QA 扩展,不能证明开放环境收益"},{"id":"2607.18575","title":"RECEIPT","url":"https://arxiv.org/abs/2607.18575","depth":"core","theme":"verification","evidence":"同一 Claude Opus 自评 27 份 XSS 报告只有 10 个 TP;隔离、浏览器 verdict、可重放 PoC 和角色分离后,作者报告 30/30 TP;消融从 45% 到 100% precision","supports":"独立、不可游戏、可重放的 verifier 比 actor 自评可靠","boundary":"单模型、白盒 XSS 和特定 harness;高 precision 可能伴随未测量的 false negative"},{"id":"2607.07405","title":"Reason Less, Verify More","url":"https://arxiv.org/abs/2607.07405","depth":"core","theme":"verification","evidence":"tau2 airline 中确定性前置 gate 把成功从 29.6% 提到 42.0%,独立种子复现约 +12.3pp;gate precision 从 5% 到 100% 不等;retail 无正增益","supports":"可判定、高价值、工具允许违规的政策适合前置 gate","boundary":"任务与 gate 同源;只有 cancellation gate 明显承担收益;block 与结构化反馈效应未分离"},{"id":"2607.06624","title":"AgentLens","url":"https://arxiv.org/abs/2607.06624","depth":"support","theme":"verification","evidence":"16 个 Java 场景 × 2 persona;分开评 final result、instruction compliance、pitfalls、tool calls、pleasantness 和 formal verification;两个 judge 在 23% pairwise 比较上选不同赢家,18% 为各自偏向同家族","supports":"轨迹质量包含多个非冗余能力;judge 家族偏差和 provider/harness 故障会污染总分","boundary":"场景很小且 Java-only;正式人类一致性研究尚未完成;QI 是未加权代理分"},{"id":"2503.15223","title":"Are \"Solved Issues\" in SWE-bench Really Solved Correctly?","url":"https://arxiv.org/abs/2503.15223","depth":"core","theme":"protocol","evidence":"检查 877 个已通过 benchmark 的补丁;补跑全部开发者测试直接发现平均 7.8% 错误;PatchDiff 暴露 29.6% 行为差异,人工审计后估计约 11% plausible patch 不正确","supports":"通过有限测试不等于需求完成;应检查测试覆盖和行为差异","boundary":"差分测试也可能误报;部分 issue 规格本身不充分,人工结论有 uncertain 类"},{"id":"2410.06992","title":"SWE-Bench+","url":"https://arxiv.org/abs/2410.06992","depth":"support","theme":"protocol","evidence":"手工筛 251 个 SWE-Agent + GPT-4 通过项;报告 32.67% answer leakage、31.08% 弱测试,过滤后分数 12.47% 降到 3.97%","supports":"泄漏和弱测试可显著抬高 coding-agent 分数","boundary":"判定定义和方法存在较强主观性,作为前述 ICSE 审计的旁证而非主证据"},{"id":"2607.22368","title":"Protocol Validity / HackDetect","url":"https://arxiv.org/abs/2607.22368","depth":"core","theme":"protocol","evidence":"审计 15 个 benchmark、2,385 条轨迹;53 条手标样本上 LLM audit F1 为 0.84;发现任务到分数协议可能允许绕过预期能力","supports":"benchmark 要证明“目标能力对得分仍然必要”,不能只检查数据和 metric","boundary":"部分系统只预选可疑样本,不能解释为全体 prevalence;依赖同一类 LLM 审计"},{"id":"2606.26300","title":"The Verification Horizon","url":"https://arxiv.org/abs/2606.26300","depth":"support","theme":"protocol","evidence":"coding-agent reward 中,过程监控抑制测试篡改/查答案等 hack;报告内部 SWE、前端和长程 reward 的多组改进","supports":"verifier 必须随 Agent 能力和攻击面共同演化,最终测试不能覆盖所有过程 hack","boundary":"大量结果来自 Qwen 内部 benchmark 和训练管线,外部可复核性有限"},{"id":"2607.13085","title":"Baselines Before Architecture","url":"https://arxiv.org/abs/2607.13085","depth":"core","theme":"protocol","evidence":"XBOW 104 个任务、两次运行;普通 Codex 基线随 GPT-5→5.2→5.5 从 67.3→79.8→92.3,模型匹配后剩余架构增益明显缩小","supports":"必须用同模型、同预算的简单基线分离模型能力和架构贡献","boundary":"公开系统并非完全同预算,且公开安全 benchmark 可能有训练污染"},{"id":"2607.22520","title":"The Regression Tax","url":"https://arxiv.org/abs/2607.22520","depth":"support","theme":"protocol","evidence":"5,832 个 office task-condition run;skills 带来 553 个 gain、324 个 regression,回归抵消约 59% 增益;Bonferroni 后仅 3/18 显著","supports":"新组件必须同时报告 gain 和 regression,平均值会藏住负迁移","boundary":"主要显著结果集中在一个模型/benchmark;部分原 grader 有 artifact"},{"id":"2508.11027","title":"Hell or High Water","url":"https://arxiv.org/abs/2508.11027","depth":"core","theme":"recovery","evidence":"830 题、4,450 个函数;第一工具不可用但保证有最多三步替代路径;多模型从 clean 到 failure 条件下降约 22-30pp,53%-66% 失败在工具搜索","supports":"外部故障后的替代路径发现是独立能力,显式错误和已知可行路径也不保证恢复","boundary":"Spider 派生、工具库很大但路径短;结果受 prompt 和工具检索接口影响"},{"id":"2606.05806","title":"ToolMaze","url":"https://arxiv.org/abs/2606.05806","depth":"core","theme":"recovery","evidence":"controlled DAG 覆盖显式/隐式 × 瞬时/永久故障;有 hint 时平均恢复率为 81.44%、27.68%、38.12%、17.58%;模型规模对 recovery 的拟合斜率远低于 task success","supports":"隐式永久故障最难;恢复能力不会随一般任务能力等速增长","boundary":"程序生成 DAG 和故障模板,不是开放生产环境;规模相关不证明因果"},{"id":"2606.21409","title":"Don't Blindly Trust It","url":"https://arxiv.org/abs/2606.21409","depth":"core","theme":"recovery","evidence":"HotpotQA/FEVER matched loop;错误或冲突反馈使结果远差于无反馈;首步预测器在 recoverable conflict 上 AUC 跌到 0.516;requery 对不同模型可正可负","supports":"检出坏反馈只是一层过滤,最终表现受 fallback 能力限制","boundary":"主要是 QA 和模拟 corruption,只有 GPT-4o 是闭源强模型;calculator 只是 pilot"},{"id":"2607.04623","title":"R2Act","url":"https://arxiv.org/abs/2607.04623","depth":"core","theme":"recovery","evidence":"302 个 Kubernetes incident;最强 RAG 的 root service 识别 91.4%-99.7%,恢复动作有效率仅 36.8%-60.3%;Qwen live replay 恢复 146/302","supports":"诊断、动作有效性和真实状态恢复是三个不同阶段","boundary":"单一微服务系统和有限故障类型;组织权限/策略未充分覆盖"},{"id":"2509.25370","title":"AgentDebug","url":"https://arxiv.org/abs/2509.25370","depth":"core","theme":"recovery","evidence":"200 条失败轨迹;检测 exact step 45%,step+module 31.3%,全部精确 24.3%;从定位点重跑在三套任务上提高恢复","supports":"定位错误步骤可比从头盲目改写更有效,但自动根因判断仍弱","boundary":"标注者 κ=0.55;样本和域小,部分图表分母不够清晰"},{"id":"2607.18754","title":"AgentDebugX","url":"https://arxiv.org/abs/2607.18754","depth":"core","theme":"recovery","evidence":"Who&When 184 条轨迹上 strict agent+step 由 21.7% 提到 28.8%;GAIA 73 条失败一次重跑修复 13 条,三个基线为 4-6 条;约 1.6× 单次读取 Token","supports":"结构化归因可提高后续恢复,但严格根因定位仍远未解决","boundary":"所有方法可见参考答案;GAIA 只用一个 policy model,比较的是完整 recipe,未隔离归因效应"},{"id":"2607.19338","title":"CodeRescue","url":"https://arxiv.org/abs/2607.19338","depth":"core","theme":"recovery","evidence":"约 27,300 次代码任务尝试;固定 reflect/replan/escalate 恢复率为 27.5/45.3/68.6%,router 为 81.7%;可恢复失败中 cheap-only/escalation-only/both 为 28/45/27%","supports":"不存在统一最优恢复动作;根据失败上下文和成本路由有直接价值","boundary":"单次路由决策、代码 benchmark 和特定模型;高成功率不代表安全恢复"},{"id":"2607.17641","title":"VRR-Stop","url":"https://arxiv.org/abs/2607.17641","depth":"core","theme":"recovery","evidence":"在 verifier/repair 不匹配 stress 中,固定五轮 repair 从 0.700 降到 0.116,VRR-Stop 为 0.722;55% 正确计划被破坏","supports":"verify-repair loop 必须有停止/保护条件,更多轮次可能主动伤害","boundary":"巨大差值来自刻意 stress;正常条件下相对 no-repair 的提升较小且 CI 可跨 0"},{"id":"2303.11366","title":"Reflexion","url":"https://arxiv.org/abs/2303.11366","depth":"support","theme":"recovery","evidence":"HumanEval Rust50 消融:base 60、无测试自反思 52、仅测试 60、测试+反思 68;自生成测试在 MBPP 有更多 false positive","supports":"反思依赖可靠外部反馈;无证据自反思会损伤正确答案","boundary":"较早模型、特定 coding/QA/ALFWorld 设置,不能代表当前 Agent"},{"id":"2305.11738","title":"CRITIC","url":"https://arxiv.org/abs/2305.11738","depth":"support","theme":"recovery","evidence":"外部工具反馈在 QA/math/toxicity 有增益;数学中修复 32.2% 初始错误,同时错误修改 14.3%,并使原本正确项下降 4.3%","supports":"外部反馈有用,但 correction 不是单调改进,必须复验","boundary":"多为短程任务,工具质量和 prompt 强烈影响结果"},{"id":"2604.18847","title":"Human-Guided Harm Recovery","url":"https://arxiv.org/abs/2604.18847","depth":"support","theme":"recovery","evidence":"775 个场景、20 名标注者、1,130 个计划偏好;在 50 个 OSWorld harm 场景上,人偏好 reward/rubric 重排计划","supports":"有伤害后的恢复涉及规范和人类偏好,不能只优化任务成功","boundary":"假设已有外部 harm classifier;评的是计划偏好,不是客观状态恢复"},{"id":"2606.01416","title":"Self-Healing Agentic Orchestrators","url":"https://arxiv.org/abs/2606.01416","depth":"support","theme":"recovery","evidence":"100 个作者构造任务、确定性工具和故障模板中报告 98.8% 成功;live model 部分只有 15 个任务,三种方法最终都 100%;作者明确承认 benchmark 与 recovery policy 可能同源","supports":"`detect-diagnose-recover-verify` 和 bounded budget 是可用系统词汇","boundary":"不能用其高分证明生产自愈;合成任务、受控故障、弱 baseline 和饱和 live set 外部效度低"},{"id":"2512.07850","title":"SABER","url":"https://arxiv.org/abs/2512.07850","depth":"core","theme":"reliability","evidence":"tau-bench 中 mutating action 仅占 14%-18%,但与失败强相关;confirmation/reflection/verifier 可提高多数组合,full 组合在部分 retail 设置反而回归","supports":"高副作用动作值得前置集中审查;保护组件也需报告回归","boundary":"mutation 与失败的回归关系可能受路径距离混杂;只研究确认型 safeguard"},{"id":"2506.07982","title":"tau2-bench","url":"https://arxiv.org/abs/2506.07982","depth":"core","theme":"reliability","evidence":"新增共享世界状态和用户工具,区分 no-user、oracle-plan 和 dual-control;telecom `pass^1` 约 34%;用户模拟器 critical error 为 6%,retail/airline 为 12%/13%","supports":"指导用户、协调共享状态和纯自主执行是不同能力;user simulator 也是误差源","boundary":"三个客服域,专家-新手差距未显式建模;域扩展仍依赖专家"},{"id":"2401.13178","title":"AgentBoard","url":"https://arxiv.org/abs/2401.13178","depth":"support","theme":"reliability","evidence":"1,013 个任务、9 个环境;把 success 与人工定义 progress rate 分开","supports":"部分进展有诊断价值,不能只保留二元成功","boundary":"progress 仍依赖任务作者定义,且模拟环境与当前模型不同"},{"id":"2410.10934","title":"Agent-as-a-Judge","url":"https://arxiv.org/abs/2410.10934","depth":"support","theme":"reliability","evidence":"DevAI 55 个任务上,Agent judge 可主动读取环境;作者专家评估耗时 58/86.5 小时,Agent judge 与人类对齐约 84%-90%","supports":"judge 主动取证比只看最终文本更合理,也可降低人工成本","boundary":"样本小、主观任务多;对齐率不足以支持自动发布"},{"id":"2304.05128","title":"Self-Debugging","url":"https://arxiv.org/abs/2304.05128","depth":"context","theme":"context","evidence":"","supports":"","boundary":"早期代码自调试背景;本轮已有 Reflexion、CRITIC 和更新的恢复对照承担该结论"},{"id":"2307.13854","title":"WebArena","url":"https://arxiv.org/abs/2307.13854","depth":"context","theme":"context","evidence":"","supports":"","boundary":"作为交互环境历史锚点;完成证明由 tau/AppWorld/OSWorld 的更直接状态设计承担"},{"id":"2310.06770","title":"SWE-bench","url":"https://arxiv.org/abs/2310.06770","depth":"context","theme":"context","evidence":"","supports":"","boundary":"作为 coding benchmark 历史锚点;测试误认证由 2503.15223 的专门审计承担"}]};