[ { "id": "2607.17331", "arxiv_id": "2607.17331", "source": "arxiv", "source_id": "arxiv:2607.17331", "title": "Agentic ERP: Multi-Agent Large Language Model Architecture for Autonomous Enterprise Resource Planning", "url": "https://arxiv.org/abs/2607.17331", "pdf_url": "https://arxiv.org/pdf/2607.17331", "published": "2026-07-19", "updated": "2026-07-19", "authors": [ "Zhihao Liu", "Tianyu Wang", "Xi Vincent Wang", "Lihui Wang" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "agent-safety", "multi-agent", "planning", "tool-use", "workflow-agent", "world-model" ], "score": 24, "relevance": "high", "primary_query": "llm-agent", "matched_queries": [ "llm-agent", "multi-agent-llm", "planning-agent" ] }, { "id": "2607.17528", "arxiv_id": "2607.17528", "source": "arxiv", "source_id": "arxiv:2607.17528", "title": "Can AI Agents Really Complete RTL-to-GDS? Lessons from Benchmarking Tool-Interactive EDA Workflows", "url": "https://arxiv.org/abs/2607.17528", "pdf_url": "https://arxiv.org/pdf/2607.17528", "published": "2026-07-22", "updated": "2026-07-22", "authors": [ "Jinyuan Deng", "Zhengrui Chen", "Xufeng Wei", "Tianyu Xing", "Chenyi Wen", "Qi Sun", "Cheng Zhuo" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "coding-agent", "planning", "tool-use", "workflow-agent" ], "score": 23, "relevance": "high", "primary_query": "ai-agent", "matched_queries": [ "ai-agent", "coding-agent", "llm-agent" ] }, { "id": "2607.18366", "arxiv_id": "2607.18366", "source": "arxiv", "source_id": "arxiv:2607.18366", "title": "Operational Hallucination and Safety Drift in AI Agents", "url": "https://arxiv.org/abs/2607.18366", "pdf_url": "https://arxiv.org/pdf/2607.18366", "published": "2026-07-20", "updated": "2026-07-20", "authors": [ "Shasha Yu", "Fiona Carroll", "Barry L. Bentley" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "agent-safety", "computer-use", "planning", "reasoning", "tool-use", "world-model" ], "score": 22, "relevance": "high", "primary_query": "agentic-ai", "matched_queries": [ "agentic-ai", "ai-agent", "autonomous-agent-llm", "tool-use" ] }, { "id": "2607.10994", "arxiv_id": "2607.10994", "source": "arxiv", "source_id": "arxiv:2607.10994", "title": "A Multi-Agent Framework for Zero-Dimensional Reduced-Order Model Planning", "url": "https://arxiv.org/abs/2607.10994", "pdf_url": "https://arxiv.org/pdf/2607.10994", "published": "2026-07-12", "updated": "2026-07-12", "authors": [ "Bingteng Sun", "Hao Yin", "Yiling Chen", "Renjie Xiao", "Lei Xie", "Shanyou Wang", "Ruonan Wang", "Shubao Chen", "Qingzong Xu", "Lin Lu", "Qiang Du", "Junqiang Zhu" ], "categories": [ "cs.LG" ], "topics": [ "agent-evaluation", "computer-use", "embodied-agent", "multi-agent", "planning", "rag", "reasoning", "tool-use", "workflow-agent" ], "score": 22, "relevance": "high", "primary_query": "multi-agent-llm", "matched_queries": [ "multi-agent-llm", "planning-agent", "rag-agent" ] }, { "id": "2607.15781", "arxiv_id": "2607.15781", "source": "arxiv", "source_id": "arxiv:2607.15781", "title": "AgentFAIR: A Multi-Agent Collaborative Framework for FAIRness Evaluation of Geospatial Datasets", "url": "https://arxiv.org/abs/2607.15781", "pdf_url": "https://arxiv.org/pdf/2607.15781", "published": "2026-07-24", "updated": "2026-07-24", "authors": [ "Ming Chen", "Pranav Pai" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "agent-safety", "coding-agent", "multi-agent", "planning", "rag", "tool-use" ], "score": 21, "relevance": "high", "primary_query": "multi-agent-llm", "matched_queries": [ "multi-agent-llm" ] }, { "id": "2607.15434", "arxiv_id": "2607.15434", "source": "arxiv", "source_id": "arxiv:2607.15434", "title": "Coercion and Deception in AI-to-AI Management: An Agentic Benchmark of Unprompted Escalation", "url": "https://arxiv.org/abs/2607.15434", "pdf_url": "https://arxiv.org/pdf/2607.15434", "published": "2026-07-22", "updated": "2026-07-22", "authors": [ "Jasmine Brazilek", "Maheep Chaudhary", "Zoe Lu", "Miles Tidmarsh" ], "categories": [ "cs.MA" ], "topics": [ "agent-evaluation", "multi-agent", "tool-use" ], "score": 21, "relevance": "high", "primary_query": "agent-evaluation", "matched_queries": [ "agent-evaluation", "ai-agent", "multi-agent-llm" ] }, { "id": "2607.14642", "arxiv_id": "2607.14642", "source": "arxiv", "source_id": "arxiv:2607.14642", "title": "MCPEvol-Bench: Benchmarking LLM Agent Performance Across Dynamic Evolutions of MCP Servers", "url": "https://arxiv.org/abs/2607.14642", "pdf_url": "https://arxiv.org/pdf/2607.14642", "published": "2026-07-16", "updated": "2026-07-16", "authors": [ "Huanxi Liu", "Kun Hu", "Jiaqi Liao", "Qiang Wang", "Pengfei Qian", "YuanZhao Zhai", "Dawei Feng", "Bo Ding", "Huaimin Wang" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "planning", "rag", "reasoning", "tool-use", "workflow-agent" ], "score": 21, "relevance": "high", "primary_query": "llm-agent", "matched_queries": [ "llm-agent", "planning-agent", "tool-use" ] }, { "id": "2607.21106", "arxiv_id": "2607.21106", "source": "arxiv", "source_id": "arxiv:2607.21106", "title": "AttriMem: Attribution-Guided Process Feedback for Agent Memory Learning", "url": "https://arxiv.org/abs/2607.21106", "pdf_url": "https://arxiv.org/pdf/2607.21106", "published": "2026-07-23", "updated": "2026-07-23", "authors": [ "Qinfeng Li", "Yuntai Bao", "Xinyan Yu", "Hongze Chen", "Wenqi Zhang", "Xuhong Zhang" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "computer-use", "memory", "planning", "rag", "reasoning", "tool-use" ], "score": 20, "relevance": "high", "primary_query": "agent-memory", "matched_queries": [ "agent-memory", "llm-agent" ] }, { "id": "2607.20709", "arxiv_id": "2607.20709", "source": "arxiv", "source_id": "arxiv:2607.20709", "title": "NVIDIA-labs OO Agents: Native Python Object-Oriented Agents", "url": "https://arxiv.org/abs/2607.20709", "pdf_url": "https://arxiv.org/pdf/2607.20709", "published": "2026-07-22", "updated": "2026-07-22", "authors": [ "Paul Furgale", "Severin Klingler", "James Nolan", "Matt Staats", "Gaia Di Lorenzo", "Elisa Martinez Abad", "Christian Schüller", "Razvan Dinu", "Alessio Devoto", "Pascal Berard", "Gal Kaplun", "Elad Sarafian", "Riccardo Roveri", "Leon Derczynski", "Ricardo Silveira Cabral" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "coding-agent", "memory", "rag", "reasoning", "tool-use", "workflow-agent" ], "score": 20, "relevance": "high", "primary_query": "ai-agent", "matched_queries": [ "ai-agent", "coding-agent" ] }, { "id": "2607.18039", "arxiv_id": "2607.18039", "source": "arxiv", "source_id": "arxiv:2607.18039", "title": "Evidence-in-the-Loop: Trace-Driven Optimization for Customer-Service LLM Agents", "url": "https://arxiv.org/abs/2607.18039", "pdf_url": "https://arxiv.org/pdf/2607.18039", "published": "2026-07-20", "updated": "2026-07-20", "authors": [ "Chunming Wu", "Dafei Qiu", "Congde Yuan", "Charles Quan", "Jun Wu", "Suipeng Li", "Mo Wu", "Gavin Xie", "Hope Chen", "Max Yao" ], "categories": [ "cs.IR" ], "topics": [ "agent-evaluation", "agent-safety", "computer-use", "memory", "rag", "tool-use", "workflow-agent" ], "score": 20, "relevance": "high", "primary_query": "agentic-ai", "matched_queries": [ "agentic-ai", "llm-agent", "rag-agent" ] }, { "id": "2607.17545", "arxiv_id": "2607.17545", "source": "arxiv", "source_id": "arxiv:2607.17545", "title": "Retain or Consolidate? Budget-Dependent Operator Selection for Language Agent Memory", "url": "https://arxiv.org/abs/2607.17545", "pdf_url": "https://arxiv.org/pdf/2607.17545", "published": "2026-07-20", "updated": "2026-07-20", "authors": [ "Qingcan Kang", "Mingyang Liu", "Shixiong Kai", "Kaichao Liang", "Zhentao Tang", "Yuqi Cui", "Tao Zhong", "Mingxuan Yuan" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "agent-safety", "memory", "rag", "tool-use" ], "score": 20, "relevance": "high", "primary_query": "agent-memory", "matched_queries": [ "agent-memory", "language-agent" ] }, { "id": "2607.15660", "arxiv_id": "2607.15660", "source": "arxiv", "source_id": "arxiv:2607.15660", "title": "ToolVerse: Unlocking Massive Environments and Long-Horizon Tasks for Agentic Reinforcement Learning", "url": "https://arxiv.org/abs/2607.15660", "pdf_url": "https://arxiv.org/pdf/2607.15660", "published": "2026-07-17", "updated": "2026-07-17", "authors": [ "Shuaiyu Zhou", "Fengpeng Yue", "Zengjie Hu", "Yuanzhe Shen", "Chenyang Zhang", "feng hong", "Cao Liu", "Ke Zeng" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "planning", "reasoning", "tool-use", "world-model" ], "score": 20, "relevance": "high", "primary_query": "agent-evaluation", "matched_queries": [ "agent-evaluation", "llm-agent", "tool-use" ] }, { "id": "2607.15079", "arxiv_id": "2607.15079", "source": "arxiv", "source_id": "arxiv:2607.15079", "title": "BrainPilot: Automating Brain Discovery with Agentic Research", "url": "https://arxiv.org/abs/2607.15079", "pdf_url": "https://arxiv.org/pdf/2607.15079", "published": "2026-07-17", "updated": "2026-07-17", "authors": [ "Haoxuan Li", "Tianci Gao", "Jianhe Li", "Yang Fan", "Runze Shi", "Weiran Wang", "Tianxiang Zhao", "Zezhao Wu", "Xiaoyang Jiang", "Qihui Zhang", "Jia Li", "Xiao Xiao", "Kai Du", "Xiaoxuan Jia", "Chao Xie", "Lu Mi" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "multi-agent", "rag", "reasoning", "tool-use", "workflow-agent" ], "score": 20, "relevance": "high", "primary_query": "ai-agent", "matched_queries": [ "ai-agent", "tool-use" ] }, { "id": "2607.14651", "arxiv_id": "2607.14651", "source": "arxiv", "source_id": "arxiv:2607.14651", "title": "MemPoison: Uncovering Persistent Memory Threats and Structural Blind Spots in LLM Agents", "url": "https://arxiv.org/abs/2607.14651", "pdf_url": "https://arxiv.org/pdf/2607.14651", "published": "2026-07-16", "updated": "2026-07-16", "authors": [ "Jifeng Gao", "Kang Xia", "Yi Zhang", "Xiaobin Hong", "Mingkai Lin", "Xingshen Wei", "Wenzhong Li", "Sanglu Lu" ], "categories": [ "cs.CR" ], "topics": [ "agent-evaluation", "agent-safety", "memory", "rag", "tool-use" ], "score": 20, "relevance": "high", "primary_query": "llm-agent", "matched_queries": [ "llm-agent" ] }, { "id": "2607.08448", "arxiv_id": "2607.08448", "source": "arxiv", "source_id": "arxiv:2607.08448", "title": "Harness VLA: Steering Frozen VLAs into Reliable Manipulation Primitives via Memory-Guided Agents", "url": "https://arxiv.org/abs/2607.08448", "pdf_url": "https://arxiv.org/pdf/2607.08448", "published": "2026-07-15", "updated": "2026-07-15", "authors": [ "Yixian Zhang", "Huanming Zhang", "Feng Gao", "Xiao Li", "Zhihao Liu", "Chunyang Zhu", "Jiaxing Qiu", "Yuchen Yan", "Jiyuan Liu", "Wenhao Tang", "Zhengru Fang", "Yi Nie", "Changxu Wei", "Yu Wang", "Wenbo Ding", "Chao Yu" ], "categories": [ "cs.RO" ], "topics": [ "coding-agent", "computer-use", "embodied-agent", "memory", "planning", "rag", "reasoning", "tool-use" ], "score": 20, "relevance": "high", "primary_query": "coding-agent", "matched_queries": [ "coding-agent" ] }, { "id": "2607.09653", "arxiv_id": "2607.09653", "source": "arxiv", "source_id": "arxiv:2607.09653", "title": "VEXAIoT: Autonomous IoT Vulnerability EXploitation using AI Agents", "url": "https://arxiv.org/abs/2607.09653", "pdf_url": "https://arxiv.org/pdf/2607.09653", "published": "2026-07-10", "updated": "2026-07-10", "authors": [ "Katherine Swinea", "Kshitiz Aryal", "Lopamudra Praharaj", "Maanak Gupta" ], "categories": [ "cs.CR" ], "topics": [ "agent-evaluation", "agent-safety", "multi-agent", "planning", "rag", "reasoning", "tool-use", "workflow-agent" ], "score": 20, "relevance": "high", "primary_query": "multi-agent-llm", "matched_queries": [ "multi-agent-llm", "planning-agent" ] }, { "id": "2607.22083", "arxiv_id": "2607.22083", "source": "arxiv", "source_id": "arxiv:2607.22083", "title": "Nanbeige4.2-3B: Unlocking Agentic Capabilities in a Compact Mode", "url": "https://arxiv.org/abs/2607.22083", "pdf_url": "https://arxiv.org/pdf/2607.22083", "published": "2026-07-24", "updated": "2026-07-24", "authors": [ "Nanbeige Lab", ":", "Chen Yang", "Chengrui Huang", "Fufeng Lan", "Hanhui Chen", "Hao Zhou", "Huatong Song", "Jiaqi Cao", "Jiaying Zhu", "Jinlin Niu", "Kai Wang", "Lisheng Huang", "Qiliang Liang", "Ran Le", "Ruixiang Feng", "Shuang Sun", "Tao Gu", "Tao Zhang", "Tianyu Luo", "Yang Song", "Yun Xing", "Yuntao Wen", "Ziyao Xu", "Zongchao Chen", "et al. (1 additional authors not shown)" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "agent-safety", "coding-agent", "planning", "reasoning", "tool-use", "workflow-agent" ], "score": 19, "relevance": "high", "primary_query": "agent-evaluation", "matched_queries": [ "agent-evaluation", "coding-agent", "tool-use" ] }, { "id": "2607.21920", "arxiv_id": "2607.21920", "source": "arxiv", "source_id": "arxiv:2607.21920", "title": "Systematic Literature Reviews With Two Multi-Agentic Systems And Human-In-The-Loop", "url": "https://arxiv.org/abs/2607.21920", "pdf_url": "https://arxiv.org/pdf/2607.21920", "published": "2026-07-23", "updated": "2026-07-23", "authors": [ "Zexin Ren", "Zixuan Zhao", "Qiyun Li", "Yawen Wu", "Lanjing Wang", "Renjie Luo", "Yi Xu", "Qing Guo", "Jin Shi", "En Xie", "Feifang Hu", "Qian Shi" ], "categories": [ "stat.AP" ], "topics": [ "agent-evaluation", "multi-agent", "rag", "tool-use" ], "score": 19, "relevance": "high", "primary_query": "llm-agent", "matched_queries": [ "llm-agent", "multi-agent-llm" ] }, { "id": "2607.20064", "arxiv_id": "2607.20064", "source": "arxiv", "source_id": "arxiv:2607.20064", "title": "PRO-LONG: Programmatic Memory Enables Long-Horizon Reasoning", "url": "https://arxiv.org/abs/2607.20064", "pdf_url": "https://arxiv.org/pdf/2607.20064", "published": "2026-07-23", "updated": "2026-07-23", "authors": [ "Alexis Fox", "Junlin Wang", "Paul Rosu", "Bhuwan Dhingra" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "coding-agent", "memory", "planning", "rag", "reasoning", "tool-use" ], "score": 19, "relevance": "high", "primary_query": "coding-agent", "matched_queries": [ "coding-agent", "llm-agent" ] }, { "id": "2607.20121", "arxiv_id": "2607.20121", "source": "arxiv", "source_id": "arxiv:2607.20121", "title": "OpenSkillRisk: Benchmarking Agent Safety When Using Real-World Risky Third-Party Skills", "url": "https://arxiv.org/abs/2607.20121", "pdf_url": "https://arxiv.org/pdf/2607.20121", "published": "2026-07-23", "updated": "2026-07-23", "authors": [ "Qiyuan Liu", "Tingfeng Hui", "Kun Zhan", "Kaike Zhang", "Ning Miao" ], "categories": [ "cs.CL" ], "topics": [ "agent-evaluation", "agent-safety", "rag", "reasoning", "tool-use" ], "score": 19, "relevance": "high", "primary_query": "agent-safety", "matched_queries": [ "agent-safety" ] }, { "id": "2607.14573", "arxiv_id": "2607.14573", "source": "arxiv", "source_id": "arxiv:2607.14573", "title": "Alipay-PIBench: A Realistic Payment Integration Benchmark for Coding Agents", "url": "https://arxiv.org/abs/2607.14573", "pdf_url": "https://arxiv.org/pdf/2607.14573", "published": "2026-07-21", "updated": "2026-07-21", "authors": [ "Shiyu Ying", "Xuejie Cao", "Yingfan Ma", "Yuanhao Dong", "Wenyu Chen", "Bowen Song", "Lin Zhu" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "agent-safety", "coding-agent", "computer-use", "rag", "tool-use" ], "score": 19, "relevance": "high", "primary_query": "coding-agent", "matched_queries": [ "coding-agent" ] }, { "id": "2607.18485", "arxiv_id": "2607.18485", "source": "arxiv", "source_id": "arxiv:2607.18485", "title": "Trusted Credentials, Untrusted Behavior: Benchmarking LLM-Agent Security in High-Performance Computing", "url": "https://arxiv.org/abs/2607.18485", "pdf_url": "https://arxiv.org/pdf/2607.18485", "published": "2026-07-20", "updated": "2026-07-20", "authors": [ "Jie Li" ], "categories": [ "cs.CR" ], "topics": [ "agent-evaluation", "agent-safety", "planning", "rag", "tool-use", "workflow-agent", "world-model" ], "score": 19, "relevance": "high", "primary_query": "agent-safety", "matched_queries": [ "agent-safety", "llm-agent", "planning-agent" ] }, { "id": "2607.19430", "arxiv_id": "2607.19430", "source": "arxiv", "source_id": "arxiv:2607.19430", "title": "ChannelGuard: Safe Models Do Not Compose into Safe Multi-Agent Systems", "url": "https://arxiv.org/abs/2607.19430", "pdf_url": "https://arxiv.org/pdf/2607.19430", "published": "2026-07-20", "updated": "2026-07-20", "authors": [ "Elias Hossain", "Md Mehedi Hasan Nipu", "Fatema Tuj Johora Faria", "Tasfia Nuzhat Ornee", "Maleeha Sheikh" ], "categories": [ "cs.CR" ], "topics": [ "agent-evaluation", "agent-safety", "memory", "multi-agent", "planning", "reasoning", "tool-use" ], "score": 19, "relevance": "high", "primary_query": "multi-agent-llm", "matched_queries": [ "multi-agent-llm" ] }, { "id": "2607.17437", "arxiv_id": "2607.17437", "source": "arxiv", "source_id": "arxiv:2607.17437", "title": "Empirical Grounding Improves the Realism of LLM Agents Simulating Human Behavior During Disruptions", "url": "https://arxiv.org/abs/2607.17437", "pdf_url": "https://arxiv.org/pdf/2607.17437", "published": "2026-07-19", "updated": "2026-07-19", "authors": [ "Chen Xia", "Zexi Kuang", "Yuqing Hu" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "memory", "planning", "rag", "reasoning", "world-model" ], "score": 19, "relevance": "high", "primary_query": "llm-agent", "matched_queries": [ "llm-agent", "planning-agent" ] }, { "id": "2607.15535", "arxiv_id": "2607.15535", "source": "arxiv", "source_id": "arxiv:2607.15535", "title": "Symbolic Predicate-Guided Language Agents for Inverse Design of Perovskite Oxides", "url": "https://arxiv.org/abs/2607.15535", "pdf_url": "https://arxiv.org/pdf/2607.15535", "published": "2026-07-16", "updated": "2026-07-16", "authors": [ "Dong Hyeon Mok", "Seoin Back", "Victor Fung", "Guoxiang Hu" ], "categories": [ "cond-mat.mtrl-sci" ], "topics": [ "agent-evaluation", "computer-use", "multi-agent", "rag", "reasoning" ], "score": 19, "relevance": "high", "primary_query": "language-agent", "matched_queries": [ "language-agent", "llm-agent", "multi-agent-llm" ] }, { "id": "2607.14989", "arxiv_id": "2607.14989", "source": "arxiv", "source_id": "arxiv:2607.14989", "title": "OmniaBench: Benchmarking General AI Agents Across Diverse Scenarios", "url": "https://arxiv.org/abs/2607.14989", "pdf_url": "https://arxiv.org/pdf/2607.14989", "published": "2026-07-16", "updated": "2026-07-16", "authors": [ "Chengyu Shen", "Yujie Fu", "Gangtao Xin", "Yanheng Hou", "Wenlong Fei", "Guojie Zhu", "Jiawei Li", "Hongcheng Gao", "Runming He", "Zhen Hao Wong", "Meiyi Qiang", "Hao Liang", "Zhao Cao", "Hao Jiang", "Chong Chen", "Wentao Zhang" ], "categories": [ "cs.CL" ], "topics": [ "agent-evaluation", "planning", "rag", "tool-use" ], "score": 19, "relevance": "high", "primary_query": "agent-evaluation", "matched_queries": [ "agent-evaluation", "ai-agent" ] }, { "id": "2607.13591", "arxiv_id": "2607.13591", "source": "arxiv", "source_id": "arxiv:2607.13591", "title": "Memory as a Controlled Process: Learned Adaptive Memory Management for LLM Agents", "url": "https://arxiv.org/abs/2607.13591", "pdf_url": "https://arxiv.org/pdf/2607.13591", "published": "2026-07-15", "updated": "2026-07-15", "authors": [ "Eric Hanchen Jiang", "Zhi Zhang", "Yuchen Wu", "Levina Li", "Dong Liu", "Xiao Liang", "Rui Sun", "Yubei Li", "Edward Sun", "Haozheng Luo", "Zhaolu Kang", "Aylin Caliskan", "Kai-Wei Chang", "Ying Nian Wu" ], "categories": [ "cs.CL" ], "topics": [ "agent-evaluation", "memory", "planning", "rag" ], "score": 19, "relevance": "high", "primary_query": "planning-agent", "matched_queries": [ "planning-agent" ] }, { "id": "2607.06624", "arxiv_id": "2607.06624", "source": "arxiv", "source_id": "arxiv:2607.06624", "title": "AgentLens: Production-Assessed Trajectory Reviews for Coding Agent Evaluation", "url": "https://arxiv.org/abs/2607.06624", "pdf_url": "https://arxiv.org/pdf/2607.06624", "published": "2026-07-14", "updated": "2026-07-14", "authors": [ "Andrey Podivilov", "Vadim Lomshakov", "Sergey Savin", "Matvei Startsev", "Roman Pozharskiy", "Maksim Parshin", "Sergey Nikolenko" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "coding-agent", "planning", "tool-use" ], "score": 19, "relevance": "high", "primary_query": "agent-evaluation", "matched_queries": [ "agent-evaluation", "coding-agent" ] }, { "id": "2607.12267", "arxiv_id": "2607.12267", "source": "arxiv", "source_id": "arxiv:2607.12267", "title": "Track, Rank, Crack: Epistemic Working Memory Scales Multi-Hop Reasoning in Language Agents", "url": "https://arxiv.org/abs/2607.12267", "pdf_url": "https://arxiv.org/pdf/2607.12267", "published": "2026-07-13", "updated": "2026-07-13", "authors": [ "Ning Liu" ], "categories": [ "cs.LG" ], "topics": [ "agent-evaluation", "memory", "rag", "reasoning", "tool-use" ], "score": 19, "relevance": "high", "primary_query": "language-agent", "matched_queries": [ "language-agent", "tool-use" ] }, { "id": "2607.20531", "arxiv_id": "2607.20531", "source": "arxiv", "source_id": "arxiv:2607.20531", "title": "DynamicMCPBench: A Trace-Grounded, Effect-Scored Benchmark for LLM Agents over Live MCP Servers", "url": "https://arxiv.org/abs/2607.20531", "pdf_url": "https://arxiv.org/pdf/2607.20531", "published": "2026-07-10", "updated": "2026-07-10", "authors": [ "Jerzy Kamiński", "Ilya Galyukshev", "Artem Kuznetsov", "Sergey Chuprin", "Kirill Redko", "Aidar Shumbalov", "Anna Kalyuzhnaya" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "memory", "rag", "tool-use" ], "score": 19, "relevance": "high", "primary_query": "llm-agent", "matched_queries": [ "llm-agent", "tool-use" ] }, { "id": "2607.08960", "arxiv_id": "2607.08960", "source": "arxiv", "source_id": "arxiv:2607.08960", "title": "Eluna: An Agentic LLM System for Automating Warehouse Operations with Reasoning and Task Execution", "url": "https://arxiv.org/abs/2607.08960", "pdf_url": "https://arxiv.org/pdf/2607.08960", "published": "2026-07-09", "updated": "2026-07-09", "authors": [ "Ning Liu", "Kalle Kujanpää", "Zhaoxuan Zhu", "P Aditya Sreekar", "Kaiwen Liu", "Chuanneng Sun", "Jorge Marchena Menendez", "Matthew Bales", "Tianyu Yang", "Shahnawaz Alam", "Rose Yu", "Baoyuan Liu", "Kristina Klinkner", "Shervin Malmasi" ], "categories": [ "cs.LG" ], "topics": [ "agent-evaluation", "computer-use", "memory", "multi-agent", "reasoning" ], "score": 19, "relevance": "high", "primary_query": "multi-agent-llm", "matched_queries": [ "multi-agent-llm" ] }, { "id": "2607.22368", "arxiv_id": "2607.22368", "source": "arxiv", "source_id": "arxiv:2607.22368", "title": "Do Agent Benchmarks Measure Capability? Protocol Validity in the Age of Agentic AI", "url": "https://arxiv.org/abs/2607.22368", "pdf_url": "https://arxiv.org/pdf/2607.22368", "published": "2026-07-24", "updated": "2026-07-24", "authors": [ "Jiaqi Shao", "Hanck Chen", "Wei Zhang", "Maxm Pan", "Bing Luo" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "coding-agent", "planning", "tool-use" ], "score": 18, "relevance": "high", "primary_query": "agent-evaluation", "matched_queries": [ "agent-evaluation", "agentic-ai" ] }, { "id": "2607.21217", "arxiv_id": "2607.21217", "source": "arxiv", "source_id": "arxiv:2607.21217", "title": "ICAE-Bench: Evaluating Coding Agents as Interactive Project Builders", "url": "https://arxiv.org/abs/2607.21217", "pdf_url": "https://arxiv.org/pdf/2607.21217", "published": "2026-07-23", "updated": "2026-07-23", "authors": [ "Zhongyuan Peng", "Dan Huang", "Chuyu Zhang", "Caijun Xu", "Changyi Xiao", "Shibo Hong", "David Lo", "Lin Qiu", "Xuezhi Cao", "Jiyuan He", "Yixin Cao" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "coding-agent", "computer-use", "planning", "tool-use", "workflow-agent", "world-model" ], "score": 18, "relevance": "high", "primary_query": "coding-agent", "matched_queries": [ "coding-agent", "tool-use" ] }, { "id": "2607.20972", "arxiv_id": "2607.20972", "source": "arxiv", "source_id": "arxiv:2607.20972", "title": "Delivery, Not Storage: Cue-Anchored Working Memory as a Harness Property for Coding Agents", "url": "https://arxiv.org/abs/2607.20972", "pdf_url": "https://arxiv.org/pdf/2607.20972", "published": "2026-07-23", "updated": "2026-07-23", "authors": [ "Swapnanil Saha" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "coding-agent", "memory", "planning", "rag", "tool-use" ], "score": 18, "relevance": "high", "primary_query": "coding-agent", "matched_queries": [ "coding-agent" ] }, { "id": "2607.19913", "arxiv_id": "2607.19913", "source": "arxiv", "source_id": "arxiv:2607.19913", "title": "JANUS: Foreseeing Latent Risk for Long-Horizon Agent Safety", "url": "https://arxiv.org/abs/2607.19913", "pdf_url": "https://arxiv.org/pdf/2607.19913", "published": "2026-07-22", "updated": "2026-07-22", "authors": [ "Yuan Xiong", "Linji Hao", "Shizhu He", "Yequan Wang", "Lijun Li" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "agent-safety", "multi-agent", "planning", "rag", "tool-use", "world-model" ], "score": 18, "relevance": "high", "primary_query": "agent-safety", "matched_queries": [ "agent-safety", "tool-use" ] }, { "id": "2607.19595", "arxiv_id": "2607.19595", "source": "arxiv", "source_id": "arxiv:2607.19595", "title": "Twin Agent: Context Residual Compression for Privilege Separated Agents", "url": "https://arxiv.org/abs/2607.19595", "pdf_url": "https://arxiv.org/pdf/2607.19595", "published": "2026-07-21", "updated": "2026-07-21", "authors": [ "Zhanhao Hu", "Dennis Jacob", "Xiao Huang", "Zhaorun Chen", "Bo Li", "David Wagner" ], "categories": [ "cs.CR" ], "topics": [ "agent-evaluation", "agent-safety", "coding-agent", "planning", "reasoning", "tool-use" ], "score": 18, "relevance": "high", "primary_query": "coding-agent", "matched_queries": [ "coding-agent", "llm-agent", "tool-use" ] }, { "id": "2607.18754", "arxiv_id": "2607.18754", "source": "arxiv", "source_id": "arxiv:2607.18754", "title": "AgentDebugX: An Open-Source Toolkit for Failure Observability, Attribution, and Recovery in LLM Agents", "url": "https://arxiv.org/abs/2607.18754", "pdf_url": "https://arxiv.org/pdf/2607.18754", "published": "2026-07-21", "updated": "2026-07-21", "authors": [ "Kunlun Zhu", "Xuyan Ye", "Zhiguang Han", "Yuchen Zhao", "Bingxuan Li", "Weijia Zhang", "Muxin Tian", "Xiangru Tang", "Pan Lu", "James Zou", "Jiaxuan You", "Heng Ji" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "computer-use", "memory", "tool-use", "workflow-agent" ], "score": 18, "relevance": "high", "primary_query": "llm-agent", "matched_queries": [ "llm-agent" ] }, { "id": "2607.17288", "arxiv_id": "2607.17288", "source": "arxiv", "source_id": "arxiv:2607.17288", "title": "SAGA: Synthetic Agentic Graph Architecture for Temporal Benchmark Generation", "url": "https://arxiv.org/abs/2607.17288", "pdf_url": "https://arxiv.org/pdf/2607.17288", "published": "2026-07-19", "updated": "2026-07-19", "authors": [ "Jiacheng Ding", "Xiaofei Zhang" ], "categories": [ "cs.DB" ], "topics": [ "agent-evaluation", "agent-safety", "rag" ], "score": 18, "relevance": "high", "primary_query": "llm-agent", "matched_queries": [ "llm-agent", "rag-agent" ] }, { "id": "2607.16851", "arxiv_id": "2607.16851", "source": "arxiv", "source_id": "arxiv:2607.16851", "title": "AgentBrew: Lifelong Knowledge Brewing from Strong Teachers to Weak LLM Agents", "url": "https://arxiv.org/abs/2607.16851", "pdf_url": "https://arxiv.org/pdf/2607.16851", "published": "2026-07-18", "updated": "2026-07-18", "authors": [ "Yangqin Jiang", "Chao Huang" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "coding-agent", "computer-use", "memory", "tool-use" ], "score": 18, "relevance": "high", "primary_query": "llm-agent", "matched_queries": [ "llm-agent", "tool-use" ] }, { "id": "2607.15715", "arxiv_id": "2607.15715", "source": "arxiv", "source_id": "arxiv:2607.15715", "title": "Behavioral Controllability of Agentic Models for Information Extraction: From Fixed Workflows to Reflective Agents", "url": "https://arxiv.org/abs/2607.15715", "pdf_url": "https://arxiv.org/pdf/2607.15715", "published": "2026-07-17", "updated": "2026-07-17", "authors": [ "Lujia Zhang", "Xingzhou Chen", "Hongwei Feng" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "memory", "rag", "reasoning", "tool-use", "workflow-agent" ], "score": 18, "relevance": "high", "primary_query": "llm-agent", "matched_queries": [ "llm-agent" ] }, { "id": "2607.14277", "arxiv_id": "2607.14277", "source": "arxiv", "source_id": "arxiv:2607.14277", "title": "Multi-Head Latent Control: A Unified Interface for LLM Agent Decision Making", "url": "https://arxiv.org/abs/2607.14277", "pdf_url": "https://arxiv.org/pdf/2607.14277", "published": "2026-07-15", "updated": "2026-07-15", "authors": [ "Amirhosein Ghasemabadi", "Ruichen Chen", "Bahador Rashidi", "Di Niu" ], "categories": [ "cs.CL" ], "topics": [ "agent-evaluation", "rag", "reasoning", "tool-use" ], "score": 18, "relevance": "high", "primary_query": "llm-agent", "matched_queries": [ "llm-agent", "tool-use" ] }, { "id": "2607.14178", "arxiv_id": "2607.14178", "source": "arxiv", "source_id": "arxiv:2607.14178", "title": "ReasFlow: Assisting Reasoning-Centric Scientific Discovery in Applied Mathematics via a Knowledge-Based Multi-Agent System", "url": "https://arxiv.org/abs/2607.14178", "pdf_url": "https://arxiv.org/pdf/2607.14178", "published": "2026-07-15", "updated": "2026-07-15", "authors": [ "Yutong He", "Daibo Li", "Guohong Li", "Jiahe Geng", "Zhengyang Huang", "Can Ren", "Zekun Zhang", "Yifan Liu", "Shuchen Zhu", "Hengrui Zhang", "Boao Kong", "Ming Sun", "Shu Li", "Chenyi Li", "Jiang Hu", "Kun Yuan", "Zaiwen Wen", "Pingwen Zhang" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "multi-agent", "rag", "reasoning" ], "score": 18, "relevance": "high", "primary_query": "ai-agent", "matched_queries": [ "ai-agent", "autonomous-agent-llm" ] }, { "id": "2607.11751", "arxiv_id": "2607.11751", "source": "arxiv", "source_id": "arxiv:2607.11751", "title": "When Local Monitors Miss Compositional Harm: Diagnosing Distributed Backdoors in Multi-Agent Systems", "url": "https://arxiv.org/abs/2607.11751", "pdf_url": "https://arxiv.org/pdf/2607.11751", "published": "2026-07-13", "updated": "2026-07-13", "authors": [ "Yibo Hu", "Ren Wang" ], "categories": [ "cs.CR" ], "topics": [ "agent-evaluation", "agent-safety", "coding-agent", "multi-agent", "rag", "tool-use" ], "score": 18, "relevance": "high", "primary_query": "multi-agent-llm", "matched_queries": [ "multi-agent-llm", "tool-use" ] }, { "id": "2607.10608", "arxiv_id": "2607.10608", "source": "arxiv", "source_id": "arxiv:2607.10608", "title": "The Compliance Trap: Diagnosing How AI Agents Consume Conflicting Memory", "url": "https://arxiv.org/abs/2607.10608", "pdf_url": "https://arxiv.org/pdf/2607.10608", "published": "2026-07-12", "updated": "2026-07-12", "authors": [ "Yixiong Chen", "Xinyi Bai", "Alan Yuille" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "computer-use", "memory", "planning", "rag", "tool-use" ], "score": 18, "relevance": "high", "primary_query": "ai-agent", "matched_queries": [ "ai-agent" ] }, { "id": "2607.09101", "arxiv_id": "2607.09101", "source": "arxiv", "source_id": "arxiv:2607.09101", "title": "Multi-Agent LLM Collaboration for Unit Test Generation via Human-Testing-Inspired Workflows", "url": "https://arxiv.org/abs/2607.09101", "pdf_url": "https://arxiv.org/pdf/2607.09101", "published": "2026-07-10", "updated": "2026-07-10", "authors": [ "Quanjun Zhang", "Ye Shang", "Siqi Gu", "Jianyi Zhou", "Chunrong Fang", "Zhenyu Chen", "Liang Xiao" ], "categories": [ "cs.SE" ], "topics": [ "agent-evaluation", "coding-agent", "multi-agent", "planning", "rag", "reasoning", "tool-use", "workflow-agent" ], "score": 18, "relevance": "high", "primary_query": "multi-agent-llm", "matched_queries": [ "multi-agent-llm" ] }, { "id": "2607.08681", "arxiv_id": "2607.08681", "source": "arxiv", "source_id": "arxiv:2607.08681", "title": "SolarChain-Eval: A Physics-Constrained Benchmark for Trustworthy Economic Agents in Decentralized Energy Markets", "url": "https://arxiv.org/abs/2607.08681", "pdf_url": "https://arxiv.org/pdf/2607.08681", "published": "2026-07-09", "updated": "2026-07-09", "authors": [ "Shilin Ou", "Yifan Xu", "Luyao Zhang" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "agent-safety", "planning", "tool-use" ], "score": 18, "relevance": "high", "primary_query": "agent-evaluation", "matched_queries": [ "agent-evaluation", "agentic-ai", "autonomous-agent-llm" ] }, { "id": "2607.19653", "arxiv_id": "2607.19653", "source": "arxiv", "source_id": "arxiv:2607.19653", "title": "PerfAgent: Profiler-Guided Iterative Refinement for Repository-Level Code Optimization", "url": "https://arxiv.org/abs/2607.19653", "pdf_url": "https://arxiv.org/pdf/2607.19653", "published": "2026-07-21", "updated": "2026-07-21", "authors": [ "Ryan Deng", "Yuanzhe Liu", "Bastian Lipka", "Yao Ma", "Xuhao Chen", "Tim Kaler", "Jatin Ganhotra" ], "categories": [ "cs.SE" ], "topics": [ "agent-evaluation", "coding-agent", "computer-use", "reasoning", "tool-use", "workflow-agent" ], "score": 17, "relevance": "high", "primary_query": "coding-agent", "matched_queries": [ "coding-agent", "llm-agent" ] }, { "id": "2607.18659", "arxiv_id": "2607.18659", "source": "arxiv", "source_id": "arxiv:2607.18659", "title": "Broken Gates: Re-evaluating Web Bot Defenses in the Age of LLM Agents", "url": "https://arxiv.org/abs/2607.18659", "pdf_url": "https://arxiv.org/pdf/2607.18659", "published": "2026-07-20", "updated": "2026-07-20", "authors": [ "Behzad Ousat", "Nikita Turkmen", "Lalchandra Rampersaud", "Dillan Bailey", "Amin Kharraz" ], "categories": [ "cs.CR" ], "topics": [ "agent-evaluation", "agent-safety", "computer-use", "tool-use", "workflow-agent" ], "score": 17, "relevance": "high", "primary_query": "llm-agent", "matched_queries": [ "llm-agent", "web-gui-agent" ] }, { "id": "2607.19433", "arxiv_id": "2607.19433", "source": "arxiv", "source_id": "arxiv:2607.19433", "title": "The Chronos Vulnerability: A Taxonomy of Temporal Persistence and Memory-Based Deception in Agentic AI", "url": "https://arxiv.org/abs/2607.19433", "pdf_url": "https://arxiv.org/pdf/2607.19433", "published": "2026-07-20", "updated": "2026-07-20", "authors": [ "Om Narayan", "Ramkinker Singh", "Praveen Baskar" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "agent-safety", "memory", "planning", "workflow-agent" ], "score": 17, "relevance": "high", "primary_query": "agentic-ai", "matched_queries": [ "agentic-ai" ] }, { "id": "2607.12463", "arxiv_id": "2607.12463", "source": "arxiv", "source_id": "arxiv:2607.12463", "title": "Function-Aware Fill-in-the-Middle as Mid-Training for Coding Agent Foundation Models", "url": "https://arxiv.org/abs/2607.12463", "pdf_url": "https://arxiv.org/pdf/2607.12463", "published": "2026-07-19", "updated": "2026-07-19", "authors": [ "Yubo Wang", "Jiarong Liang", "Yuxuan Zhang", "Xuye Liu", "Cong Wei", "Yuyu Zhang", "Ping Nie", "Wenhu Chen" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "coding-agent", "reasoning", "tool-use" ], "score": 17, "relevance": "high", "primary_query": "coding-agent", "matched_queries": [ "coding-agent", "function-calling", "tool-use" ] }, { "id": "2607.15246", "arxiv_id": "2607.15246", "source": "arxiv", "source_id": "arxiv:2607.15246", "title": "ARMOR++: Agentic Orchestration of a Multi-Domain Primitive Set for Transferable Attacks on Deepfake Detectors", "url": "https://arxiv.org/abs/2607.15246", "pdf_url": "https://arxiv.org/pdf/2607.15246", "published": "2026-07-16", "updated": "2026-07-16", "authors": [ "Christos Korgialas", "Gabriel Lee Jun Rong", "Dion Jia Xu Ho", "Pai Chet Ng", "Xiaoxiao Miao", "Konstantinos N. Plataniotis" ], "categories": [ "cs.CV" ], "topics": [ "agent-evaluation", "multi-agent", "rag" ], "score": 17, "relevance": "high", "primary_query": "agent-evaluation", "matched_queries": [ "agent-evaluation", "multi-agent-llm" ] }, { "id": "2606.31650", "arxiv_id": "2606.31650", "source": "arxiv", "source_id": "arxiv:2606.31650", "title": "ECHO: Prune To Act, Trace To Learn With Selective Turn Memory In Agentic RL", "url": "https://arxiv.org/abs/2606.31650", "pdf_url": "https://arxiv.org/pdf/2606.31650", "published": "2026-07-14", "updated": "2026-07-14", "authors": [ "Zijun Xie", "Binbin Zheng", "Enlei Gong", "Jihua Liu", "Yuyang You", "Lingfeng Liu", "Jiayao Tang", "Guanqun Zhao", "Aoqi Hu", "Zeyu Chen" ], "categories": [ "cs.LG" ], "topics": [ "agent-evaluation", "memory", "planning", "tool-use" ], "score": 17, "relevance": "high", "primary_query": "language-agent", "matched_queries": [ "language-agent" ] }, { "id": "2607.11141", "arxiv_id": "2607.11141", "source": "arxiv", "source_id": "arxiv:2607.11141", "title": "NextFund: A Unified Performance Tracking Platform for Agentic Portfolio Management", "url": "https://arxiv.org/abs/2607.11141", "pdf_url": "https://arxiv.org/pdf/2607.11141", "published": "2026-07-13", "updated": "2026-07-13", "authors": [ "Changlun Li", "Peixian Ma", "Qiqi Duan", "Zhenyu Lin", "Peineng Wu" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "agent-safety", "multi-agent", "tool-use" ], "score": 17, "relevance": "high", "primary_query": "multi-agent-llm", "matched_queries": [ "multi-agent-llm" ] }, { "id": "2607.11042", "arxiv_id": "2607.11042", "source": "arxiv", "source_id": "arxiv:2607.11042", "title": "BackendForge: Benchmarking Agentic End-to-End Code Generation with Backend Services", "url": "https://arxiv.org/abs/2607.11042", "pdf_url": "https://arxiv.org/pdf/2607.11042", "published": "2026-07-12", "updated": "2026-07-12", "authors": [ "Yuzhe Guo", "Mengzhou Wu", "Yuan Cao", "Jialei Wei", "Dezhi Ran", "Wei Yang", "Tao Xie" ], "categories": [ "cs.SE" ], "topics": [ "agent-evaluation", "coding-agent", "tool-use" ], "score": 17, "relevance": "high", "primary_query": "coding-agent", "matched_queries": [ "coding-agent" ] }, { "id": "2607.10789", "arxiv_id": "2607.10789", "source": "arxiv", "source_id": "arxiv:2607.10789", "title": "Imaging-101: Benchmarking LLM Coding Agents on Scientific Computational Imaging", "url": "https://arxiv.org/abs/2607.10789", "pdf_url": "https://arxiv.org/pdf/2607.10789", "published": "2026-07-12", "updated": "2026-07-12", "authors": [ "Siyi Chen", "Jiahe Ying", "Yixuan Jia", "Yuxuan Gu", "Enze Ye", "Weimin Bai", "Zhijun Zeng", "Shaochi Ren", "Binhong Gao", "Yubing Li", "Tianhan Zhang", "He Sun" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "coding-agent", "planning" ], "score": 17, "relevance": "high", "primary_query": "coding-agent", "matched_queries": [ "coding-agent" ] }, { "id": "2607.10582", "arxiv_id": "2607.10582", "source": "arxiv", "source_id": "arxiv:2607.10582", "title": "MemDecay: Region-Aware KV Cache Eviction for Efficient LLM Agent Inference", "url": "https://arxiv.org/abs/2607.10582", "pdf_url": "https://arxiv.org/pdf/2607.10582", "published": "2026-07-12", "updated": "2026-07-12", "authors": [ "Venkatesha Matam", "Keon Kim" ], "categories": [ "cs.LG" ], "topics": [ "agent-evaluation", "coding-agent", "memory", "planning", "reasoning", "tool-use" ], "score": 17, "relevance": "high", "primary_query": "planning-agent", "matched_queries": [ "planning-agent" ] }, { "id": "2606.29116", "arxiv_id": "2606.29116", "source": "arxiv", "source_id": "arxiv:2606.29116", "title": "Characterizing Large Language Model Agentic Workflows: A Study on N8n Ecosystem", "url": "https://arxiv.org/abs/2606.29116", "pdf_url": "https://arxiv.org/pdf/2606.29116", "published": "2026-07-11", "updated": "2026-07-11", "authors": [ "Yutian Tang", "Yuming Zhou", "Huaming Chen" ], "categories": [ "cs.AI" ], "topics": [ "agent-safety", "planning", "rag", "tool-use", "workflow-agent" ], "score": 17, "relevance": "high", "primary_query": "planning-agent", "matched_queries": [ "planning-agent" ] }, { "id": "2607.09195", "arxiv_id": "2607.09195", "source": "arxiv", "source_id": "arxiv:2607.09195", "title": "Toward Auditable AI Scientists: A Hypothesis Evolution Protocol for LLM Agents", "url": "https://arxiv.org/abs/2607.09195", "pdf_url": "https://arxiv.org/pdf/2607.09195", "published": "2026-07-10", "updated": "2026-07-10", "authors": [ "Izumi Takahara", "Teruyasu Mizoguchi" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "planning", "reasoning", "tool-use" ], "score": 17, "relevance": "high", "primary_query": "planning-agent", "matched_queries": [ "planning-agent", "tool-use" ] }, { "id": "2607.09179", "arxiv_id": "2607.09179", "source": "arxiv", "source_id": "arxiv:2607.09179", "title": "Malaika: Understanding Malware through Tri-Grounded Agentic Reasoning", "url": "https://arxiv.org/abs/2607.09179", "pdf_url": "https://arxiv.org/pdf/2607.09179", "published": "2026-07-10", "updated": "2026-07-10", "authors": [ "Xingzhi Qian", "Xinran Zheng", "Yiling He", "Lorenzo Cavallaro" ], "categories": [ "cs.CR" ], "topics": [ "agent-evaluation", "agent-safety", "multi-agent", "rag", "reasoning", "tool-use" ], "score": 17, "relevance": "high", "primary_query": "multi-agent-llm", "matched_queries": [ "multi-agent-llm" ] }, { "id": "2607.09153", "arxiv_id": "2607.09153", "source": "arxiv", "source_id": "arxiv:2607.09153", "title": "KV-PRM: Efficient Process Reward Modeling via KV-Cache Transfer for Multi-Agent Test-Time Scaling", "url": "https://arxiv.org/abs/2607.09153", "pdf_url": "https://arxiv.org/pdf/2607.09153", "published": "2026-07-10", "updated": "2026-07-10", "authors": [ "Peng Kuang", "Haibo Jin", "Xiaoyu Han", "Yanli Wang", "Xiaopeng Yuan", "Ye Yu", "Kaidi Xu", "Haohan Wang" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "coding-agent", "computer-use", "memory", "multi-agent" ], "score": 17, "relevance": "high", "primary_query": "multi-agent-llm", "matched_queries": [ "multi-agent-llm" ] }, { "id": "2607.08282", "arxiv_id": "2607.08282", "source": "arxiv", "source_id": "arxiv:2607.08282", "title": "Multi-Agent Firewall Architecture for Privacy Protection of Sensitive Data in Interactions with Language Models", "url": "https://arxiv.org/abs/2607.08282", "pdf_url": "https://arxiv.org/pdf/2607.08282", "published": "2026-07-09", "updated": "2026-07-09", "authors": [ "Hugo García Cuesta", "Pablo Mateo Torrejón", "Alfonso Sánchez-Macián" ], "categories": [ "cs.CR" ], "topics": [ "agent-evaluation", "agent-safety", "computer-use", "multi-agent", "tool-use", "workflow-agent" ], "score": 17, "relevance": "high", "primary_query": "multi-agent-llm", "matched_queries": [ "multi-agent-llm" ] }, { "id": "2607.22520", "arxiv_id": "2607.22520", "source": "arxiv", "source_id": "arxiv:2607.22520", "title": "The Regression Tax: Decomposing Why Skills Help and Hurt LLM Agents", "url": "https://arxiv.org/abs/2607.22520", "pdf_url": "https://arxiv.org/pdf/2607.22520", "published": "2026-07-24", "updated": "2026-07-24", "authors": [ "Darshan Tank", "Baran Nama" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "computer-use", "rag", "workflow-agent" ], "score": 16, "relevance": "high", "primary_query": "llm-agent", "matched_queries": [ "llm-agent" ] }, { "id": "2607.22465", "arxiv_id": "2607.22465", "source": "arxiv", "source_id": "arxiv:2607.22465", "title": "TRACE-ROUTER: Task-Consistent and Adaptive Online Routing for Agentic AI", "url": "https://arxiv.org/abs/2607.22465", "pdf_url": "https://arxiv.org/pdf/2607.22465", "published": "2026-07-24", "updated": "2026-07-24", "authors": [ "Ritik Raj", "Souvik Kundu", "Sarbartha Banerjee", "Dheemanth Joshi", "Ishita Vohra", "Tushar Krishna" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "planning", "rag", "workflow-agent" ], "score": 16, "relevance": "high", "primary_query": "agent-evaluation", "matched_queries": [ "agent-evaluation", "agentic-ai" ] }, { "id": "2607.21835", "arxiv_id": "2607.21835", "source": "arxiv", "source_id": "arxiv:2607.21835", "title": "ToolGuardian: Declarative Security for AI Agent-Tool Interactions", "url": "https://arxiv.org/abs/2607.21835", "pdf_url": "https://arxiv.org/pdf/2607.21835", "published": "2026-07-23", "updated": "2026-07-23", "authors": [ "Arun Ravindran", "Saurabh Deochake" ], "categories": [ "cs.CR" ], "topics": [ "agent-evaluation", "agent-safety", "reasoning", "tool-use" ], "score": 16, "relevance": "high", "primary_query": "ai-agent", "matched_queries": [ "ai-agent", "llm-agent" ] }, { "id": "2607.21419", "arxiv_id": "2607.21419", "source": "arxiv", "source_id": "arxiv:2607.21419", "title": "PATS: Policy-Aware Training Scaffolding for Agentic Reinforcement Learning", "url": "https://arxiv.org/abs/2607.21419", "pdf_url": "https://arxiv.org/pdf/2607.21419", "published": "2026-07-23", "updated": "2026-07-23", "authors": [ "Yipeng Shi", "Zhipeng Ma", "Yue Wang", "Qitai Tan", "Yang Li", "Peng Chen", "Zhengzhou Zhu" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "computer-use", "planning" ], "score": 16, "relevance": "high", "primary_query": "llm-agent", "matched_queries": [ "llm-agent" ] }, { "id": "2607.21125", "arxiv_id": "2607.21125", "source": "arxiv", "source_id": "arxiv:2607.21125", "title": "Causal-AgentIR: Self-Evolving Causal Memory for Adaptive Image Restoration Agents", "url": "https://arxiv.org/abs/2607.21125", "pdf_url": "https://arxiv.org/pdf/2607.21125", "published": "2026-07-23", "updated": "2026-07-23", "authors": [ "Hu Gao", "Yulong Chen", "Lizhuang Ma" ], "categories": [ "cs.CV" ], "topics": [ "agent-evaluation", "memory", "multi-agent", "planning", "rag", "reasoning", "tool-use" ], "score": 16, "relevance": "high", "primary_query": "tool-use", "matched_queries": [ "tool-use" ] }, { "id": "2607.21832", "arxiv_id": "2607.21832", "source": "arxiv", "source_id": "arxiv:2607.21832", "title": "How Do AI Coding Agents Contribute to Software Development? an Empirical Study of Agentic Pull Requests", "url": "https://arxiv.org/abs/2607.21832", "pdf_url": "https://arxiv.org/pdf/2607.21832", "published": "2026-07-23", "updated": "2026-07-23", "authors": [ "Iren Mazloomzadeh", "Mohammad Mehdi Morovati", "Foutse Khomh" ], "categories": [ "cs.SE" ], "topics": [ "coding-agent", "tool-use", "workflow-agent" ], "score": 16, "relevance": "high", "primary_query": "coding-agent", "matched_queries": [ "coding-agent" ] }, { "id": "2607.19605", "arxiv_id": "2607.19605", "source": "arxiv", "source_id": "arxiv:2607.19605", "title": "RIME: Enabling Large-Scale Agentic Music Post-Production", "url": "https://arxiv.org/abs/2607.19605", "pdf_url": "https://arxiv.org/pdf/2607.19605", "published": "2026-07-23", "updated": "2026-07-23", "authors": [ "Noah Schaffer", "Nikhil Singh" ], "categories": [ "cs.SD" ], "topics": [ "agent-evaluation", "coding-agent", "rag", "tool-use", "workflow-agent" ], "score": 16, "relevance": "high", "primary_query": "coding-agent", "matched_queries": [ "coding-agent" ] }, { "id": "2607.21912", "arxiv_id": "2607.21912", "source": "arxiv", "source_id": "arxiv:2607.21912", "title": "Reliability-Contagion Feasibility in LLM Multi-Agent Networks", "url": "https://arxiv.org/abs/2607.21912", "pdf_url": "https://arxiv.org/pdf/2607.21912", "published": "2026-07-23", "updated": "2026-07-23", "authors": [ "Ruiwu Niu", "Xincheng Shu", "Ying Zhao" ], "categories": [ "cs.MA" ], "topics": [ "agent-evaluation", "agent-safety", "multi-agent", "tool-use", "world-model" ], "score": 16, "relevance": "high", "primary_query": "multi-agent-llm", "matched_queries": [ "multi-agent-llm" ] }, { "id": "2607.19837", "arxiv_id": "2607.19837", "source": "arxiv", "source_id": "arxiv:2607.19837", "title": "Know Your Agent: Reconnaissance-Driven Pentesting of AI Agents", "url": "https://arxiv.org/abs/2607.19837", "pdf_url": "https://arxiv.org/pdf/2607.19837", "published": "2026-07-22", "updated": "2026-07-22", "authors": [ "Or Zion Eliav", "Eyal Lenga", "Shir Bernstien", "Yisroel Mirsky" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "agent-safety", "coding-agent", "rag" ], "score": 16, "relevance": "high", "primary_query": "agent-safety", "matched_queries": [ "agent-safety", "ai-agent", "coding-agent" ] }, { "id": "2607.19336", "arxiv_id": "2607.19336", "source": "arxiv", "source_id": "arxiv:2607.19336", "title": "Agents in the Wild: Where Research Meets Deployment", "url": "https://arxiv.org/abs/2607.19336", "pdf_url": "https://arxiv.org/pdf/2607.19336", "published": "2026-07-21", "updated": "2026-07-21", "authors": [ "Grace Hui Yang", "Pranav N. Venkit", "Hooman Sedghamiz", "Enrico Santus", "Victor Dibia", "Ioana Baldini" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "agent-safety", "coding-agent", "planning", "reasoning", "tool-use" ], "score": 16, "relevance": "high", "primary_query": "multi-agent-llm", "matched_queries": [ "multi-agent-llm" ] }, { "id": "2607.18847", "arxiv_id": "2607.18847", "source": "arxiv", "source_id": "arxiv:2607.18847", "title": "Data Leakage Prevention in Agentic Applications via Preemptive Hardening", "url": "https://arxiv.org/abs/2607.18847", "pdf_url": "https://arxiv.org/pdf/2607.18847", "published": "2026-07-21", "updated": "2026-07-21", "authors": [ "Akansha Shukla", "Emily Bellov", "Parth Atulbhai Gandhi", "Yuval Elovici", "Asaf Shabtai" ], "categories": [ "cs.CR" ], "topics": [ "agent-evaluation", "agent-safety", "planning", "tool-use", "workflow-agent" ], "score": 16, "relevance": "high", "primary_query": "multi-agent-llm", "matched_queries": [ "multi-agent-llm" ] }, { "id": "2607.18575", "arxiv_id": "2607.18575", "source": "arxiv", "source_id": "arxiv:2607.18575", "title": "RECEIPT: Deterministic, Reward-Hacking-Resistant Verification for White-Box Agentic XSS Discovery", "url": "https://arxiv.org/abs/2607.18575", "pdf_url": "https://arxiv.org/pdf/2607.18575", "published": "2026-07-20", "updated": "2026-07-20", "authors": [ "Muxi Lyu", "Karen Shieh", "Yiwei Hou", "Hao Wang", "Koushik Sen", "David Wagner" ], "categories": [ "cs.CR" ], "topics": [ "agent-evaluation", "coding-agent", "computer-use", "planning", "reasoning" ], "score": 16, "relevance": "high", "primary_query": "coding-agent", "matched_queries": [ "coding-agent" ] }, { "id": "2607.17050", "arxiv_id": "2607.17050", "source": "arxiv", "source_id": "arxiv:2607.17050", "title": "EvoGUI: An Evolution-Aware Benchmark for GUI State-Transition Understanding", "url": "https://arxiv.org/abs/2607.17050", "pdf_url": "https://arxiv.org/pdf/2607.17050", "published": "2026-07-18", "updated": "2026-07-18", "authors": [ "Yaohan Yang", "Minglei Shi", "Borui Zhang", "Jie Zhou", "Jiwen Lu" ], "categories": [ "cs.CV" ], "topics": [ "agent-evaluation", "computer-use", "planning", "rag", "tool-use" ], "score": 16, "relevance": "high", "primary_query": "agent-evaluation", "matched_queries": [ "agent-evaluation", "web-gui-agent" ] }, { "id": "2607.16610", "arxiv_id": "2607.16610", "source": "arxiv", "source_id": "arxiv:2607.16610", "title": "Just A Rather Very Intelligent Spoken Agent", "url": "https://arxiv.org/abs/2607.16610", "pdf_url": "https://arxiv.org/pdf/2607.16610", "published": "2026-07-17", "updated": "2026-07-17", "authors": [ "Chen Chen", "Zhehuai Chen" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "computer-use", "multi-agent", "planning", "tool-use", "workflow-agent" ], "score": 16, "relevance": "high", "primary_query": "agentic-ai", "matched_queries": [ "agentic-ai", "ai-agent" ] }, { "id": "2607.15263", "arxiv_id": "2607.15263", "source": "arxiv", "source_id": "arxiv:2607.15263", "title": "Beyond Success Rate: Cost-Aware Evaluation of Offensive and Defensive Security Agents", "url": "https://arxiv.org/abs/2607.15263", "pdf_url": "https://arxiv.org/pdf/2607.15263", "published": "2026-07-17", "updated": "2026-07-17", "authors": [ "Paul Kassianik", "Blaine Nelson", "Yaron Singer" ], "categories": [ "cs.CR" ], "topics": [ "agent-evaluation", "agent-safety", "embodied-agent", "reasoning", "tool-use" ], "score": 16, "relevance": "high", "primary_query": "agent-evaluation", "matched_queries": [ "agent-evaluation", "tool-use" ] }, { "id": "2607.15367", "arxiv_id": "2607.15367", "source": "arxiv", "source_id": "arxiv:2607.15367", "title": "AnovaX: A Local, Multi-Agent Voice Assistant with LLM Planning, Typed Executors, and Adaptive Recovery", "url": "https://arxiv.org/abs/2607.15367", "pdf_url": "https://arxiv.org/pdf/2607.15367", "published": "2026-07-16", "updated": "2026-07-16", "authors": [ "Raunak B Sinha" ], "categories": [ "cs.AI" ], "topics": [ "agent-safety", "computer-use", "multi-agent", "planning", "rag", "tool-use" ], "score": 16, "relevance": "high", "primary_query": "multi-agent-llm", "matched_queries": [ "multi-agent-llm" ] }, { "id": "2607.14165", "arxiv_id": "2607.14165", "source": "arxiv", "source_id": "arxiv:2607.14165", "title": "Towards Reliable AI-Assisted Analog Design: Template-Constrained LLM Agents for SAR ADC Generation", "url": "https://arxiv.org/abs/2607.14165", "pdf_url": "https://arxiv.org/pdf/2607.14165", "published": "2026-07-15", "updated": "2026-07-15", "authors": [ "Dimple Vijay Kochar", "Hae-Seung Lee", "Anantha P. Chandrakasan" ], "categories": [ "cs.SE" ], "topics": [ "planning", "rag", "workflow-agent", "world-model" ], "score": 16, "relevance": "high", "primary_query": "llm-agent", "matched_queries": [ "llm-agent", "planning-agent" ] }, { "id": "2606.05711", "arxiv_id": "2606.05711", "source": "arxiv", "source_id": "arxiv:2606.05711", "title": "Beyond tokens: a unified framework for latent communication in LLM-based multi-agent systems", "url": "https://arxiv.org/abs/2606.05711", "pdf_url": "https://arxiv.org/pdf/2606.05711", "published": "2026-07-15", "updated": "2026-07-15", "authors": [ "Yingzhuo Liu" ], "categories": [ "cs.CL" ], "topics": [ "agent-safety", "computer-use", "multi-agent", "planning", "reasoning", "tool-use" ], "score": 16, "relevance": "high", "primary_query": "language-agent", "matched_queries": [ "language-agent" ] }, { "id": "2607.12406", "arxiv_id": "2607.12406", "source": "arxiv", "source_id": "arxiv:2607.12406", "title": "Isolation as a First-Class Principle for LLM-Agent System Safety: Concepts, Taxonomy, Challenges and Future Directions", "url": "https://arxiv.org/abs/2607.12406", "pdf_url": "https://arxiv.org/pdf/2607.12406", "published": "2026-07-14", "updated": "2026-07-14", "authors": [ "Huihao Jing", "Wenbin Hu", "Shaojin Chen", "Haochen Shi", "Sirui Zhang", "Hanyu Yang", "Changxuan Fan", "Zhongwei Xie", "Hongyu Luo", "Wun Yu Chan", "Wei Fan", "Haoran Li", "Yangqiu Song" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "agent-safety", "memory", "rag", "tool-use", "workflow-agent" ], "score": 16, "relevance": "high", "primary_query": "agentic-ai", "matched_queries": [ "agentic-ai" ] }, { "id": "2607.13081", "arxiv_id": "2607.13081", "source": "arxiv", "source_id": "arxiv:2607.13081", "title": "SingGuard-NSFA: Extensible Guardrails for Agentic AI via Generative Reasoning and Real-Time Classification", "url": "https://arxiv.org/abs/2607.13081", "pdf_url": "https://arxiv.org/pdf/2607.13081", "published": "2026-07-13", "updated": "2026-07-13", "authors": [ "SingGuard Team" ], "categories": [ "cs.CR" ], "topics": [ "agent-evaluation", "agent-safety", "computer-use", "reasoning", "tool-use" ], "score": 16, "relevance": "high", "primary_query": "agent-safety", "matched_queries": [ "agent-safety", "agentic-ai" ] }, { "id": "2607.13085", "arxiv_id": "2607.13085", "source": "arxiv", "source_id": "arxiv:2607.13085", "title": "Baselines Before Architecture: Evaluating Coding Agents for Autonomous Penetration Testing", "url": "https://arxiv.org/abs/2607.13085", "pdf_url": "https://arxiv.org/pdf/2607.13085", "published": "2026-07-13", "updated": "2026-07-13", "authors": [ "Ananda Dhakal", "Krish Neupane", "Aarjan Chaudhary" ], "categories": [ "cs.CR" ], "topics": [ "agent-evaluation", "agent-safety", "coding-agent", "rag" ], "score": 16, "relevance": "high", "primary_query": "coding-agent", "matched_queries": [ "coding-agent" ] }, { "id": "2607.08894", "arxiv_id": "2607.08894", "source": "arxiv", "source_id": "arxiv:2607.08894", "title": "GATS: Graph-Augmented Tree Search with Layered World Models for Efficient Agent Planning", "url": "https://arxiv.org/abs/2607.08894", "pdf_url": "https://arxiv.org/pdf/2607.08894", "published": "2026-07-09", "updated": "2026-07-09", "authors": [ "Maureese Williams", "Dymitr Nowicki" ], "categories": [ "cs.AI" ], "topics": [ "coding-agent", "computer-use", "embodied-agent", "planning", "tool-use", "workflow-agent", "world-model" ], "score": 16, "relevance": "high", "primary_query": "language-agent", "matched_queries": [ "language-agent", "planning-agent" ] }, { "id": "2607.19215", "arxiv_id": "2607.19215", "source": "arxiv", "source_id": "arxiv:2607.19215", "title": "HACO: Hedged Agent Computing for Reliable LLM Systems", "url": "https://arxiv.org/abs/2607.19215", "pdf_url": "https://arxiv.org/pdf/2607.19215", "published": "2026-07-21", "updated": "2026-07-21", "authors": [ "Enhan Li", "Hongyang Du" ], "categories": [ "cs.NI" ], "topics": [ "agent-evaluation", "memory", "tool-use", "workflow-agent" ], "score": 15, "relevance": "high", "primary_query": "llm-agent", "matched_queries": [ "llm-agent", "tool-use" ] }, { "id": "2605.28787", "arxiv_id": "2605.28787", "source": "arxiv", "source_id": "arxiv:2605.28787", "title": "Do Data Agents Need Semantic Metadata? A Comparative Study in Agentic Data Retrieval", "url": "https://arxiv.org/abs/2605.28787", "pdf_url": "https://arxiv.org/pdf/2605.28787", "published": "2026-07-21", "updated": "2026-07-21", "authors": [ "Shiyu Chen", "Tarfah Alrashed", "Alon Halevy", "Natasha Noy" ], "categories": [ "cs.IR" ], "topics": [ "agent-evaluation", "rag", "tool-use", "workflow-agent" ], "score": 15, "relevance": "high", "primary_query": "autonomous-agent-llm", "matched_queries": [ "autonomous-agent-llm" ] }, { "id": "2607.18064", "arxiv_id": "2607.18064", "source": "arxiv", "source_id": "arxiv:2607.18064", "title": "Autoresearch with Coding Agents: Generalizers and Metric-Maximizers on Quran Recitation Data", "url": "https://arxiv.org/abs/2607.18064", "pdf_url": "https://arxiv.org/pdf/2607.18064", "published": "2026-07-20", "updated": "2026-07-20", "authors": [ "Nursultan Askarbekuly", "Mohamad Al Mdfaa", "Ahmed Helaly", "Gonzalo Ferrer", "Manuel Mazzara" ], "categories": [ "cs.SE" ], "topics": [ "agent-evaluation", "agent-safety", "coding-agent", "memory", "reasoning" ], "score": 15, "relevance": "high", "primary_query": "coding-agent", "matched_queries": [ "coding-agent" ] }, { "id": "2607.05690", "arxiv_id": "2607.05690", "source": "arxiv", "source_id": "arxiv:2607.05690", "title": "Memory in the Loop: In-Process Retrieval as Extended Working Memory for Language Agents", "url": "https://arxiv.org/abs/2607.05690", "pdf_url": "https://arxiv.org/pdf/2607.05690", "published": "2026-07-19", "updated": "2026-07-19", "authors": [ "Yusuf Khan", "Carlo Lipizzi" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "memory", "rag", "tool-use" ], "score": 15, "relevance": "high", "primary_query": "language-agent", "matched_queries": [ "language-agent" ] }, { "id": "2607.15593", "arxiv_id": "2607.15593", "source": "arxiv", "source_id": "arxiv:2607.15593", "title": "Scalable LLM Agent Tool Access in the Cloud", "url": "https://arxiv.org/abs/2607.15593", "pdf_url": "https://arxiv.org/pdf/2607.15593", "published": "2026-07-16", "updated": "2026-07-16", "authors": [ "Mingxin Li", "Enge Song", "Yueshang Zuo", "Xiaodong Liu", "Rong Wen", "Qiang Fu", "Gianni Antichi", "Jian He", "Jing Tie", "Zhou Shao", "Xiaobo Xue", "Xiong Xiao", "Luyao Zhong", "Shaokai Zhang", "Jiangu Zhao", "Jianyuan Lu", "Shize Zhang", "Xiaoqing Sun", "Changgang Zheng", "Zihao Fan", "Haonan Li", "Tian Pan", "Xiaomin Wu", "Yang Song", "Xing Li", "et al. (5 additional authors not shown)" ], "categories": [ "cs.DC" ], "topics": [ "agent-evaluation", "memory", "planning", "rag", "tool-use" ], "score": 15, "relevance": "high", "primary_query": "llm-agent", "matched_queries": [ "llm-agent" ] }, { "id": "2607.14777", "arxiv_id": "2607.14777", "source": "arxiv", "source_id": "arxiv:2607.14777", "title": "SEED: Self-Evolving On-Policy Distillation for Agentic Reinforcement Learning", "url": "https://arxiv.org/abs/2607.14777", "pdf_url": "https://arxiv.org/pdf/2607.14777", "published": "2026-07-16", "updated": "2026-07-16", "authors": [ "Jinyang Wu", "Shuo Yang", "Zhengxi Lu", "Fan Zhang", "Yuhao Shen", "Lang Feng", "Haoran Luo", "Zheng Lian", "Shuai Zhang", "Zhengqi Wen", "Jianhua Tao" ], "categories": [ "cs.CL" ], "topics": [ "computer-use", "planning", "tool-use", "workflow-agent" ], "score": 15, "relevance": "high", "primary_query": "tool-use", "matched_queries": [ "tool-use" ] }, { "id": "2607.13987", "arxiv_id": "2607.13987", "source": "arxiv", "source_id": "arxiv:2607.13987", "title": "Agent Skill Security: Threat Models, Attacks, Defenses, and Evaluation", "url": "https://arxiv.org/abs/2607.13987", "pdf_url": "https://arxiv.org/pdf/2607.13987", "published": "2026-07-15", "updated": "2026-07-15", "authors": [ "Sanket Badhe", "Priyanka Tiwari" ], "categories": [ "cs.CR" ], "topics": [ "agent-evaluation", "agent-safety", "coding-agent", "planning", "rag" ], "score": 15, "relevance": "high", "primary_query": "llm-agent", "matched_queries": [ "llm-agent" ] }, { "id": "2607.14264", "arxiv_id": "2607.14264", "source": "arxiv", "source_id": "arxiv:2607.14264", "title": "MonteRET: AI Agent Enhancing Multimodal LLMs with Multi-granularity Knowledge Retrieval for Chest CT Report Generation", "url": "https://arxiv.org/abs/2607.14264", "pdf_url": "https://arxiv.org/pdf/2607.14264", "published": "2026-07-15", "updated": "2026-07-15", "authors": [ "Yi Lin", "Yihao Ding", "Elana Benishay", "Elefterios Trikantzopoulos", "David Nauheim", "Hanley Ong", "Jiang Bian", "Hua Xu", "Yuzhe Yang", "George Shih", "Yifan Peng" ], "categories": [ "cs.CV" ], "topics": [ "agent-evaluation", "agent-safety", "computer-use", "rag" ], "score": 15, "relevance": "high", "primary_query": "ai-agent", "matched_queries": [ "ai-agent" ] }, { "id": "2607.13854", "arxiv_id": "2607.13854", "source": "arxiv", "source_id": "arxiv:2607.13854", "title": "SPyCE: Skill-Policy Co-evolution for Multimodal Agents", "url": "https://arxiv.org/abs/2607.13854", "pdf_url": "https://arxiv.org/pdf/2607.13854", "published": "2026-07-15", "updated": "2026-07-15", "authors": [ "Ru Zhang", "Weijie Qiu" ], "categories": [ "cs.CL" ], "topics": [ "agent-evaluation", "computer-use", "memory", "rag", "reasoning", "tool-use", "workflow-agent" ], "score": 15, "relevance": "high", "primary_query": "tool-use", "matched_queries": [ "tool-use" ] }, { "id": "2607.13196", "arxiv_id": "2607.13196", "source": "arxiv", "source_id": "arxiv:2607.13196", "title": "From Human-Centric to Agentic Code Review: The Impact of Different Generations of Generative AI Technology on Review Quality", "url": "https://arxiv.org/abs/2607.13196", "pdf_url": "https://arxiv.org/pdf/2607.13196", "published": "2026-07-14", "updated": "2026-07-14", "authors": [ "Suzhen Zhong", "Shayan Noei", "Bram Adams", "Ying Zou" ], "categories": [ "cs.SE" ], "topics": [ "computer-use", "multi-agent", "planning", "tool-use" ], "score": 15, "relevance": "high", "primary_query": "ai-agent", "matched_queries": [ "ai-agent" ] }, { "id": "2607.12397", "arxiv_id": "2607.12397", "source": "arxiv", "source_id": "arxiv:2607.12397", "title": "Critic Experience Bank: Self-Evolving Step-Level Confidence Estimation for LLM Agents", "url": "https://arxiv.org/abs/2607.12397", "pdf_url": "https://arxiv.org/pdf/2607.12397", "published": "2026-07-14", "updated": "2026-07-14", "authors": [ "Yaopei Zeng", "Congchao Wang", "JianHang Chen", "Nan Wang", "Yurui Chang", "Lu Lin" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "memory", "tool-use" ], "score": 15, "relevance": "high", "primary_query": "agent-evaluation", "matched_queries": [ "agent-evaluation" ] }, { "id": "2607.14145", "arxiv_id": "2607.14145", "source": "arxiv", "source_id": "arxiv:2607.14145", "title": "ToolAnchor: Anchoring Counterfactual Context to Boost Agentic Tool-use Capability", "url": "https://arxiv.org/abs/2607.14145", "pdf_url": "https://arxiv.org/pdf/2607.14145", "published": "2026-07-14", "updated": "2026-07-14", "authors": [ "Weiting Liu", "Jieyi Bi", "Wanqi Zhou", "Jianfeng Feng", "Yining Ma", "Ai Han", "Wenlian Lu" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "planning", "reasoning", "tool-use" ], "score": 15, "relevance": "high", "primary_query": "tool-use", "matched_queries": [ "tool-use" ] }, { "id": "2607.12068", "arxiv_id": "2607.12068", "source": "arxiv", "source_id": "arxiv:2607.12068", "title": "Beyond Test Presence: Assessing the Quality and Robustness of Agent-Generated Tests in Open-Source Projects", "url": "https://arxiv.org/abs/2607.12068", "pdf_url": "https://arxiv.org/pdf/2607.12068", "published": "2026-07-13", "updated": "2026-07-13", "authors": [ "Preet Jhanglani", "Zeel Kaushal Desai", "Vidhi Kansara", "Eman Abdullah AlOmar" ], "categories": [ "cs.SE" ], "topics": [ "agent-evaluation", "agent-safety", "coding-agent", "rag" ], "score": 15, "relevance": "high", "primary_query": "ai-agent", "matched_queries": [ "ai-agent", "coding-agent" ] }, { "id": "2607.11250", "arxiv_id": "2607.11250", "source": "arxiv", "source_id": "arxiv:2607.11250", "title": "Multi-Agent LLMs Fail to Explore Each Other", "url": "https://arxiv.org/abs/2607.11250", "pdf_url": "https://arxiv.org/pdf/2607.11250", "published": "2026-07-13", "updated": "2026-07-13", "authors": [ "Hyeong Kyu Choi", "Jiatong Li", "Wendi Li", "Xin Eric Wang", "Sharon Li" ], "categories": [ "cs.MA" ], "topics": [ "agent-evaluation", "computer-use", "multi-agent", "tool-use" ], "score": 15, "relevance": "high", "primary_query": "multi-agent-llm", "matched_queries": [ "multi-agent-llm" ] }, { "id": "2607.11046", "arxiv_id": "2607.11046", "source": "arxiv", "source_id": "arxiv:2607.11046", "title": "Retrieval-Oriented Code Representations in Agentic Bug Localization", "url": "https://arxiv.org/abs/2607.11046", "pdf_url": "https://arxiv.org/pdf/2607.11046", "published": "2026-07-12", "updated": "2026-07-12", "authors": [ "Genevieve Caumartin", "Tse-Hsun", "Chen", "Diego Elias Costa" ], "categories": [ "cs.SE" ], "topics": [ "agent-evaluation", "coding-agent", "computer-use", "rag" ], "score": 15, "relevance": "high", "primary_query": "coding-agent", "matched_queries": [ "coding-agent" ] }, { "id": "2607.10490", "arxiv_id": "2607.10490", "source": "arxiv", "source_id": "arxiv:2607.10490", "title": "NetInjectBench: Benchmarking Indirect Prompt Injection in Tool-Using Large Language Model Agents for Network Operations", "url": "https://arxiv.org/abs/2607.10490", "pdf_url": "https://arxiv.org/pdf/2607.10490", "published": "2026-07-11", "updated": "2026-07-11", "authors": [ "Ruksat Khan Shayoni", "Muhammad Faraz Shoaib", "S M Asif Hossain", "M. F. Mridha" ], "categories": [ "cs.CR" ], "topics": [ "agent-evaluation", "agent-safety", "tool-use" ], "score": 15, "relevance": "high", "primary_query": "tool-use", "matched_queries": [ "tool-use" ] }, { "id": "2607.10286", "arxiv_id": "2607.10286", "source": "arxiv", "source_id": "arxiv:2607.10286", "title": "Can Agentic Trading Systems Pay for Their Own Intelligence?", "url": "https://arxiv.org/abs/2607.10286", "pdf_url": "https://arxiv.org/pdf/2607.10286", "published": "2026-07-11", "updated": "2026-07-11", "authors": [ "Qiqi Duan", "Changlun Li", "Chen Wang", "Fan Zhang", "Mengxiang Wang", "Dayi Miao", "Peixian Ma", "Jiangpeng Yan", "Liyuan Chen", "Shuoling Liu", "Preslav Nakov", "Yuyu Luo", "Nan Tang" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "reasoning", "tool-use" ], "score": 15, "relevance": "high", "primary_query": "tool-use", "matched_queries": [ "tool-use" ] }, { "id": "2607.07405", "arxiv_id": "2607.07405", "source": "arxiv", "source_id": "arxiv:2607.07405", "title": "Reason Less, Verify More: Deterministic Gates Recover a Silent Policy-Violation Failure Mode in Tool-Using LLM Agents", "url": "https://arxiv.org/abs/2607.07405", "pdf_url": "https://arxiv.org/pdf/2607.07405", "published": "2026-07-11", "updated": "2026-07-11", "authors": [ "Vikas Reddy", "Sumanth Reddy Challaram", "Abhishek Basu" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "reasoning", "tool-use" ], "score": 15, "relevance": "high", "primary_query": "tool-use", "matched_queries": [ "tool-use" ] }, { "id": "2607.10463", "arxiv_id": "2607.10463", "source": "arxiv", "source_id": "arxiv:2607.10463", "title": "GRASP: GRanularity-Aware Search Policy for Agentic RAG", "url": "https://arxiv.org/abs/2607.10463", "pdf_url": "https://arxiv.org/pdf/2607.10463", "published": "2026-07-11", "updated": "2026-07-11", "authors": [ "Varun Gandhi", "Jaewook Lee", "Shantanu Todmal", "Franck Dernoncourt", "Ryan Rossi", "Zichao Wang", "Andrew Lan" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "rag", "reasoning", "tool-use" ], "score": 15, "relevance": "high", "primary_query": "rag-agent", "matched_queries": [ "rag-agent" ] }, { "id": "2607.09092", "arxiv_id": "2607.09092", "source": "arxiv", "source_id": "arxiv:2607.09092", "title": "AgentKGV: Agentic LLM-RAG Framework with Two-Stage Training for the Fact Verification of Knowledge Graphs", "url": "https://arxiv.org/abs/2607.09092", "pdf_url": "https://arxiv.org/pdf/2607.09092", "published": "2026-07-10", "updated": "2026-07-10", "authors": [ "Yumin Heo", "Hyeon-gu Lee", "Sumin Seo", "Youngjoong Ko" ], "categories": [ "cs.CL" ], "topics": [ "agent-evaluation", "rag", "reasoning", "tool-use" ], "score": 15, "relevance": "high", "primary_query": "rag-agent", "matched_queries": [ "rag-agent" ] }, { "id": "2607.09076", "arxiv_id": "2607.09076", "source": "arxiv", "source_id": "arxiv:2607.09076", "title": "Neuro-Agentic Control: A Deep Learning-based LLM-Powered Agentic AI Framework for Controlling Security Controls", "url": "https://arxiv.org/abs/2607.09076", "pdf_url": "https://arxiv.org/pdf/2607.09076", "published": "2026-07-09", "updated": "2026-07-09", "authors": [ "Saroj Gopali", "Bipin Chhetri", "Deepika Giri", "Sima Siami-Namini", "Akbar Siami Namin" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "agent-safety", "planning", "reasoning", "tool-use" ], "score": 15, "relevance": "high", "primary_query": "agentic-ai", "matched_queries": [ "agentic-ai" ] }, { "id": "2607.08180", "arxiv_id": "2607.08180", "source": "arxiv", "source_id": "arxiv:2607.08180", "title": "Out of Sight: Compression-Aware Content Protection against Agentic Crawlers", "url": "https://arxiv.org/abs/2607.08180", "pdf_url": "https://arxiv.org/pdf/2607.08180", "published": "2026-07-09", "updated": "2026-07-09", "authors": [ "Xuefei Wang" ], "categories": [ "cs.CR" ], "topics": [ "computer-use", "memory", "reasoning", "workflow-agent" ], "score": 15, "relevance": "high", "primary_query": "agentic-ai", "matched_queries": [ "agentic-ai" ] }, { "id": "2607.22443", "arxiv_id": "2607.22443", "source": "arxiv", "source_id": "arxiv:2607.22443", "title": "A Human-Augmenting Agentic Workflow for Observational Causal Inference", "url": "https://arxiv.org/abs/2607.22443", "pdf_url": "https://arxiv.org/pdf/2607.22443", "published": "2026-07-24", "updated": "2026-07-24", "authors": [ "Winston Chou", "Adrien Alexandre", "Lars Olds", "Yi Zhang", "Nathan Kallus" ], "categories": [ "stat.CO" ], "topics": [ "agent-evaluation", "rag", "tool-use", "workflow-agent" ], "score": 14, "relevance": "high", "primary_query": "agentic-ai", "matched_queries": [ "agentic-ai" ] }, { "id": "2607.21273", "arxiv_id": "2607.21273", "source": "arxiv", "source_id": "arxiv:2607.21273", "title": "The Dark Room in the Reward Channel: Dense Prediction Rewards Collapse GRPO-Trained LLM Agents -- and What Actually Works", "url": "https://arxiv.org/abs/2607.21273", "pdf_url": "https://arxiv.org/pdf/2607.21273", "published": "2026-07-23", "updated": "2026-07-23", "authors": [ "Yu Wang" ], "categories": [ "cs.LG" ], "topics": [ "memory", "planning", "tool-use" ], "score": 14, "relevance": "high", "primary_query": "llm-agent", "matched_queries": [ "llm-agent" ] }, { "id": "2607.15557", "arxiv_id": "2607.15557", "source": "arxiv", "source_id": "arxiv:2607.15557", "title": "SkillCorpus: Consolidating and Evaluating the Open Skill Ecosystem for Real-World LLM Agents", "url": "https://arxiv.org/abs/2607.15557", "pdf_url": "https://arxiv.org/pdf/2607.15557", "published": "2026-07-23", "updated": "2026-07-23", "authors": [ "Yanze Wang", "Pengfei Yao", "Tianyi Sun", "Chuanrui Hu", "Yan Xiao", "Yunyun Han", "Yifan Chen", "Jun Sun", "Yafeng Deng" ], "categories": [ "cs.CL" ], "topics": [ "agent-evaluation", "agent-safety", "rag" ], "score": 14, "relevance": "high", "primary_query": "llm-agent", "matched_queries": [ "llm-agent" ] }, { "id": "2607.20827", "arxiv_id": "2607.20827", "source": "arxiv", "source_id": "arxiv:2607.20827", "title": "Auditing Provenance Sensitivity in LLM Agent Action Selection", "url": "https://arxiv.org/abs/2607.20827", "pdf_url": "https://arxiv.org/pdf/2607.20827", "published": "2026-07-22", "updated": "2026-07-22", "authors": [ "Junchi Liao" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "memory", "tool-use" ], "score": 14, "relevance": "high", "primary_query": "llm-agent", "matched_queries": [ "llm-agent" ] }, { "id": "2607.19947", "arxiv_id": "2607.19947", "source": "arxiv", "source_id": "arxiv:2607.19947", "title": "ETPDesigner: Multi-Agent Orchestration for Interactive Multimodal Electronic Theater Program", "url": "https://arxiv.org/abs/2607.19947", "pdf_url": "https://arxiv.org/pdf/2607.19947", "published": "2026-07-22", "updated": "2026-07-22", "authors": [ "Mengtian Li", "Xinru Guo", "Xiaoru Lin", "Xiao Rong", "Zhifeng Xie", "Chaofeng Chen" ], "categories": [ "cs.CV" ], "topics": [ "agent-evaluation", "multi-agent" ], "score": 14, "relevance": "high", "primary_query": "multi-agent-llm", "matched_queries": [ "multi-agent-llm" ] }, { "id": "2607.16900", "arxiv_id": "2607.16900", "source": "arxiv", "source_id": "arxiv:2607.16900", "title": "Environment-free Synthetic Data Generation for API-Calling Agents", "url": "https://arxiv.org/abs/2607.16900", "pdf_url": "https://arxiv.org/pdf/2607.16900", "published": "2026-07-21", "updated": "2026-07-21", "authors": [ "Seanie Lee", "Sanjoy Chowdhury", "Chao Jiang", "Cheng-Yu Hsieh", "Ting-Yao Hu", "Alexander T Toshev", "Oncel Tuzel", "Raviteja Vemulapalli" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "memory", "rag", "tool-use", "workflow-agent", "world-model" ], "score": 14, "relevance": "high", "primary_query": "llm-agent", "matched_queries": [ "llm-agent" ] }, { "id": "2607.19262", "arxiv_id": "2607.19262", "source": "arxiv", "source_id": "arxiv:2607.19262", "title": "BioSecBench-Surveillance: A Verifiable Benchmark for AI Agents in Pathogen Genomic Surveillance", "url": "https://arxiv.org/abs/2607.19262", "pdf_url": "https://arxiv.org/pdf/2607.19262", "published": "2026-07-21", "updated": "2026-07-21", "authors": [ "Harmon Bhasin", "Kevin Flyangolts", "Dianzhuo Wang", "Evan Seeyave", "Arjun Banerjee", "Amanda Darling", "Joshua Stallings", "David Stern", "Shawn Higdon", "Claire Duvallet", "Bryan Tegomoh", "Kenny Workman" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "workflow-agent" ], "score": 14, "relevance": "high", "primary_query": "ai-agent", "matched_queries": [ "ai-agent" ] }, { "id": "2607.19096", "arxiv_id": "2607.19096", "source": "arxiv", "source_id": "arxiv:2607.19096", "title": "Supra Cognitive Modes: A Routed Architecture for Agent Memory", "url": "https://arxiv.org/abs/2607.19096", "pdf_url": "https://arxiv.org/pdf/2607.19096", "published": "2026-07-21", "updated": "2026-07-21", "authors": [ "Joshua Tobkin", "David Yang" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "coding-agent", "memory", "rag", "reasoning" ], "score": 14, "relevance": "high", "primary_query": "agent-memory", "matched_queries": [ "agent-memory" ] }, { "id": "2607.17951", "arxiv_id": "2607.17951", "source": "arxiv", "source_id": "arxiv:2607.17951", "title": "RT-SHCUA: Real-Time Self-Hosted Computer-Use Agent for UAV Control", "url": "https://arxiv.org/abs/2607.17951", "pdf_url": "https://arxiv.org/pdf/2607.17951", "published": "2026-07-20", "updated": "2026-07-20", "authors": [ "Di Lu", "Bo Zhang", "Xiyuan Li", "Yongzhi Liao", "Xuewen Dong", "Yulong Shen", "Zhiquan Liu", "Jianfeng Ma" ], "categories": [ "cs.CR" ], "topics": [ "agent-evaluation", "agent-safety", "reasoning", "tool-use" ], "score": 14, "relevance": "high", "primary_query": "language-agent", "matched_queries": [ "language-agent", "tool-use", "web-gui-agent" ] }, { "id": "2607.17621", "arxiv_id": "2607.17621", "source": "arxiv", "source_id": "arxiv:2607.17621", "title": "Mechanistic Attention Guidance for Agent Memory Refinement", "url": "https://arxiv.org/abs/2607.17621", "pdf_url": "https://arxiv.org/pdf/2607.17621", "published": "2026-07-20", "updated": "2026-07-20", "authors": [ "Yechao Hong", "Haiquan Qiu", "Yaqing Wang", "Quanming Yao" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "computer-use", "memory", "rag", "reasoning" ], "score": 14, "relevance": "high", "primary_query": "agent-memory", "matched_queries": [ "agent-memory" ] }, { "id": "2607.18665", "arxiv_id": "2607.18665", "source": "arxiv", "source_id": "arxiv:2607.18665", "title": "SciHazard: A Benchmark for Measuring Scientific Safety Risks with Decomposed Harm Scoring", "url": "https://arxiv.org/abs/2607.18665", "pdf_url": "https://arxiv.org/pdf/2607.18665", "published": "2026-07-20", "updated": "2026-07-20", "authors": [ "Chunxiao Li", "Yuan Xiong", "Lijun Li", "Tianyi Du", "Wenlong Zhang", "Lei Bai", "Jing Shao" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "agent-safety", "computer-use", "rag", "tool-use" ], "score": 14, "relevance": "high", "primary_query": "autonomous-agent-llm", "matched_queries": [ "autonomous-agent-llm" ] }, { "id": "2607.17879", "arxiv_id": "2607.17879", "source": "arxiv", "source_id": "arxiv:2607.17879", "title": "Exploratory and Assimilating Reflection: Reflective Recall Cycle for Long-term Memory", "url": "https://arxiv.org/abs/2607.17879", "pdf_url": "https://arxiv.org/pdf/2607.17879", "published": "2026-07-20", "updated": "2026-07-20", "authors": [ "Ganesh Senrayan", "Moyuru Yamada", "Ishan Jindal", "Kiran Purohit" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "memory", "rag", "reasoning", "tool-use" ], "score": 14, "relevance": "high", "primary_query": "autonomous-agent-llm", "matched_queries": [ "autonomous-agent-llm" ] }, { "id": "2607.13705", "arxiv_id": "2607.13705", "source": "arxiv", "source_id": "arxiv:2607.13705", "title": "AgentCompass: A Unified Evaluation Infrastructure for Agent Capabilities", "url": "https://arxiv.org/abs/2607.13705", "pdf_url": "https://arxiv.org/pdf/2607.13705", "published": "2026-07-20", "updated": "2026-07-20", "authors": [ "Kai Chen", "Zichen Ding", "Jiaye Ge", "Shufan Jiang", "Mo Li", "Qingqiu Li", "Zehao Li", "Zonglin Li", "Tianhao Liang", "Shudong Liu", "Zerun Ma", "Zixin Shang", "Wenhui Tian", "Zun Wang", "Liwei Wu", "Zhenyu Wu", "Jun Xu", "Bowen Yang", "Dingbo Yuan", "Qi Zhang", "Songyang Zhang", "Peiheng Zhou", "Dongsheng Zhu" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "rag", "tool-use" ], "score": 14, "relevance": "high", "primary_query": "autonomous-agent-llm", "matched_queries": [ "autonomous-agent-llm" ] }, { "id": "2607.16708", "arxiv_id": "2607.16708", "source": "arxiv", "source_id": "arxiv:2607.16708", "title": "Model-Driven Discipline for Multi-Agent LLMs: Requirement-to-Verification Generation of Traceable System Models", "url": "https://arxiv.org/abs/2607.16708", "pdf_url": "https://arxiv.org/pdf/2607.16708", "published": "2026-07-18", "updated": "2026-07-18", "authors": [ "Ran Wei", "Le Zhu", "Haochi Wang", "Ruizhe Yang", "Jiapeng Guan", "Siyuan Ji", "Yuchen Hu", "Zhe Jiang", "Xiangyang Ji" ], "categories": [ "cs.SE" ], "topics": [ "agent-evaluation", "agent-safety", "multi-agent", "tool-use" ], "score": 14, "relevance": "high", "primary_query": "multi-agent-llm", "matched_queries": [ "multi-agent-llm" ] }, { "id": "2607.16133", "arxiv_id": "2607.16133", "source": "arxiv", "source_id": "arxiv:2607.16133", "title": "When Do Multi-Agent Systems Help? An Information Bottleneck Perspective", "url": "https://arxiv.org/abs/2607.16133", "pdf_url": "https://arxiv.org/pdf/2607.16133", "published": "2026-07-17", "updated": "2026-07-17", "authors": [ "Wendi Yu", "Lianhao Zhou", "Xiangjue Dong", "Sai Sudarshan Barath", "Declan Staunton", "Byung-Jun Yoon", "Xiaoning Qian", "James Caverlee", "Shuiwang Ji" ], "categories": [ "cs.LG" ], "topics": [ "agent-evaluation", "multi-agent", "reasoning" ], "score": 14, "relevance": "high", "primary_query": "multi-agent-llm", "matched_queries": [ "multi-agent-llm" ] }, { "id": "2607.15193", "arxiv_id": "2607.15193", "source": "arxiv", "source_id": "arxiv:2607.15193", "title": "Plover: Steering GUI Agents through Plan-Centric Interaction", "url": "https://arxiv.org/abs/2607.15193", "pdf_url": "https://arxiv.org/pdf/2607.15193", "published": "2026-07-16", "updated": "2026-07-16", "authors": [ "Madhumitha Venkatesan", "Shicheng Wen", "Jiajing Guo", "Jorge Piazentin Ono", "Liu Ren", "Dongyu Liu" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "computer-use", "planning", "tool-use", "workflow-agent" ], "score": 14, "relevance": "high", "primary_query": "web-gui-agent", "matched_queries": [ "web-gui-agent" ] }, { "id": "2607.13618", "arxiv_id": "2607.13618", "source": "arxiv", "source_id": "arxiv:2607.13618", "title": "STOCKTAKE: Measuring the Gap Between Perception and Action in LLM Agents with a Fair Oracle", "url": "https://arxiv.org/abs/2607.13618", "pdf_url": "https://arxiv.org/pdf/2607.13618", "published": "2026-07-15", "updated": "2026-07-15", "authors": [ "Sagar Deb", "Ashwanth Krishnan" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "tool-use" ], "score": 14, "relevance": "high", "primary_query": "llm-agent", "matched_queries": [ "llm-agent" ] }, { "id": "2607.14386", "arxiv_id": "2607.14386", "source": "arxiv", "source_id": "arxiv:2607.14386", "title": "CIPHER: A Decoupled Exploration-Selection Framework for Test-Time Scaling of Data Science Agents", "url": "https://arxiv.org/abs/2607.14386", "pdf_url": "https://arxiv.org/pdf/2607.14386", "published": "2026-07-15", "updated": "2026-07-15", "authors": [ "Maxime Heuillet", "Sharadind Peddiraju" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "rag", "tool-use", "workflow-agent" ], "score": 14, "relevance": "high", "primary_query": "ai-agent", "matched_queries": [ "ai-agent" ] }, { "id": "2607.12122", "arxiv_id": "2607.12122", "source": "arxiv", "source_id": "arxiv:2607.12122", "title": "An Agentic AI Scientific Community for Automated Neural Operator Discovery", "url": "https://arxiv.org/abs/2607.12122", "pdf_url": "https://arxiv.org/pdf/2607.12122", "published": "2026-07-13", "updated": "2026-07-13", "authors": [ "Luis Loo", "Ulisses Braga-Neto" ], "categories": [ "cs.LG" ], "topics": [ "agent-evaluation", "planning", "world-model" ], "score": 14, "relevance": "high", "primary_query": "agentic-ai", "matched_queries": [ "agentic-ai" ] }, { "id": "2607.11444", "arxiv_id": "2607.11444", "source": "arxiv", "source_id": "arxiv:2607.11444", "title": "UMoE:Unlocking Every Expert in Domain-Specific Training", "url": "https://arxiv.org/abs/2607.11444", "pdf_url": "https://arxiv.org/pdf/2607.11444", "published": "2026-07-13", "updated": "2026-07-13", "authors": [ "Xuefeng Li", "Pengfei Liu" ], "categories": [ "cs.CL" ], "topics": [ "agent-evaluation", "coding-agent", "rag", "tool-use" ], "score": 14, "relevance": "high", "primary_query": "coding-agent", "matched_queries": [ "coding-agent", "tool-use" ] }, { "id": "2607.11126", "arxiv_id": "2607.11126", "source": "arxiv", "source_id": "arxiv:2607.11126", "title": "ToolAtlas: Learning Once, Reusing Everywhere with Tool-Side Memory", "url": "https://arxiv.org/abs/2607.11126", "pdf_url": "https://arxiv.org/pdf/2607.11126", "published": "2026-07-13", "updated": "2026-07-13", "authors": [ "Yue Fang", "Zhibang Yang", "Fangkai Yang", "Xiaoting Qin", "Liqun Li", "Qingwei Lin", "Saravan Rajmohan", "Dongmei Zhang" ], "categories": [ "cs.LG" ], "topics": [ "agent-evaluation", "computer-use", "memory", "tool-use" ], "score": 14, "relevance": "high", "primary_query": "tool-use", "matched_queries": [ "tool-use" ] }, { "id": "2607.10265", "arxiv_id": "2607.10265", "source": "arxiv", "source_id": "arxiv:2607.10265", "title": "TGMS: An Agent-Native Bi-Temporal Graph Management System", "url": "https://arxiv.org/abs/2607.10265", "pdf_url": "https://arxiv.org/pdf/2607.10265", "published": "2026-07-11", "updated": "2026-07-11", "authors": [ "Xiaofei Zhang" ], "categories": [ "cs.DB" ], "topics": [ "agent-evaluation", "planning", "rag", "reasoning", "tool-use" ], "score": 14, "relevance": "high", "primary_query": "planning-agent", "matched_queries": [ "planning-agent", "rag-agent" ] }, { "id": "2607.10057", "arxiv_id": "2607.10057", "source": "arxiv", "source_id": "arxiv:2607.10057", "title": "Quantum Circuit Vision: Cost-Aware Evaluation of Visual AI Agents for Quantum Code Generation", "url": "https://arxiv.org/abs/2607.10057", "pdf_url": "https://arxiv.org/pdf/2607.10057", "published": "2026-07-10", "updated": "2026-07-10", "authors": [ "Dongping Liu", "Aoyu Zhang", "Luyao Zhang" ], "categories": [ "quant-ph" ], "topics": [ "agent-evaluation", "reasoning" ], "score": 14, "relevance": "high", "primary_query": "ai-agent", "matched_queries": [ "ai-agent" ] }, { "id": "2607.09996", "arxiv_id": "2607.09996", "source": "arxiv", "source_id": "arxiv:2607.09996", "title": "Who&When Pro: Can LLMs Really Attribute Failures in AI Agents?", "url": "https://arxiv.org/abs/2607.09996", "pdf_url": "https://arxiv.org/pdf/2607.09996", "published": "2026-07-10", "updated": "2026-07-10", "authors": [ "Jiale Liu", "Huajun Xi", "Shaokun Zhang", "Yifan Zeng", "Tianwei Yue", "Chi Wang", "Jian Kang", "Qingyun Wu", "Huazheng Wang" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "computer-use" ], "score": 14, "relevance": "high", "primary_query": "ai-agent", "matched_queries": [ "ai-agent" ] }, { "id": "2607.09902", "arxiv_id": "2607.09902", "source": "arxiv", "source_id": "arxiv:2607.09902", "title": "Do These Violent Delights Have Violent Ends? Measuring the Post-Merge Fate of Agentic Code", "url": "https://arxiv.org/abs/2607.09902", "pdf_url": "https://arxiv.org/pdf/2607.09902", "published": "2026-07-10", "updated": "2026-07-10", "authors": [ "Chunqiu Steven Xia", "Courtney Miller" ], "categories": [ "cs.SE" ], "topics": [ "agent-evaluation", "agent-safety", "coding-agent", "rag", "tool-use" ], "score": 14, "relevance": "high", "primary_query": "coding-agent", "matched_queries": [ "coding-agent" ] }, { "id": "2607.09553", "arxiv_id": "2607.09553", "source": "arxiv", "source_id": "arxiv:2607.09553", "title": "Writing Bug Reports for Software Repair Agents: What Information Matters Most?", "url": "https://arxiv.org/abs/2607.09553", "pdf_url": "https://arxiv.org/pdf/2607.09553", "published": "2026-07-10", "updated": "2026-07-10", "authors": [ "Vincenzo Luigi Bruno", "Alessandro Giagnorio", "Daniele Bifolco", "Leon Wienges", "Massimiliano Di Penta", "Gabriele Bavota" ], "categories": [ "cs.SE" ], "topics": [ "agent-evaluation", "coding-agent", "workflow-agent" ], "score": 14, "relevance": "high", "primary_query": "coding-agent", "matched_queries": [ "coding-agent" ] }, { "id": "2607.08716", "arxiv_id": "2607.08716", "source": "arxiv", "source_id": "arxiv:2607.08716", "title": "Remember When It Matters: Proactive Memory Agent for Long-Horizon Agents", "url": "https://arxiv.org/abs/2607.08716", "pdf_url": "https://arxiv.org/pdf/2607.08716", "published": "2026-07-09", "updated": "2026-07-09", "authors": [ "Yifan Wu", "Lizhu Zhang", "Yuhang Zhou", "Mingyi Wang", "Bo Peng", "Serena Li", "Xiangjun Fan", "Zhuokai Zhao" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "computer-use", "memory", "planning", "rag", "tool-use" ], "score": 14, "relevance": "high", "primary_query": "agent-memory", "matched_queries": [ "agent-memory" ] }, { "id": "2607.08983", "arxiv_id": "2607.08983", "source": "arxiv", "source_id": "arxiv:2607.08983", "title": "SCATE: Learning to Supervise Coding Agents for Cost-Effective Test Generation", "url": "https://arxiv.org/abs/2607.08983", "pdf_url": "https://arxiv.org/pdf/2607.08983", "published": "2026-07-09", "updated": "2026-07-09", "authors": [ "Sijia Gu", "Noor Nashid", "Ali Mesbah" ], "categories": [ "cs.SE" ], "topics": [ "agent-evaluation", "coding-agent", "rag", "tool-use" ], "score": 14, "relevance": "high", "primary_query": "coding-agent", "matched_queries": [ "coding-agent" ] }, { "id": "2607.22031", "arxiv_id": "2607.22031", "source": "arxiv", "source_id": "arxiv:2607.22031", "title": "IDSTune: A Multi-Agent Collaborative Framework for Integrated Database System Tuning", "url": "https://arxiv.org/abs/2607.22031", "pdf_url": "https://arxiv.org/pdf/2607.22031", "published": "2026-07-24", "updated": "2026-07-24", "authors": [ "Yiyan Li", "Guanli Liu", "Renata Borovica-Gajic", "Haoyang Li", "Zihang Qiu", "Xinmei Huang", "Andreas Kipf", "Cuiping Li", "Hong Chen" ], "categories": [ "cs.DB" ], "topics": [ "agent-evaluation", "multi-agent", "rag" ], "score": 13, "relevance": "high", "primary_query": "multi-agent-llm", "matched_queries": [ "multi-agent-llm" ] }, { "id": "2607.21957", "arxiv_id": "2607.21957", "source": "arxiv", "source_id": "arxiv:2607.21957", "title": "KaPilot: LLM-Assisted Generation of Kani Specifications for Unsafe Rust Verification", "url": "https://arxiv.org/abs/2607.21957", "pdf_url": "https://arxiv.org/pdf/2607.21957", "published": "2026-07-24", "updated": "2026-07-24", "authors": [ "Minghua Wang", "Yuxi Ling", "Mingzhi Gao", "Yuwei Liu", "Lin Huang" ], "categories": [ "cs.SE" ], "topics": [ "agent-evaluation", "agent-safety", "computer-use", "memory", "multi-agent", "tool-use" ], "score": 13, "relevance": "high", "primary_query": "multi-agent-llm", "matched_queries": [ "multi-agent-llm" ] }, { "id": "2607.21824", "arxiv_id": "2607.21824", "source": "arxiv", "source_id": "arxiv:2607.21824", "title": "Protocol-Level Attacks on Agentic Commerce Platforms: A Cross-Platform Taxonomy, AIP-Bench, and Unified Defense", "url": "https://arxiv.org/abs/2607.21824", "pdf_url": "https://arxiv.org/pdf/2607.21824", "published": "2026-07-23", "updated": "2026-07-23", "authors": [ "Yedidel Louck" ], "categories": [ "cs.CR" ], "topics": [ "agent-evaluation", "agent-safety", "tool-use" ], "score": 13, "relevance": "high", "primary_query": "ai-agent", "matched_queries": [ "ai-agent" ] }, { "id": "2607.21495", "arxiv_id": "2607.21495", "source": "arxiv", "source_id": "arxiv:2607.21495", "title": "Toward Continuous Assurance for the Democratization of AI Agent Creation in Industry", "url": "https://arxiv.org/abs/2607.21495", "pdf_url": "https://arxiv.org/pdf/2607.21495", "published": "2026-07-23", "updated": "2026-07-23", "authors": [ "Natan Levy", "Harel Berger" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "computer-use", "rag", "tool-use", "workflow-agent" ], "score": 13, "relevance": "high", "primary_query": "ai-agent", "matched_queries": [ "ai-agent" ] }, { "id": "2607.20062", "arxiv_id": "2607.20062", "source": "arxiv", "source_id": "arxiv:2607.20062", "title": "Solar Open 2 Technical Report", "url": "https://arxiv.org/abs/2607.20062", "pdf_url": "https://arxiv.org/pdf/2607.20062", "published": "2026-07-23", "updated": "2026-07-23", "authors": [ "Sungrae Park", "Sanghoon Kim", "Gyoungjin Gim", "Jungho Cho", "Hyunwoong Ko", "Minbyul Jeong", "Minjeong Kim", "Keunwoo Choi", "Chaehun Shin", "Chanwoong Yoon", "Dongjun Kim", "Eunwon Kim", "Gyungin Shin", "Hyeonju Lee", "Hyungkyu Kang", "Inseo Song", "Jisu Bae", "Jiyoon Han", "Jiyun Lee", "Joonkee Kim", "Junyeop Lee", "Mikyoung Cha", "Sangwon Yu", "Sehwan Joo", "Seokyoon Kang", "et al. (28 additional authors not shown)" ], "categories": [ "cs.CL" ], "topics": [ "agent-evaluation", "coding-agent", "planning", "rag", "tool-use", "workflow-agent" ], "score": 13, "relevance": "high", "primary_query": "agent-evaluation", "matched_queries": [ "agent-evaluation" ] }, { "id": "2607.21268", "arxiv_id": "2607.21268", "source": "arxiv", "source_id": "arxiv:2607.21268", "title": "pAI-Econ-claude: A Gated Human-in-the-Loop Multi-Agent Architecture for AI-Assisted Economic Theory Development", "url": "https://arxiv.org/abs/2607.21268", "pdf_url": "https://arxiv.org/pdf/2607.21268", "published": "2026-07-23", "updated": "2026-07-23", "authors": [ "Chen Zhu", "Xiaolu Wang", "Weilong Zhang" ], "categories": [ "cs.MA" ], "topics": [ "agent-evaluation", "multi-agent", "workflow-agent" ], "score": 13, "relevance": "high", "primary_query": "multi-agent-llm", "matched_queries": [ "multi-agent-llm" ] }, { "id": "2607.03316", "arxiv_id": "2607.03316", "source": "arxiv", "source_id": "arxiv:2607.03316", "title": "Is Agentic Code Review Helpful? Mining Developers' Feedback to CodeRabbit Reviews in the Wild", "url": "https://arxiv.org/abs/2607.03316", "pdf_url": "https://arxiv.org/pdf/2607.03316", "published": "2026-07-23", "updated": "2026-07-23", "authors": [ "Hong Yi Lin", "Mingzhao Liang", "Patanamon Thongtanunam", "Kla Tantithamthavorn" ], "categories": [ "cs.SE" ], "topics": [ "agent-safety", "coding-agent", "workflow-agent" ], "score": 13, "relevance": "high", "primary_query": "autonomous-agent-llm", "matched_queries": [ "autonomous-agent-llm" ] }, { "id": "2607.19338", "arxiv_id": "2607.19338", "source": "arxiv", "source_id": "arxiv:2607.19338", "title": "CodeRescue: Budget-Calibrated Recovery Routing for Coding Agents", "url": "https://arxiv.org/abs/2607.19338", "pdf_url": "https://arxiv.org/pdf/2607.19338", "published": "2026-07-21", "updated": "2026-07-21", "authors": [ "Qijia He", "Jiayi Cheng", "Chenqian Le", "Rui Wang", "Xunmei Liu", "Yixian Chen", "Jie Mei", "Zhihao Wang", "Xupeng Chen", "Yuhuan Chen", "Tao Wang" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "agent-safety", "coding-agent", "tool-use" ], "score": 13, "relevance": "high", "primary_query": "coding-agent", "matched_queries": [ "coding-agent" ] }, { "id": "2607.18138", "arxiv_id": "2607.18138", "source": "arxiv", "source_id": "arxiv:2607.18138", "title": "AI Agent Communications in AI-Native 6G Network: Status, Challenges and Opportunities", "url": "https://arxiv.org/abs/2607.18138", "pdf_url": "https://arxiv.org/pdf/2607.18138", "published": "2026-07-20", "updated": "2026-07-20", "authors": [ "Qiang Duan" ], "categories": [ "cs.NI" ], "topics": [ "multi-agent", "tool-use" ], "score": 13, "relevance": "high", "primary_query": "agentic-ai", "matched_queries": [ "agentic-ai", "ai-agent" ] }, { "id": "2607.16057", "arxiv_id": "2607.16057", "source": "arxiv", "source_id": "arxiv:2607.16057", "title": "Frontier AI performance across the business disciplines: a case-grounded benchmark of knowledge work and analytical reasoning", "url": "https://arxiv.org/abs/2607.16057", "pdf_url": "https://arxiv.org/pdf/2607.16057", "published": "2026-07-20", "updated": "2026-07-20", "authors": [ "Ajay Patel", "Kartik Hosanagar", "Ramayya Krishnan", "Chris Callison-Burch", "Karim Lakhani" ], "categories": [ "cs.CL" ], "topics": [ "agent-evaluation", "coding-agent", "reasoning", "tool-use" ], "score": 13, "relevance": "high", "primary_query": "tool-use", "matched_queries": [ "tool-use" ] }, { "id": "2607.18171", "arxiv_id": "2607.18171", "source": "arxiv", "source_id": "arxiv:2607.18171", "title": "FlashRT: Agent Harness for Guiding Agents to Deploy Real-Time Multimodal Applications", "url": "https://arxiv.org/abs/2607.18171", "pdf_url": "https://arxiv.org/pdf/2607.18171", "published": "2026-07-20", "updated": "2026-07-20", "authors": [ "Krish Agarwal", "Zhuoming Chen", "Yanyuan Qin", "Zhenyu Gu", "Atri Rudra", "Beidi Chen" ], "categories": [ "cs.LG" ], "topics": [ "agent-evaluation", "coding-agent", "computer-use", "world-model" ], "score": 13, "relevance": "high", "primary_query": "coding-agent", "matched_queries": [ "coding-agent" ] }, { "id": "2607.17527", "arxiv_id": "2607.17527", "source": "arxiv", "source_id": "arxiv:2607.17527", "title": "Sidekick: Designing Communication for Effective Multitasking with Computer Use Agents", "url": "https://arxiv.org/abs/2607.17527", "pdf_url": "https://arxiv.org/pdf/2607.17527", "published": "2026-07-20", "updated": "2026-07-20", "authors": [ "Ruei-Che Chang", "Wenqian Xu", "Dingzeyu Li", "Bryan Wang", "Anhong Guo" ], "categories": [ "cs.HC" ], "topics": [ "computer-use", "multi-agent", "planning", "reasoning", "tool-use" ], "score": 13, "relevance": "high", "primary_query": "web-gui-agent", "matched_queries": [ "web-gui-agent" ] }, { "id": "2607.17149", "arxiv_id": "2607.17149", "source": "arxiv", "source_id": "arxiv:2607.17149", "title": "A Diagnostic Framework for AI Agent Behavior", "url": "https://arxiv.org/abs/2607.17149", "pdf_url": "https://arxiv.org/pdf/2607.17149", "published": "2026-07-19", "updated": "2026-07-19", "authors": [ "Xichen Zhang", "Yingjie Zhang", "Tianshu Sun" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "memory", "tool-use" ], "score": 13, "relevance": "high", "primary_query": "ai-agent", "matched_queries": [ "ai-agent" ] }, { "id": "2607.15518", "arxiv_id": "2607.15518", "source": "arxiv", "source_id": "arxiv:2607.15518", "title": "A Tool-Invariant Framework for Teaching and Assessing Computational Methods in the Age of Agentic AI", "url": "https://arxiv.org/abs/2607.15518", "pdf_url": "https://arxiv.org/pdf/2607.15518", "published": "2026-07-16", "updated": "2026-07-16", "authors": [ "Larry Engelhardt" ], "categories": [ "physics.ed-ph" ], "topics": [ "agent-evaluation", "coding-agent", "tool-use", "world-model" ], "score": 13, "relevance": "high", "primary_query": "agentic-ai", "matched_queries": [ "agentic-ai", "language-agent" ] }, { "id": "2607.14548", "arxiv_id": "2607.14548", "source": "arxiv", "source_id": "arxiv:2607.14548", "title": "HyMobileAgent: Data-Environment Co-Scaling for Efficient GUI Agents", "url": "https://arxiv.org/abs/2607.14548", "pdf_url": "https://arxiv.org/pdf/2607.14548", "published": "2026-07-16", "updated": "2026-07-16", "authors": [ "Hy Vision Team", "Huawen Shen", "Zhengyang Tang", "Shangpin Peng", "Liang Wu", "Anran Zhang", "Weinong Wang", "Yiduo Guo", "Chenxin Li", "Zhengyao Fang", "Yang Ding", "Junyi Li", "Fei Tang", "Zheng Ruan", "Yi Zhang", "Xingran Zhou", "Dingchen Yang", "Sunqi Fan", "Zhiyi Wan", "Han Hu", "Xin Lai", "Pengyuan Lyu", "Chengquan Zhang" ], "categories": [ "cs.CV" ], "topics": [ "computer-use", "embodied-agent", "planning", "reasoning", "tool-use" ], "score": 13, "relevance": "high", "primary_query": "web-gui-agent", "matched_queries": [ "web-gui-agent" ] }, { "id": "2607.14456", "arxiv_id": "2607.14456", "source": "arxiv", "source_id": "arxiv:2607.14456", "title": "Beyond Generalist LLMs: Specialist Agentic Systems for Structured Code Workflow Execution", "url": "https://arxiv.org/abs/2607.14456", "pdf_url": "https://arxiv.org/pdf/2607.14456", "published": "2026-07-15", "updated": "2026-07-15", "authors": [ "Harris Borman", "Herman Wandabwa", "Fusun Yu", "Sandeepa Kannangara", "Justin Liu", "Anna Leontjeva", "Ritchie Ng" ], "categories": [ "cs.SE" ], "topics": [ "tool-use", "workflow-agent" ], "score": 13, "relevance": "high", "primary_query": "agentic-ai", "matched_queries": [ "agentic-ai", "tool-use" ] }, { "id": "2607.13027", "arxiv_id": "2607.13027", "source": "arxiv", "source_id": "arxiv:2607.13027", "title": "PalmClaw: A Native On-Device Agent Framework for Mobile Phones", "url": "https://arxiv.org/abs/2607.13027", "pdf_url": "https://arxiv.org/pdf/2607.13027", "published": "2026-07-14", "updated": "2026-07-14", "authors": [ "Hongru Cai", "Yongqi Li", "Ran Wei", "Wenjie Li" ], "categories": [ "cs.CL" ], "topics": [ "computer-use", "memory", "tool-use", "workflow-agent" ], "score": 13, "relevance": "high", "primary_query": "tool-use", "matched_queries": [ "tool-use" ] }, { "id": "2607.12056", "arxiv_id": "2607.12056", "source": "arxiv", "source_id": "arxiv:2607.12056", "title": "Designing Agent-Ready Websites for AI Web Agents: A Framework for Machine Readability, Actionability, and Decision Reliability", "url": "https://arxiv.org/abs/2607.12056", "pdf_url": "https://arxiv.org/pdf/2607.12056", "published": "2026-07-13", "updated": "2026-07-13", "authors": [ "Said Elnaffar", "Farzad Rashidi" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "computer-use", "rag", "tool-use", "workflow-agent" ], "score": 13, "relevance": "high", "primary_query": "ai-agent", "matched_queries": [ "ai-agent", "web-gui-agent" ] }, { "id": "2607.11423", "arxiv_id": "2607.11423", "source": "arxiv", "source_id": "arxiv:2607.11423", "title": "ToFu: A White-Box, Token-Efficient Agent Harness for Researchers", "url": "https://arxiv.org/abs/2607.11423", "pdf_url": "https://arxiv.org/pdf/2607.11423", "published": "2026-07-13", "updated": "2026-07-13", "authors": [ "Junhao Ruan", "Yuan Ge", "Bei Li", "Yongjing Yin", "Yuchun Fan", "Xin Chen", "Jingang Wang", "Chenglong Wang", "Jingbo Zhu", "Tong Xiao" ], "categories": [ "cs.CL" ], "topics": [ "agent-evaluation", "coding-agent", "tool-use", "workflow-agent" ], "score": 13, "relevance": "high", "primary_query": "tool-use", "matched_queries": [ "tool-use" ] }, { "id": "2607.11565", "arxiv_id": "2607.11565", "source": "arxiv", "source_id": "arxiv:2607.11565", "title": "Heuristic Learning for Active Flow Control Using Coding Agents", "url": "https://arxiv.org/abs/2607.11565", "pdf_url": "https://arxiv.org/pdf/2607.11565", "published": "2026-07-13", "updated": "2026-07-13", "authors": [ "Paul Garnier", "Jonathan Viquerat", "Elie Hachem" ], "categories": [ "cs.LG" ], "topics": [ "agent-evaluation", "coding-agent", "tool-use", "world-model" ], "score": 13, "relevance": "high", "primary_query": "coding-agent", "matched_queries": [ "coding-agent" ] }, { "id": "2607.11388", "arxiv_id": "2607.11388", "source": "arxiv", "source_id": "arxiv:2607.11388", "title": "StructAgent: Harness Long-horizon Digital Agents with Unified Causal Structure", "url": "https://arxiv.org/abs/2607.11388", "pdf_url": "https://arxiv.org/pdf/2607.11388", "published": "2026-07-13", "updated": "2026-07-13", "authors": [ "Wenyi Wu", "Sibo Zhu", "Kun Zhou", "Aayush Salvi", "Zixuan Song", "Biwei Huang" ], "categories": [ "cs.AI" ], "topics": [ "computer-use", "planning", "reasoning", "tool-use", "workflow-agent" ], "score": 13, "relevance": "high", "primary_query": "web-gui-agent", "matched_queries": [ "web-gui-agent" ] }, { "id": "2607.11185", "arxiv_id": "2607.11185", "source": "arxiv", "source_id": "arxiv:2607.11185", "title": "SCALECUA: Scaling Computer Use Agents with Verifiable Task Synthesis and Efficient Online RL", "url": "https://arxiv.org/abs/2607.11185", "pdf_url": "https://arxiv.org/pdf/2607.11185", "published": "2026-07-13", "updated": "2026-07-13", "authors": [ "Bowen Lv", "Xiao Liu", "Yanyu Ren", "Hanyu Lai", "Bohao Jing", "Hanchen Zhang", "Yanxiao Zhao", "Shuntian Yao", "Jie Tang", "Yuxiao Dong" ], "categories": [ "cs.AI" ], "topics": [ "computer-use", "multi-agent", "tool-use", "workflow-agent" ], "score": 13, "relevance": "high", "primary_query": "web-gui-agent", "matched_queries": [ "web-gui-agent" ] }, { "id": "2607.11119", "arxiv_id": "2607.11119", "source": "arxiv", "source_id": "arxiv:2607.11119", "title": "VIA: Visual Interface Agent for Robot Control", "url": "https://arxiv.org/abs/2607.11119", "pdf_url": "https://arxiv.org/pdf/2607.11119", "published": "2026-07-13", "updated": "2026-07-13", "authors": [ "Hengyuan Hu", "Priya Sundaresan", "Jensen Gao", "Dorsa Sadigh" ], "categories": [ "cs.RO" ], "topics": [ "coding-agent", "computer-use", "embodied-agent", "planning", "rag", "reasoning", "tool-use" ], "score": 13, "relevance": "high", "primary_query": "web-gui-agent", "matched_queries": [ "web-gui-agent" ] }, { "id": "2607.12215", "arxiv_id": "2607.12215", "source": "arxiv", "source_id": "arxiv:2607.12215", "title": "Fine-Tuned Multi-Agent Framework for Detecting OCEAN in Life Narratives", "url": "https://arxiv.org/abs/2607.12215", "pdf_url": "https://arxiv.org/pdf/2607.12215", "published": "2026-07-13", "updated": "2026-07-13", "authors": [ "Rasiq Hussain", "Darshil Italiya", "Joshua Oltmanns", "Mehak Gupta" ], "categories": [ "cs.CL" ], "topics": [ "agent-evaluation", "multi-agent", "reasoning" ], "score": 13, "relevance": "high", "primary_query": "multi-agent-llm", "matched_queries": [ "multi-agent-llm" ] }, { "id": "2607.21013", "arxiv_id": "2607.21013", "source": "arxiv", "source_id": "arxiv:2607.21013", "title": "EmoAgent-R1: Towards Multimodal Emotion Understanding with Reinforcement Learning-based Dynamic Agent Specialization", "url": "https://arxiv.org/abs/2607.21013", "pdf_url": "https://arxiv.org/pdf/2607.21013", "published": "2026-07-23", "updated": "2026-07-23", "authors": [ "Lihuang Fang", "Yuchen Zou", "kebin Jin", "Jinghui Qin" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "reasoning", "workflow-agent" ], "score": 12, "relevance": "medium", "primary_query": "agentic-ai", "matched_queries": [ "agentic-ai" ] }, { "id": "2607.21461", "arxiv_id": "2607.21461", "source": "arxiv", "source_id": "arxiv:2607.21461", "title": "AREX: Towards a Recursively Self-Improving Agent for Deep Research", "url": "https://arxiv.org/abs/2607.21461", "pdf_url": "https://arxiv.org/pdf/2607.21461", "published": "2026-07-23", "updated": "2026-07-23", "authors": [ "Shuqi Lu", "Chaofan Li", "Kun Luo", "Zhang Zhang", "Hui Wang", "Hongwang Xiao", "Lei Xiong", "Jiahao Wang", "Sen Wang", "Xiyan Jiang", "Wanli Li", "Yuyang Hu", "Hongjin Qian", "Bingyu Yan", "Jianlyu Chen", "Ziyi Xia", "Yingxia Shao", "Kang Liu", "Zhicheng Dou", "Di He", "Chaozhuo Li", "Qiwei Ye", "Zhongyuan Wang", "Zheng Liu" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "computer-use", "planning", "reasoning", "tool-use" ], "score": 12, "relevance": "medium", "primary_query": "tool-use", "matched_queries": [ "tool-use" ] }, { "id": "2607.20690", "arxiv_id": "2607.20690", "source": "arxiv", "source_id": "arxiv:2607.20690", "title": "Learning to Detect UI Principle Violations via Reinforcement Learning", "url": "https://arxiv.org/abs/2607.20690", "pdf_url": "https://arxiv.org/pdf/2607.20690", "published": "2026-07-22", "updated": "2026-07-22", "authors": [ "Nishi Mehta", "Swathi Alse", "Himani Kumavat", "Yue Yu", "Pratik Jayarao" ], "categories": [ "cs.CL" ], "topics": [ "agent-evaluation", "coding-agent", "rag", "reasoning", "tool-use" ], "score": 12, "relevance": "medium", "primary_query": "coding-agent", "matched_queries": [ "coding-agent" ] }, { "id": "2607.19794", "arxiv_id": "2607.19794", "source": "arxiv", "source_id": "arxiv:2607.19794", "title": "TriAgent: Divergence-Aware Multi-Agent Committees for Cost-Efficient Financial Sentiment Analysis", "url": "https://arxiv.org/abs/2607.19794", "pdf_url": "https://arxiv.org/pdf/2607.19794", "published": "2026-07-22", "updated": "2026-07-22", "authors": [ "Isabel Xu", "Cynthia Xu", "Rachel Ren", "Cong Guo", "Jiacheng Ding" ], "categories": [ "cs.CL" ], "topics": [ "agent-safety", "multi-agent" ], "score": 12, "relevance": "medium", "primary_query": "multi-agent-llm", "matched_queries": [ "multi-agent-llm" ] }, { "id": "2607.19703", "arxiv_id": "2607.19703", "source": "arxiv", "source_id": "arxiv:2607.19703", "title": "Bridging Behavior and Implementation: Automated Java Glue Code Generation for Behavior-Driven Development", "url": "https://arxiv.org/abs/2607.19703", "pdf_url": "https://arxiv.org/pdf/2607.19703", "published": "2026-07-21", "updated": "2026-07-21", "authors": [ "Xinyu Shi", "Zhou Yang", "An Ran Chen" ], "categories": [ "cs.SE" ], "topics": [ "agent-evaluation", "multi-agent", "rag", "reasoning", "tool-use", "workflow-agent" ], "score": 12, "relevance": "medium", "primary_query": "multi-agent-llm", "matched_queries": [ "multi-agent-llm" ] }, { "id": "2607.17937", "arxiv_id": "2607.17937", "source": "arxiv", "source_id": "arxiv:2607.17937", "title": "How Agent Skills Fail under Long Contexts: A White-Box Study in Code Auditing", "url": "https://arxiv.org/abs/2607.17937", "pdf_url": "https://arxiv.org/pdf/2607.17937", "published": "2026-07-20", "updated": "2026-07-20", "authors": [ "Yue Xue" ], "categories": [ "cs.SE" ], "topics": [ "agent-evaluation", "coding-agent", "rag", "tool-use", "workflow-agent" ], "score": 12, "relevance": "medium", "primary_query": "agent-evaluation", "matched_queries": [ "agent-evaluation", "coding-agent", "tool-use" ] }, { "id": "2607.17225", "arxiv_id": "2607.17225", "source": "arxiv", "source_id": "arxiv:2607.17225", "title": "Specifying the Delegated-Autonomy Boundary: Requirements Engineering for Agentic AI", "url": "https://arxiv.org/abs/2607.17225", "pdf_url": "https://arxiv.org/pdf/2607.17225", "published": "2026-07-19", "updated": "2026-07-19", "authors": [ "Chetan Arora", "Andreas Vogelsang", "Abbi Sharma" ], "categories": [ "cs.SE" ], "topics": [ "agent-safety", "planning", "tool-use" ], "score": 12, "relevance": "medium", "primary_query": "agentic-ai", "matched_queries": [ "agentic-ai" ] }, { "id": "2607.15684", "arxiv_id": "2607.15684", "source": "arxiv", "source_id": "arxiv:2607.15684", "title": "Understanding Agent-Reactive Bugs at the Model-Harness Boundary: An Empirical Study of LLM Agent Issue Reports", "url": "https://arxiv.org/abs/2607.15684", "pdf_url": "https://arxiv.org/pdf/2607.15684", "published": "2026-07-17", "updated": "2026-07-17", "authors": [ "Jingyi Chen", "Songqiang Chen", "Hengcheng Zhu", "Jialun Cao", "Jiasi Shen", "Shing-Chi Cheung" ], "categories": [ "cs.SE" ], "topics": [ "agent-safety", "tool-use" ], "score": 12, "relevance": "medium", "primary_query": "llm-agent", "matched_queries": [ "llm-agent" ] }, { "id": "2607.14754", "arxiv_id": "2607.14754", "source": "arxiv", "source_id": "arxiv:2607.14754", "title": "FlowGuard: From Signals to Evidence for MCP Security Detection", "url": "https://arxiv.org/abs/2607.14754", "pdf_url": "https://arxiv.org/pdf/2607.14754", "published": "2026-07-16", "updated": "2026-07-16", "authors": [ "Baichao An", "Pei Chen", "Geng Hong", "Yueyue Chen", "Mengying Wu" ], "categories": [ "cs.CR" ], "topics": [ "agent-evaluation", "agent-safety", "computer-use", "tool-use" ], "score": 12, "relevance": "medium", "primary_query": "llm-agent", "matched_queries": [ "llm-agent" ] }, { "id": "2607.15143", "arxiv_id": "2607.15143", "source": "arxiv", "source_id": "arxiv:2607.15143", "title": "Setup Complete, Now You Are Compromised: Weaponizing Setup Instructions Against AI Coding Agents", "url": "https://arxiv.org/abs/2607.15143", "pdf_url": "https://arxiv.org/pdf/2607.15143", "published": "2026-07-16", "updated": "2026-07-16", "authors": [ "Aadesh Bagmar", "Pushkar Saraf" ], "categories": [ "cs.CR" ], "topics": [ "agent-evaluation", "agent-safety", "coding-agent" ], "score": 12, "relevance": "medium", "primary_query": "coding-agent", "matched_queries": [ "coding-agent" ] }, { "id": "2607.11098", "arxiv_id": "2607.11098", "source": "arxiv", "source_id": "arxiv:2607.11098", "title": "AgentCheck: A Reproduce-Intervene-Mitigate Workbench for LLM Agents over MCP", "url": "https://arxiv.org/abs/2607.11098", "pdf_url": "https://arxiv.org/pdf/2607.11098", "published": "2026-07-15", "updated": "2026-07-15", "authors": [ "Aritra Mazumder", "Nusrat jahan Lia" ], "categories": [ "cs.SE" ], "topics": [ "agent-evaluation", "tool-use" ], "score": 12, "relevance": "medium", "primary_query": "tool-use", "matched_queries": [ "tool-use" ] }, { "id": "2607.13548", "arxiv_id": "2607.13548", "source": "arxiv", "source_id": "arxiv:2607.13548", "title": "How Far Can Root Cause Analysis Go on Real-World Telemetry Data?", "url": "https://arxiv.org/abs/2607.13548", "pdf_url": "https://arxiv.org/pdf/2607.13548", "published": "2026-07-15", "updated": "2026-07-15", "authors": [ "Athira Gopal", "Ashwanth Krishnan" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "computer-use", "multi-agent", "reasoning", "tool-use" ], "score": 12, "relevance": "medium", "primary_query": "multi-agent-llm", "matched_queries": [ "multi-agent-llm" ] }, { "id": "2607.12823", "arxiv_id": "2607.12823", "source": "arxiv", "source_id": "arxiv:2607.12823", "title": "Human-AI Agent Interaction as a Neuroplastic Training Environment", "url": "https://arxiv.org/abs/2607.12823", "pdf_url": "https://arxiv.org/pdf/2607.12823", "published": "2026-07-14", "updated": "2026-07-14", "authors": [ "Eranga Bandara", "Ross Gore", "Asanga Gunaratna", "Ravi Mukkamala", "Nihal Siriwardanagea", "Gihan Siriwardanagea", "Sachini Rajapakse", "Isurunima Kularathna", "Pramoda Karunarathna", "Chalani Rajapakse", "Sachin Shetty", "Christopher K. Rhea", "Ng Wee Keong", "Kasun De Zoysa", "Amin Hass", "Shaifali Kaushik", "Wathsala Herath", "Preston Samuel", "Anita H. Clayton", "Atmaram Yarlagadd" ], "categories": [ "cs.AI" ], "topics": [ "coding-agent", "computer-use", "memory", "tool-use" ], "score": 12, "relevance": "medium", "primary_query": "ai-agent", "matched_queries": [ "ai-agent" ] }, { "id": "2607.01531", "arxiv_id": "2607.01531", "source": "arxiv", "source_id": "arxiv:2607.01531", "title": "OPINE-World: Programmatic World Modeling with Ontology-error-Prioritized Interactive Exploration for ARC-AGI-3", "url": "https://arxiv.org/abs/2607.01531", "pdf_url": "https://arxiv.org/pdf/2607.01531", "published": "2026-07-14", "updated": "2026-07-14", "authors": [ "David Courtis", "Wenhao Li", "Scott Sanner" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "computer-use", "planning", "tool-use", "world-model" ], "score": 12, "relevance": "medium", "primary_query": "planning-agent", "matched_queries": [ "planning-agent" ] }, { "id": "2607.08193", "arxiv_id": "2607.08193", "source": "arxiv", "source_id": "arxiv:2607.08193", "title": "Open-ended Multi-agent Autocurricula via Visual Inspection of Policies with Multi-modal LLMs", "url": "https://arxiv.org/abs/2607.08193", "pdf_url": "https://arxiv.org/pdf/2607.08193", "published": "2026-07-09", "updated": "2026-07-09", "authors": [ "Lorenzo Pantè", "Andrea Fanti", "Roberto Capobianco" ], "categories": [ "cs.LG" ], "topics": [ "multi-agent", "rag" ], "score": 12, "relevance": "medium", "primary_query": "multi-agent-llm", "matched_queries": [ "multi-agent-llm" ] }, { "id": "2607.22448", "arxiv_id": "2607.22448", "source": "arxiv", "source_id": "arxiv:2607.22448", "title": "Where FactsGo Missing: A LayerwiseTaxonomy and Per-Layer Attribution of Information Omissionin Air-Gapped LLM Agent Pipelines", "url": "https://arxiv.org/abs/2607.22448", "pdf_url": "https://arxiv.org/pdf/2607.22448", "published": "2026-07-24", "updated": "2026-07-24", "authors": [ "Santhiya Rajan" ], "categories": [ "cs.MA" ], "topics": [ "tool-use" ], "score": 11, "relevance": "medium", "primary_query": "llm-agent", "matched_queries": [ "llm-agent" ] }, { "id": "2607.22529", "arxiv_id": "2607.22529", "source": "arxiv", "source_id": "arxiv:2607.22529", "title": "Skill Self-Play: Pushing the Frontier of LLM Capability with Co-Evolving Skills", "url": "https://arxiv.org/abs/2607.22529", "pdf_url": "https://arxiv.org/pdf/2607.22529", "published": "2026-07-24", "updated": "2026-07-24", "authors": [ "Siyuan Huang", "Pengyu Cheng", "Haotian Liu", "Tao Chen", "Yihao Liu", "Jingwei Ni", "Shijie Zhou", "Ziyi Yang", "Gangwei Jiang", "Mengyu Zhou", "Yu Cheng", "Xiaoxi Jiang", "Guanjun Jiang" ], "categories": [ "cs.CL" ], "topics": [ "agent-evaluation", "rag", "reasoning", "tool-use" ], "score": 11, "relevance": "medium", "primary_query": "tool-use", "matched_queries": [ "tool-use" ] }, { "id": "2607.21412", "arxiv_id": "2607.21412", "source": "arxiv", "source_id": "arxiv:2607.21412", "title": "Euclid-MCP: A Model Context Protocol Server for Deterministic Logical Reasoning via Prolog", "url": "https://arxiv.org/abs/2607.21412", "pdf_url": "https://arxiv.org/pdf/2607.21412", "published": "2026-07-23", "updated": "2026-07-23", "authors": [ "Bartolomeo Bogliolo" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "agent-safety", "rag", "reasoning", "tool-use" ], "score": 11, "relevance": "medium", "primary_query": "rag-agent", "matched_queries": [ "rag-agent" ] }, { "id": "2607.19450", "arxiv_id": "2607.19450", "source": "arxiv", "source_id": "arxiv:2607.19450", "title": "REGEN: Replay-recycling for Expert-to-Generalist distillation with Offline Reinforcement Learning", "url": "https://arxiv.org/abs/2607.19450", "pdf_url": "https://arxiv.org/pdf/2607.19450", "published": "2026-07-21", "updated": "2026-07-21", "authors": [ "Yunjie Chen", "Xiaoxin Chen", "Fang Wang" ], "categories": [ "cs.LG" ], "topics": [ "memory", "reasoning", "tool-use" ], "score": 11, "relevance": "medium", "primary_query": "tool-use", "matched_queries": [ "tool-use" ] }, { "id": "2607.17384", "arxiv_id": "2607.17384", "source": "arxiv", "source_id": "arxiv:2607.17384", "title": "Quantifying Diversity of Thought: A Predictive Law of Weighted LLM Ensemble Lift", "url": "https://arxiv.org/abs/2607.17384", "pdf_url": "https://arxiv.org/pdf/2607.17384", "published": "2026-07-21", "updated": "2026-07-21", "authors": [ "Junade Ali" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "agent-safety", "tool-use" ], "score": 11, "relevance": "medium", "primary_query": "tool-use", "matched_queries": [ "tool-use" ] }, { "id": "2607.18462", "arxiv_id": "2607.18462", "source": "arxiv", "source_id": "arxiv:2607.18462", "title": "Beyond Resolved Rate: A Non-Functional Quality Study", "url": "https://arxiv.org/abs/2607.18462", "pdf_url": "https://arxiv.org/pdf/2607.18462", "published": "2026-07-20", "updated": "2026-07-20", "authors": [ "Xin Sun", "Daniel Ståhl", "Kristian Sandahl", "Christoph Kessler" ], "categories": [ "cs.SE" ], "topics": [ "agent-evaluation", "coding-agent", "memory", "rag" ], "score": 11, "relevance": "medium", "primary_query": "coding-agent", "matched_queries": [ "coding-agent" ] }, { "id": "2607.17363", "arxiv_id": "2607.17363", "source": "arxiv", "source_id": "arxiv:2607.17363", "title": "ORB5X v1.0: a performance-portable global electromagnetic gyrokinetic PIC code in C++/Kokkos built using Agentic AI", "url": "https://arxiv.org/abs/2607.17363", "pdf_url": "https://arxiv.org/pdf/2607.17363", "published": "2026-07-19", "updated": "2026-07-19", "authors": [ "Mohsen Sadr", "Emmanuel Lanti", "Alexey Mishchenko", "Xin Wang", "Laurent Villard" ], "categories": [ "physics.plasm-ph" ], "topics": [ "memory", "workflow-agent" ], "score": 11, "relevance": "medium", "primary_query": "agentic-ai", "matched_queries": [ "agentic-ai" ] }, { "id": "2607.16680", "arxiv_id": "2607.16680", "source": "arxiv", "source_id": "arxiv:2607.16680", "title": "Specification-Driven Development as the Foundation of AI-Native Enterprise Software Engineering", "url": "https://arxiv.org/abs/2607.16680", "pdf_url": "https://arxiv.org/pdf/2607.16680", "published": "2026-07-18", "updated": "2026-07-18", "authors": [ "Mamdouh Alenezi" ], "categories": [ "cs.SE" ], "topics": [ "agent-evaluation", "agent-safety", "coding-agent", "tool-use", "workflow-agent" ], "score": 11, "relevance": "medium", "primary_query": "agentic-ai", "matched_queries": [ "agentic-ai" ] }, { "id": "2607.14582", "arxiv_id": "2607.14582", "source": "arxiv", "source_id": "arxiv:2607.14582", "title": "MathCoPilot: An Interactive System for Human-AI Symbiotic Paradigm of Mathematical Research", "url": "https://arxiv.org/abs/2607.14582", "pdf_url": "https://arxiv.org/pdf/2607.14582", "published": "2026-07-16", "updated": "2026-07-16", "authors": [ "Junjie Zhang", "Jiayu Liu", "Wenbin Liu", "Zhenya Huang", "Doudou Wang", "Yan Jiang", "Leiye Xu", "Tao Xiong", "Wen Huang", "Qi Liu", "Guoping Hu", "Enhong Chen", "Mengping Zhang", "Xiangdong Ye" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "computer-use", "rag" ], "score": 11, "relevance": "medium", "primary_query": "ai-agent", "matched_queries": [ "ai-agent", "autonomous-agent-llm" ] }, { "id": "2607.14547", "arxiv_id": "2607.14547", "source": "arxiv", "source_id": "arxiv:2607.14547", "title": "AdaTurn: Budget-Aware Test-Time Scaling for Active Visual Perception Agents", "url": "https://arxiv.org/abs/2607.14547", "pdf_url": "https://arxiv.org/pdf/2607.14547", "published": "2026-07-16", "updated": "2026-07-16", "authors": [ "Susan Liang", "Chao Huang", "Filippos Bellos", "Jing Bi", "Jason J Corso", "Chenliang Xu" ], "categories": [ "cs.CV" ], "topics": [ "agent-evaluation", "rag", "reasoning", "tool-use" ], "score": 11, "relevance": "medium", "primary_query": "tool-use", "matched_queries": [ "tool-use" ] }, { "id": "2607.14167", "arxiv_id": "2607.14167", "source": "arxiv", "source_id": "arxiv:2607.14167", "title": "Structured Feedback Improves Repair in an LLM Agent Loop", "url": "https://arxiv.org/abs/2607.14167", "pdf_url": "https://arxiv.org/pdf/2607.14167", "published": "2026-07-15", "updated": "2026-07-15", "authors": [ "Jaideep Ray", "Ankit Goyal" ], "categories": [ "cs.SE" ], "topics": [ "coding-agent" ], "score": 11, "relevance": "medium", "primary_query": "llm-agent", "matched_queries": [ "llm-agent" ] }, { "id": "2607.12575", "arxiv_id": "2607.12575", "source": "arxiv", "source_id": "arxiv:2607.12575", "title": "How Agentic Is Agentic Commerce? A Population-Scale Measurement of x402 Adoption and Authenticity", "url": "https://arxiv.org/abs/2607.12575", "pdf_url": "https://arxiv.org/pdf/2607.12575", "published": "2026-07-14", "updated": "2026-07-14", "authors": [ "Shengchen Ling", "Yajin Zhou", "Lei Wu", "Cong Wang" ], "categories": [ "cs.CR" ], "topics": [ "agent-evaluation", "tool-use" ], "score": 11, "relevance": "medium", "primary_query": "ai-agent", "matched_queries": [ "ai-agent" ] }, { "id": "2607.12875", "arxiv_id": "2607.12875", "source": "arxiv", "source_id": "arxiv:2607.12875", "title": "MetaInfer: A Knowledge Only LLM Inference Engine Generator SKILL Toolbox", "url": "https://arxiv.org/abs/2607.12875", "pdf_url": "https://arxiv.org/pdf/2607.12875", "published": "2026-07-14", "updated": "2026-07-14", "authors": [ "Zhenwen Miao", "Honglin Wang", "Mingheng Mi" ], "categories": [ "cs.MA" ], "topics": [ "agent-evaluation", "coding-agent", "multi-agent", "rag", "tool-use" ], "score": 11, "relevance": "medium", "primary_query": "multi-agent-llm", "matched_queries": [ "multi-agent-llm" ] }, { "id": "2607.13083", "arxiv_id": "2607.13083", "source": "arxiv", "source_id": "arxiv:2607.13083", "title": "Phantom Guardrails: When Self-Improving Agent Harnesses Fix Failures That Never Happened", "url": "https://arxiv.org/abs/2607.13083", "pdf_url": "https://arxiv.org/pdf/2607.13083", "published": "2026-07-13", "updated": "2026-07-13", "authors": [ "Su Wang", "Pin Qian", "Yifan Lin", "Jingzhou Xu", "Yihang Chen", "Xiaochong Jiang", "Lifei Liu", "Haoran Yu" ], "categories": [ "cs.CR" ], "topics": [ "agent-safety", "planning", "tool-use" ], "score": 11, "relevance": "medium", "primary_query": "ai-agent", "matched_queries": [ "ai-agent" ] }, { "id": "2607.07820", "arxiv_id": "2607.07820", "source": "arxiv", "source_id": "arxiv:2607.07820", "title": "DeepSearch-World: Self-Distillation for Deep Search Agents in a Verifiable Environment", "url": "https://arxiv.org/abs/2607.07820", "pdf_url": "https://arxiv.org/pdf/2607.07820", "published": "2026-07-13", "updated": "2026-07-13", "authors": [ "Xinyu Geng", "Xuanhua He", "Sixiang Chen", "Yanjing Xiao", "Fan Zhang", "Shijue Huang", "Haitao Mi", "Zhenwen Liang", "Tianqing Fang", "Yi R. Fung" ], "categories": [ "cs.CL" ], "topics": [ "computer-use", "planning", "reasoning", "tool-use" ], "score": 11, "relevance": "medium", "primary_query": "tool-use", "matched_queries": [ "tool-use", "web-gui-agent" ] }, { "id": "2607.20543", "arxiv_id": "2607.20543", "source": "arxiv", "source_id": "arxiv:2607.20543", "title": "When RLVR Shrinks the Reasoning Boundary: Diagnosing Pass@k Inversion", "url": "https://arxiv.org/abs/2607.20543", "pdf_url": "https://arxiv.org/pdf/2607.20543", "published": "2026-07-12", "updated": "2026-07-12", "authors": [ "Todd Zhou" ], "categories": [ "cs.LG" ], "topics": [ "agent-evaluation", "agent-safety", "computer-use", "rag", "reasoning", "tool-use" ], "score": 11, "relevance": "medium", "primary_query": "language-agent", "matched_queries": [ "language-agent" ] }, { "id": "2607.08691", "arxiv_id": "2607.08691", "source": "arxiv", "source_id": "arxiv:2607.08691", "title": "ProjAgent: Procedural Similarity Retrieval for Repository-Level Code Generation", "url": "https://arxiv.org/abs/2607.08691", "pdf_url": "https://arxiv.org/pdf/2607.08691", "published": "2026-07-09", "updated": "2026-07-09", "authors": [ "QiHong Chen", "Aaron Imani", "Iftekhar Ahmed" ], "categories": [ "cs.SE" ], "topics": [ "agent-evaluation", "coding-agent", "rag", "reasoning", "workflow-agent" ], "score": 11, "relevance": "medium", "primary_query": "agentic-ai", "matched_queries": [ "agentic-ai" ] }, { "id": "2607.22157", "arxiv_id": "2607.22157", "source": "arxiv", "source_id": "arxiv:2607.22157", "title": "Learning on the Job: Continual Learning from Deployment Feedback for Frozen-Weights Agents", "url": "https://arxiv.org/abs/2607.22157", "pdf_url": "https://arxiv.org/pdf/2607.22157", "published": "2026-07-24", "updated": "2026-07-24", "authors": [ "Valentin Tablan", "Scott Taylor", "Kristoffer Bernhem" ], "categories": [ "cs.AI" ], "topics": [ "memory", "rag" ], "score": 10, "relevance": "medium", "primary_query": "ai-agent", "matched_queries": [ "ai-agent", "rag-agent" ] }, { "id": "2607.22406", "arxiv_id": "2607.22406", "source": "arxiv", "source_id": "arxiv:2607.22406", "title": "Vibe Coding: An Experiment with Test-Driven Development", "url": "https://arxiv.org/abs/2607.22406", "pdf_url": "https://arxiv.org/pdf/2607.22406", "published": "2026-07-24", "updated": "2026-07-24", "authors": [ "Moritz Mock", "Barbara Russo" ], "categories": [ "cs.SE" ], "topics": [ "coding-agent", "multi-agent", "tool-use", "workflow-agent" ], "score": 10, "relevance": "medium", "primary_query": "agentic-ai", "matched_queries": [ "agentic-ai" ] }, { "id": "2607.17063", "arxiv_id": "2607.17063", "source": "arxiv", "source_id": "arxiv:2607.17063", "title": "When LLMs Over-Answer: Measuring and Mitigating Quality Issues in LLM-Based Hardware Description Language Question Answering", "url": "https://arxiv.org/abs/2607.17063", "pdf_url": "https://arxiv.org/pdf/2607.17063", "published": "2026-07-24", "updated": "2026-07-24", "authors": [ "Ziteng Hu", "Jiachi Chen", "Wenhao Lv", "Huan Zhang", "Yingjie Xia" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "multi-agent", "rag", "tool-use" ], "score": 10, "relevance": "medium", "primary_query": "multi-agent-llm", "matched_queries": [ "multi-agent-llm" ] }, { "id": "2607.21019", "arxiv_id": "2607.21019", "source": "arxiv", "source_id": "arxiv:2607.21019", "title": "HiMe: Real-Time Self-Hosted Personal Agent Platform for Health Insights with Wearable Devices", "url": "https://arxiv.org/abs/2607.21019", "pdf_url": "https://arxiv.org/pdf/2607.21019", "published": "2026-07-23", "updated": "2026-07-23", "authors": [ "Wei Liu", "Siya Qi", "Linhai Zhang", "Lorainne Tudor Car", "Yulan He" ], "categories": [ "cs.AI" ], "topics": [ "computer-use" ], "score": 10, "relevance": "medium", "primary_query": "llm-agent", "matched_queries": [ "llm-agent" ] }, { "id": "2607.21677", "arxiv_id": "2607.21677", "source": "arxiv", "source_id": "arxiv:2607.21677", "title": "Enhancing SLMs for Sustainable Code Optimization in Radio-Astronomy", "url": "https://arxiv.org/abs/2607.21677", "pdf_url": "https://arxiv.org/pdf/2607.21677", "published": "2026-07-23", "updated": "2026-07-23", "authors": [ "Elisa Chiarotto", "Jingbo Li", "P. Chris Broekema", "Rob V. van Nieuwpoort" ], "categories": [ "cs.SE" ], "topics": [ "agent-evaluation", "rag", "tool-use" ], "score": 10, "relevance": "medium", "primary_query": "agentic-ai", "matched_queries": [ "agentic-ai", "rag-agent" ] }, { "id": "2607.17890", "arxiv_id": "2607.17890", "source": "arxiv", "source_id": "arxiv:2607.17890", "title": "Stress Testing Concept Erasure with Large Language Model Agents", "url": "https://arxiv.org/abs/2607.17890", "pdf_url": "https://arxiv.org/pdf/2607.17890", "published": "2026-07-22", "updated": "2026-07-22", "authors": [ "Yuyang Xue", "Feng Chen", "Zhihua Liu", "Edward Moroshko", "Jingyu Sun", "Steven McDonagh", "Sotirios A. Tsaftaris" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "rag" ], "score": 10, "relevance": "medium", "primary_query": "llm-agent", "matched_queries": [ "llm-agent" ] }, { "id": "2607.20764", "arxiv_id": "2607.20764", "source": "arxiv", "source_id": "arxiv:2607.20764", "title": "ArbiGraph: Arbitrarily Scalable Verifiable Task Graphs for Evaluating Context Management", "url": "https://arxiv.org/abs/2607.20764", "pdf_url": "https://arxiv.org/pdf/2607.20764", "published": "2026-07-22", "updated": "2026-07-22", "authors": [ "Pavel Golikov", "Evgenii Opryshko", "Gennady Pekhimenko", "Mark C. Jeffrey" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "reasoning", "tool-use", "workflow-agent" ], "score": 10, "relevance": "medium", "primary_query": "language-agent", "matched_queries": [ "language-agent" ] }, { "id": "2607.18727", "arxiv_id": "2607.18727", "source": "arxiv", "source_id": "arxiv:2607.18727", "title": "Formal Verification of an Out-of-Order Multiprocessor against an In-Order Weak-Memory ISA", "url": "https://arxiv.org/abs/2607.18727", "pdf_url": "https://arxiv.org/pdf/2607.18727", "published": "2026-07-21", "updated": "2026-07-21", "authors": [ "Janggun Lee", "Jeehoon Kang" ], "categories": [ "cs.PL" ], "topics": [ "memory", "reasoning" ], "score": 10, "relevance": "medium", "primary_query": "llm-agent", "matched_queries": [ "llm-agent" ] }, { "id": "2607.19432", "arxiv_id": "2607.19432", "source": "arxiv", "source_id": "arxiv:2607.19432", "title": "ChainWatch: A Kill Chain-Aligned Sequential Detection Framework for Multi-Step Attacks in MCP-Based AI Agent Systems", "url": "https://arxiv.org/abs/2607.19432", "pdf_url": "https://arxiv.org/pdf/2607.19432", "published": "2026-07-20", "updated": "2026-07-20", "authors": [ "Om Narayan", "Rashmi Jyoti", "Ramkinker Singh" ], "categories": [ "cs.CR" ], "topics": [ "agent-safety", "tool-use" ], "score": 10, "relevance": "medium", "primary_query": "ai-agent", "matched_queries": [ "ai-agent" ] }, { "id": "2607.18506", "arxiv_id": "2607.18506", "source": "arxiv", "source_id": "arxiv:2607.18506", "title": "AI Value Alignment for Evolving Social Norms", "url": "https://arxiv.org/abs/2607.18506", "pdf_url": "https://arxiv.org/pdf/2607.18506", "published": "2026-07-20", "updated": "2026-07-20", "authors": [ "Nenad Tomašev", "Matija Franklin", "Simon Osindero" ], "categories": [ "cs.CY" ], "topics": [ "agent-evaluation", "agent-safety", "memory", "tool-use", "world-model" ], "score": 10, "relevance": "medium", "primary_query": "agent-evaluation", "matched_queries": [ "agent-evaluation" ] }, { "id": "2607.03651", "arxiv_id": "2607.03651", "source": "arxiv", "source_id": "arxiv:2607.03651", "title": "LLM-Guided Transportation Hub Capacity Planning with Textual Business Inputs", "url": "https://arxiv.org/abs/2607.03651", "pdf_url": "https://arxiv.org/pdf/2607.03651", "published": "2026-07-17", "updated": "2026-07-17", "authors": [ "Xiaoyue Liu", "Zheng Dong" ], "categories": [ "cs.LG" ], "topics": [ "agent-evaluation", "computer-use", "planning", "reasoning", "workflow-agent" ], "score": 10, "relevance": "medium", "primary_query": "planning-agent", "matched_queries": [ "planning-agent" ] }, { "id": "2607.12640", "arxiv_id": "2607.12640", "source": "arxiv", "source_id": "arxiv:2607.12640", "title": "A Learning-Rate-Gated Failure of GRPO in a Small Language and Vision-Language Model Web Agent: A Controlled Null and Its Mechanism", "url": "https://arxiv.org/abs/2607.12640", "pdf_url": "https://arxiv.org/pdf/2607.12640", "published": "2026-07-14", "updated": "2026-07-14", "authors": [ "Chengguang Gan", "Zhixi Cai", "Yunhao Liang", "Hanjun Wei", "Shiwen Ni", "Qinghao Zhang" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "computer-use" ], "score": 10, "relevance": "medium", "primary_query": "web-gui-agent", "matched_queries": [ "web-gui-agent" ] }, { "id": "2607.12723", "arxiv_id": "2607.12723", "source": "arxiv", "source_id": "arxiv:2607.12723", "title": "Bulkhead: Automated Semantic Detection and Remediation of Container Escape Vulnerabilities", "url": "https://arxiv.org/abs/2607.12723", "pdf_url": "https://arxiv.org/pdf/2607.12723", "published": "2026-07-14", "updated": "2026-07-14", "authors": [ "Qiyuan Fan", "Zhi Li", "Junjie Li", "XiaoFeng Wang", "Bin Yuan", "Deqing Zou" ], "categories": [ "cs.CR" ], "topics": [ "agent-safety", "computer-use", "multi-agent", "tool-use" ], "score": 10, "relevance": "medium", "primary_query": "multi-agent-llm", "matched_queries": [ "multi-agent-llm" ] }, { "id": "2607.12057", "arxiv_id": "2607.12057", "source": "arxiv", "source_id": "arxiv:2607.12057", "title": "Predicting Acceptance and Review Effort in Human and Agent Pull Requests", "url": "https://arxiv.org/abs/2607.12057", "pdf_url": "https://arxiv.org/pdf/2607.12057", "published": "2026-07-13", "updated": "2026-07-13", "authors": [ "Kartik Ghanshyambhai Pansuriya", "Ehsan Ghorbani", "Deepak Singh", "Eman Abdullah AlOmar" ], "categories": [ "cs.SE" ], "topics": [ "agent-evaluation", "coding-agent", "tool-use", "workflow-agent" ], "score": 10, "relevance": "medium", "primary_query": "coding-agent", "matched_queries": [ "coding-agent" ] }, { "id": "2607.10621", "arxiv_id": "2607.10621", "source": "arxiv", "source_id": "arxiv:2607.10621", "title": "WebDesignIter: Co-Evolving Design Knowledge for Repository-Level Front-End Code Generation", "url": "https://arxiv.org/abs/2607.10621", "pdf_url": "https://arxiv.org/pdf/2607.10621", "published": "2026-07-12", "updated": "2026-07-12", "authors": [ "Zheng Pei", "Mingwei Liu", "Zhenxi Chen", "Zihao Wang", "Yanlin Wang" ], "categories": [ "cs.SE" ], "topics": [ "agent-evaluation", "coding-agent", "planning", "rag" ], "score": 10, "relevance": "medium", "primary_query": "coding-agent", "matched_queries": [ "coding-agent" ] }, { "id": "2607.10508", "arxiv_id": "2607.10508", "source": "arxiv", "source_id": "arxiv:2607.10508", "title": "Confining Nondeterminism: AI-Driven Research Systems as DBMSs for Reliable, Non-Wasteful, Transparent, and Collaborative Research [Vision]", "url": "https://arxiv.org/abs/2607.10508", "pdf_url": "https://arxiv.org/pdf/2607.10508", "published": "2026-07-11", "updated": "2026-07-11", "authors": [ "Kyoungmin Kim", "Anastasia Ailamaki" ], "categories": [ "cs.DB" ], "topics": [ "planning", "tool-use" ], "score": 10, "relevance": "medium", "primary_query": "planning-agent", "matched_queries": [ "planning-agent", "tool-use" ] }, { "id": "2607.09502", "arxiv_id": "2607.09502", "source": "arxiv", "source_id": "arxiv:2607.09502", "title": "All Explanations are Wrong, But Many Are Useful: Exploring the Rashomon Explanation Set with Large Language Models", "url": "https://arxiv.org/abs/2607.09502", "pdf_url": "https://arxiv.org/pdf/2607.09502", "published": "2026-07-10", "updated": "2026-07-10", "authors": [ "Pan Li" ], "categories": [ "cs.LG" ], "topics": [ "computer-use", "planning", "reasoning", "workflow-agent" ], "score": 10, "relevance": "medium", "primary_query": "agentic-ai", "matched_queries": [ "agentic-ai" ] }, { "id": "2607.22015", "arxiv_id": "2607.22015", "source": "arxiv", "source_id": "arxiv:2607.22015", "title": "Are Production Cloud Skills Adequately Tested? Measuring and Governing Skill Test Coverage in Practice", "url": "https://arxiv.org/abs/2607.22015", "pdf_url": "https://arxiv.org/pdf/2607.22015", "published": "2026-07-24", "updated": "2026-07-24", "authors": [ "Haotian Si", "Junyi Chen", "Shuyang Yu", "Ruifeng Nie", "Jiate Li", "Jianqiang Zhao", "Meng Li", "Dengcheng He" ], "categories": [ "cs.SE" ], "topics": [ "agent-evaluation", "computer-use", "rag", "workflow-agent" ], "score": 9, "relevance": "medium", "primary_query": "ai-agent", "matched_queries": [ "ai-agent" ] }, { "id": "2607.21997", "arxiv_id": "2607.21997", "source": "arxiv", "source_id": "arxiv:2607.21997", "title": "\"Go Home Copilot, You're Drunk\": Understanding Developer Responses to Agent-Generated Code Review Comments", "url": "https://arxiv.org/abs/2607.21997", "pdf_url": "https://arxiv.org/pdf/2607.21997", "published": "2026-07-24", "updated": "2026-07-24", "authors": [ "Shamse Tasnim Cynthia", "Ratnadira Widyasari", "Banani Roy", "Ting Zhang", "David Lo" ], "categories": [ "cs.SE" ], "topics": [ "agent-evaluation", "coding-agent", "workflow-agent" ], "score": 9, "relevance": "medium", "primary_query": "coding-agent", "matched_queries": [ "coding-agent" ] }, { "id": "2607.21672", "arxiv_id": "2607.21672", "source": "arxiv", "source_id": "arxiv:2607.21672", "title": "Pixels for Programs? A Cross-Provider Case Study of Input-Token Accounting for Source Code as Text and Images", "url": "https://arxiv.org/abs/2607.21672", "pdf_url": "https://arxiv.org/pdf/2607.21672", "published": "2026-07-23", "updated": "2026-07-23", "authors": [ "Ronak Bhalgami" ], "categories": [ "cs.SE" ], "topics": [ "agent-evaluation", "coding-agent", "tool-use" ], "score": 9, "relevance": "medium", "primary_query": "coding-agent", "matched_queries": [ "coding-agent" ] }, { "id": "2607.20582", "arxiv_id": "2607.20582", "source": "arxiv", "source_id": "arxiv:2607.20582", "title": "Bayesian uncertainty estimation improves clinical decision making in medical AI agents", "url": "https://arxiv.org/abs/2607.20582", "pdf_url": "https://arxiv.org/pdf/2607.20582", "published": "2026-07-22", "updated": "2026-07-22", "authors": [ "Frederik Hauke", "Patrick Wienholt", "Christiane Kuhl", "Dyke Ferber", "Jakob Nikolas Kather", "Sven Nebelung", "Daniel Truhn" ], "categories": [ "cs.LG" ], "topics": [ "agent-safety" ], "score": 9, "relevance": "medium", "primary_query": "ai-agent", "matched_queries": [ "ai-agent" ] }, { "id": "2607.17015", "arxiv_id": "2607.17015", "source": "arxiv", "source_id": "arxiv:2607.17015", "title": "Alignment of a Total Automation Economy", "url": "https://arxiv.org/abs/2607.17015", "pdf_url": "https://arxiv.org/pdf/2607.17015", "published": "2026-07-21", "updated": "2026-07-21", "authors": [ "David McAllester" ], "categories": [ "physics.soc-ph" ], "topics": [ "agent-safety", "multi-agent", "planning", "workflow-agent" ], "score": 9, "relevance": "medium", "primary_query": "agentic-ai", "matched_queries": [ "agentic-ai" ] }, { "id": "2607.10144", "arxiv_id": "2607.10144", "source": "arxiv", "source_id": "arxiv:2607.10144", "title": "IdeaTrail: Full-Process Agent Trajectories for Scientific Ideation", "url": "https://arxiv.org/abs/2607.10144", "pdf_url": "https://arxiv.org/pdf/2607.10144", "published": "2026-07-20", "updated": "2026-07-20", "authors": [ "Hengquan Guo" ], "categories": [ "cs.AI" ], "topics": [ "rag", "reasoning", "tool-use" ], "score": 9, "relevance": "medium", "primary_query": "tool-use", "matched_queries": [ "tool-use" ] }, { "id": "2607.18161", "arxiv_id": "2607.18161", "source": "arxiv", "source_id": "arxiv:2607.18161", "title": "TRIM: Reducing AI-Generated CodeSlop via Agent Trajectory Minimization", "url": "https://arxiv.org/abs/2607.18161", "pdf_url": "https://arxiv.org/pdf/2607.18161", "published": "2026-07-20", "updated": "2026-07-20", "authors": [ "Alex Mathai", "Shobini Iyer", "Aleksandr Nogikh", "Petros Maniatis", "Franjo Ivancic", "Junfeng Yang", "Baishakhi Ray" ], "categories": [ "cs.SE" ], "topics": [ "coding-agent", "computer-use" ], "score": 9, "relevance": "medium", "primary_query": "coding-agent", "matched_queries": [ "coding-agent" ] }, { "id": "2607.18029", "arxiv_id": "2607.18029", "source": "arxiv", "source_id": "arxiv:2607.18029", "title": "Natural Language Access to Domain-Specific Metadata: A Reusable Framework for LLM Query Generation", "url": "https://arxiv.org/abs/2607.18029", "pdf_url": "https://arxiv.org/pdf/2607.18029", "published": "2026-07-20", "updated": "2026-07-20", "authors": [ "Blake G. Fitch", "Cato Elia Kurtz" ], "categories": [ "cs.DB" ], "topics": [ "agent-evaluation", "multi-agent", "rag" ], "score": 9, "relevance": "medium", "primary_query": "multi-agent-llm", "matched_queries": [ "multi-agent-llm" ] }, { "id": "2607.16387", "arxiv_id": "2607.16387", "source": "arxiv", "source_id": "arxiv:2607.16387", "title": "Fantastic Adaptive Taxonomies and How to Use Them", "url": "https://arxiv.org/abs/2607.16387", "pdf_url": "https://arxiv.org/pdf/2607.16387", "published": "2026-07-17", "updated": "2026-07-17", "authors": [ "Mert Cemri", "Andrei Cojocaru", "Melissa Pan", "Shu Liu", "Shubham Agarwal", "Alexander Krentsel", "Jay Tang", "Kannan Ramchandran", "Joseph E. Gonzalez", "Matei Zaharia", "Alex Dimakis", "Ion Stoica" ], "categories": [ "cs.SE" ], "topics": [ "agent-evaluation", "coding-agent", "reasoning", "workflow-agent" ], "score": 9, "relevance": "medium", "primary_query": "coding-agent", "matched_queries": [ "coding-agent" ] }, { "id": "2607.13220", "arxiv_id": "2607.13220", "source": "arxiv", "source_id": "arxiv:2607.13220", "title": "Networked Intelligence: Active Shared Context Graphs for Human-AI Team Science", "url": "https://arxiv.org/abs/2607.13220", "pdf_url": "https://arxiv.org/pdf/2607.13220", "published": "2026-07-16", "updated": "2026-07-16", "authors": [ "Sutanay Choudhury", "Jeffrey J. Czajka", "Lummy M. O. Monteiro", "Erin Bredeweg", "Jason McDermott", "Katherine Wolf", "Alex Beliaev", "Josh Elmore", "Paul Piehowski", "Kylee Tate", "Yuqian Gao", "Aivett Bilbao", "Kelly Stratton", "Scott Baker", "Jaydeep P. Bardhan", "Kristin Burnum Johnson", "Chris Oehmen", "Robert Rallo" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "embodied-agent", "planning", "reasoning" ], "score": 9, "relevance": "medium", "primary_query": "ai-agent", "matched_queries": [ "ai-agent" ] }, { "id": "2607.13679", "arxiv_id": "2607.13679", "source": "arxiv", "source_id": "arxiv:2607.13679", "title": "When Bots Join the Team: Bot Adoption and the Institutional Fabric of Open-Source Software Projects", "url": "https://arxiv.org/abs/2607.13679", "pdf_url": "https://arxiv.org/pdf/2607.13679", "published": "2026-07-15", "updated": "2026-07-15", "authors": [ "Yongren Shi", "Wenyi Gong" ], "categories": [ "cs.AI" ], "topics": [ "memory", "multi-agent", "planning", "tool-use" ], "score": 9, "relevance": "medium", "primary_query": "ai-agent", "matched_queries": [ "ai-agent" ] }, { "id": "2607.12161", "arxiv_id": "2607.12161", "source": "arxiv", "source_id": "arxiv:2607.12161", "title": "Token Reduction Is Not Cost Reduction", "url": "https://arxiv.org/abs/2607.12161", "pdf_url": "https://arxiv.org/pdf/2607.12161", "published": "2026-07-15", "updated": "2026-07-15", "authors": [ "Sarel Weinberger", "Amir Hozez" ], "categories": [ "cs.CL" ], "topics": [ "agent-evaluation", "coding-agent", "rag", "tool-use" ], "score": 9, "relevance": "medium", "primary_query": "coding-agent", "matched_queries": [ "coding-agent" ] }, { "id": "2607.14443", "arxiv_id": "2607.14443", "source": "arxiv", "source_id": "arxiv:2607.14443", "title": "Tactile: Giving Computer-Using Agents Hands and Feet", "url": "https://arxiv.org/abs/2607.14443", "pdf_url": "https://arxiv.org/pdf/2607.14443", "published": "2026-07-15", "updated": "2026-07-15", "authors": [ "Yong Liu", "Zhenyi Zhong", "Zhanpeng Shi" ], "categories": [ "cs.AI" ], "topics": [ "computer-use", "rag", "tool-use" ], "score": 9, "relevance": "medium", "primary_query": "web-gui-agent", "matched_queries": [ "web-gui-agent" ] }, { "id": "2607.11999", "arxiv_id": "2607.11999", "source": "arxiv", "source_id": "arxiv:2607.11999", "title": "Underwriting the Agent Economy: The Blueprint for an AI Insurance Stack", "url": "https://arxiv.org/abs/2607.11999", "pdf_url": "https://arxiv.org/pdf/2607.11999", "published": "2026-07-14", "updated": "2026-07-14", "authors": [ "Cristian Trout", "Sanmi Koyejo", "Sasha Romanosky", "Giorgio Ripamonti", "Lynn Thompson", "Desiree Spain", "Alex Taylor", "Kevin Casey", "Stephen Casper", "Matthew Botvinick", "Sean McGregor", "Miles Brundage", "A. Feder Cooper", "Patricia Paskov", "Adrien Ecoffet", "Ben Bucknall", "Kevin Wei", "Markus Anderljung", "Lukasz Szpruch", "Bri Treece", "Tom Zick", "Gabriel Weil", "Ugur Ozer", "Kevin Kalinich", "Jesus Gonzalez", "et al. (12 additional authors not shown)" ], "categories": [ "cs.CY" ], "topics": [ "agent-safety", "rag", "tool-use" ], "score": 9, "relevance": "medium", "primary_query": "ai-agent", "matched_queries": [ "ai-agent" ] }, { "id": "2607.09759", "arxiv_id": "2607.09759", "source": "arxiv", "source_id": "arxiv:2607.09759", "title": "ReflectWorld-MM: An Entity-Oriented Multimodal Memory System for Open-Ended Video Streams", "url": "https://arxiv.org/abs/2607.09759", "pdf_url": "https://arxiv.org/pdf/2607.09759", "published": "2026-07-14", "updated": "2026-07-14", "authors": [ "Xiaokang Ma", "Yifan Sun", "Zhihong Jin", "Jie Gu", "Yudong Luo", "Shenyi Shao", "Chu Tang", "Jingmin Chen", "Li Pu" ], "categories": [ "cs.CV" ], "topics": [ "agent-evaluation", "memory" ], "score": 9, "relevance": "medium", "primary_query": "agent-memory", "matched_queries": [ "agent-memory" ] }, { "id": "2607.12631", "arxiv_id": "2607.12631", "source": "arxiv", "source_id": "arxiv:2607.12631", "title": "Can Induced Emotion Bias LLM Behaviors in Sequential Decision Making?", "url": "https://arxiv.org/abs/2607.12631", "pdf_url": "https://arxiv.org/pdf/2607.12631", "published": "2026-07-14", "updated": "2026-07-14", "authors": [ "Minh Khoi Ho", "Zihao Zhu", "Runchuan Zhu", "Levina Li", "Zhiwen Fan", "Zhangyang Wang", "Junyuan Hong" ], "categories": [ "cs.CL" ], "topics": [ "computer-use", "rag", "tool-use" ], "score": 9, "relevance": "medium", "primary_query": "autonomous-agent-llm", "matched_queries": [ "autonomous-agent-llm" ] }, { "id": "2607.11377", "arxiv_id": "2607.11377", "source": "arxiv", "source_id": "arxiv:2607.11377", "title": "A Glimpse into Long-term Physical Coexistence with Intelligent Robots", "url": "https://arxiv.org/abs/2607.11377", "pdf_url": "https://arxiv.org/pdf/2607.11377", "published": "2026-07-13", "updated": "2026-07-13", "authors": [ "Weiqi Jin", "Peijun Tang", "Kuncheng Luo", "Baifu Huang", "Binyan Sun", "Haotian Yang", "Shangjin Xie", "Jianan Wang" ], "categories": [ "cs.RO" ], "topics": [ "embodied-agent", "memory", "planning", "reasoning", "tool-use" ], "score": 9, "relevance": "medium", "primary_query": "agent-memory", "matched_queries": [ "agent-memory" ] }, { "id": "2607.11111", "arxiv_id": "2607.11111", "source": "arxiv", "source_id": "arxiv:2607.11111", "title": "Know Before Fix: QA-Driven Repository Knowledge Acquisition for Software Issue Resolution", "url": "https://arxiv.org/abs/2607.11111", "pdf_url": "https://arxiv.org/pdf/2607.11111", "published": "2026-07-13", "updated": "2026-07-13", "authors": [ "Haotian Lin", "Silin Chen", "Xiaodong Gu", "Yuling Shi", "Chengxi Pan", "Jiaqi Ge", "Mengfan Li", "Jianghong Huang", "Mengchieh Chuang", "Beijun Shen", "Haibing Guan" ], "categories": [ "cs.SE" ], "topics": [ "coding-agent", "rag" ], "score": 9, "relevance": "medium", "primary_query": "coding-agent", "matched_queries": [ "coding-agent" ] }, { "id": "2607.19967", "arxiv_id": "2607.19967", "source": "arxiv", "source_id": "arxiv:2607.19967", "title": "When Shippers Become Algorithms: Candidate Exposure, Information Design, and the Concentration of LLM-Mediated Freight Markets", "url": "https://arxiv.org/abs/2607.19967", "pdf_url": "https://arxiv.org/pdf/2607.19967", "published": "2026-07-22", "updated": "2026-07-22", "authors": [ "Takahiro Ezaki", "Naoto Imura", "Katsuhiro Nishinari" ], "categories": [ "physics.soc-ph" ], "topics": [ "agent-safety", "tool-use", "world-model" ], "score": 8, "relevance": "medium", "primary_query": "llm-agent", "matched_queries": [ "llm-agent" ] }, { "id": "2607.17598", "arxiv_id": "2607.17598", "source": "arxiv", "source_id": "arxiv:2607.17598", "title": "Is Progressive Disclosure All You Need for Long-Context Agents?", "url": "https://arxiv.org/abs/2607.17598", "pdf_url": "https://arxiv.org/pdf/2607.17598", "published": "2026-07-20", "updated": "2026-07-20", "authors": [ "Yifeng He", "Yinzhe Zhao", "Jicheng Wang", "Hao Chen" ], "categories": [ "cs.AI" ], "topics": [ "embodied-agent", "tool-use" ], "score": 8, "relevance": "medium", "primary_query": "agentic-ai", "matched_queries": [ "agentic-ai" ] }, { "id": "2607.17686", "arxiv_id": "2607.17686", "source": "arxiv", "source_id": "arxiv:2607.17686", "title": "Integrating High-Level Requirements to Low-Level Tests with Machine-Readable V&V Specifications", "url": "https://arxiv.org/abs/2607.17686", "pdf_url": "https://arxiv.org/pdf/2607.17686", "published": "2026-07-20", "updated": "2026-07-20", "authors": [ "Mansur Arief", "Nur Ahmad Khatim", "Ali Akarma", "Ahmad Alfan Alfian Irfan" ], "categories": [ "cs.SE" ], "topics": [ "agent-evaluation", "coding-agent", "tool-use" ], "score": 8, "relevance": "medium", "primary_query": "coding-agent", "matched_queries": [ "coding-agent" ] }, { "id": "2607.15970", "arxiv_id": "2607.15970", "source": "arxiv", "source_id": "arxiv:2607.15970", "title": "Code-Poisoning Property Inference Attacks", "url": "https://arxiv.org/abs/2607.15970", "pdf_url": "https://arxiv.org/pdf/2607.15970", "published": "2026-07-17", "updated": "2026-07-17", "authors": [ "Xukun Luan", "Yuhui Gong", "Gang Zhang", "Zixuan Huang", "Yuanguo Bi", "Xuesong Li", "Jinyan Liu" ], "categories": [ "cs.CR" ], "topics": [ "agent-evaluation", "agent-safety", "coding-agent", "tool-use" ], "score": 8, "relevance": "medium", "primary_query": "coding-agent", "matched_queries": [ "coding-agent" ] }, { "id": "2607.13190", "arxiv_id": "2607.13190", "source": "arxiv", "source_id": "arxiv:2607.13190", "title": "Reliable isomorphic physics problem generation with large language models", "url": "https://arxiv.org/abs/2607.13190", "pdf_url": "https://arxiv.org/pdf/2607.13190", "published": "2026-07-14", "updated": "2026-07-14", "authors": [ "Xian Wu", "Lindim Ismaili" ], "categories": [ "physics.ed-ph" ], "topics": [ "agent-evaluation", "workflow-agent" ], "score": 8, "relevance": "medium", "primary_query": "agentic-ai", "matched_queries": [ "agentic-ai" ] }, { "id": "2607.11560", "arxiv_id": "2607.11560", "source": "arxiv", "source_id": "arxiv:2607.11560", "title": "Technical Report on the CVPR 2026@AdvML Workshop Challenge", "url": "https://arxiv.org/abs/2607.11560", "pdf_url": "https://arxiv.org/pdf/2607.11560", "published": "2026-07-13", "updated": "2026-07-13", "authors": [ "Tianyuan Zhang", "Zonglei Jing", "Jiangfan Liu", "Ligong Zhang", "Ke Ma", "Chengzhi Sun", "Xiaohai Xu", "Zhirui Zhang", "Qianqian Xu", "Qingming Huang", "Hanyu Fang", "Junhua Liu", "Zheng Wang", "Xiaoliang Liu", "Yuanbo Li", "Shuai Gui", "Bin Wang", "Menghe Zheng", "Jing Nie", "Hanyang Meng", "Zeyang Zhang", "Xiang Zhang", "Yongxuan Zhu", "Rui Ding", "Hainan Li", "et al. (25 additional authors not shown)" ], "categories": [ "cs.CV" ], "topics": [ "agent-evaluation", "agent-safety", "reasoning" ], "score": 8, "relevance": "medium", "primary_query": "language-agent", "matched_queries": [ "language-agent" ] }, { "id": "2607.11362", "arxiv_id": "2607.11362", "source": "arxiv", "source_id": "arxiv:2607.11362", "title": "Boolean queries are all you need?", "url": "https://arxiv.org/abs/2607.11362", "pdf_url": "https://arxiv.org/pdf/2607.11362", "published": "2026-07-13", "updated": "2026-07-13", "authors": [ "Charles L. A. Clarke", "Mark D. Smucker" ], "categories": [ "cs.IR" ], "topics": [ "agent-evaluation", "rag" ], "score": 8, "relevance": "medium", "primary_query": "rag-agent", "matched_queries": [ "rag-agent" ] }, { "id": "2607.10198", "arxiv_id": "2607.10198", "source": "arxiv", "source_id": "arxiv:2607.10198", "title": "Equal Accuracy, Unequal Evidence: Search APIs as Decision Surfaces for Tool-Using Agents", "url": "https://arxiv.org/abs/2607.10198", "pdf_url": "https://arxiv.org/pdf/2607.10198", "published": "2026-07-11", "updated": "2026-07-11", "authors": [ "Sriram Selvam", "Anneswa Ghosh" ], "categories": [ "cs.CL" ], "topics": [ "agent-evaluation", "rag", "tool-use" ], "score": 8, "relevance": "medium", "primary_query": "tool-use", "matched_queries": [ "tool-use" ] }, { "id": "2607.08147", "arxiv_id": "2607.08147", "source": "arxiv", "source_id": "arxiv:2607.08147", "title": "Prismata: Confining Cross-Site Prompt Injection in Web Agents", "url": "https://arxiv.org/abs/2607.08147", "pdf_url": "https://arxiv.org/pdf/2607.08147", "published": "2026-07-09", "updated": "2026-07-09", "authors": [ "Corban Villa", "Alp Eren Ozdarendeli", "Sijun Tan", "Raluca Ada Popa" ], "categories": [ "cs.CR" ], "topics": [ "agent-safety", "computer-use", "reasoning" ], "score": 8, "relevance": "medium", "primary_query": "web-gui-agent", "matched_queries": [ "web-gui-agent" ] }, { "id": "2607.17286", "arxiv_id": "2607.17286", "source": "arxiv", "source_id": "arxiv:2607.17286", "title": "IssueExec: A Test-Driven Approach for Localizing Software Engineering Issues", "url": "https://arxiv.org/abs/2607.17286", "pdf_url": "https://arxiv.org/pdf/2607.17286", "published": "2026-07-19", "updated": "2026-07-19", "authors": [ "Jiawei Liu", "Yun Lin", "Chenyan Liu", "Yu Qian", "Yiming Liu", "Jiaxin Chang", "Weinan Zhang", "Linpeng Huang" ], "categories": [ "cs.SE" ], "topics": [ "coding-agent", "rag", "tool-use" ], "score": 7, "relevance": "medium", "primary_query": "coding-agent", "matched_queries": [ "coding-agent" ] }, { "id": "2607.16062", "arxiv_id": "2607.16062", "source": "arxiv", "source_id": "arxiv:2607.16062", "title": "When Model Merging Rivals Joint Multi-Task Reinforcement Learning: A Task-Vector Geometry Analysis", "url": "https://arxiv.org/abs/2607.16062", "pdf_url": "https://arxiv.org/pdf/2607.16062", "published": "2026-07-17", "updated": "2026-07-17", "authors": [ "S. Aaron McClendon" ], "categories": [ "cs.LG" ], "topics": [ "agent-evaluation", "computer-use", "rag" ], "score": 7, "relevance": "medium", "primary_query": "agent-evaluation", "matched_queries": [ "agent-evaluation" ] }, { "id": "2607.15053", "arxiv_id": "2607.15053", "source": "arxiv", "source_id": "arxiv:2607.15053", "title": "ANet Patu-1: The Value of Connection in the Agent Network", "url": "https://arxiv.org/abs/2607.15053", "pdf_url": "https://arxiv.org/pdf/2607.15053", "published": "2026-07-16", "updated": "2026-07-16", "authors": [ "Mu Yuan", "Jinke Song", "Zhaomeng Zhou", "Lan Zhang" ], "categories": [ "cs.NI" ], "topics": [ "multi-agent" ], "score": 7, "relevance": "medium", "primary_query": "ai-agent", "matched_queries": [ "ai-agent" ] }, { "id": "2607.14141", "arxiv_id": "2607.14141", "source": "arxiv", "source_id": "arxiv:2607.14141", "title": "Human AI Construction of Bayesian Networks for Operational Decision Support -- A Virtual Survey Approach", "url": "https://arxiv.org/abs/2607.14141", "pdf_url": "https://arxiv.org/pdf/2607.14141", "published": "2026-07-13", "updated": "2026-07-13", "authors": [ "Kumar Rahul", "Shovan Chowdhury" ], "categories": [ "cs.AI" ], "topics": [ "tool-use" ], "score": 7, "relevance": "medium", "primary_query": "ai-agent", "matched_queries": [ "ai-agent" ] }, { "id": "2607.16544", "arxiv_id": "2607.16544", "source": "arxiv", "source_id": "arxiv:2607.16544", "title": "AIMS: An uncertainty-aware AI experimentalist for quantum matter", "url": "https://arxiv.org/abs/2607.16544", "pdf_url": "https://arxiv.org/pdf/2607.16544", "published": "2026-07-17", "updated": "2026-07-17", "authors": [ "Siyuan Qiu", "Philip D. Suh", "Nhat Huy Tran", "Xirui Wang", "Heonjoon Park", "Kutay Akin", "Kevin K. S. Multani", "Seungwon Jung", "Wenkai Cai", "Xinyu Liu", "Ziyan Zhu", "Chunjing Jia", "Zhantao Chen", "Zhixun Shen", "Zhurun Ji" ], "categories": [ "cond-mat.str-el" ], "topics": [ "embodied-agent", "tool-use" ], "score": 6, "relevance": "low", "primary_query": "ai-agent", "matched_queries": [ "ai-agent" ] }, { "id": "2607.13087", "arxiv_id": "2607.13087", "source": "arxiv", "source_id": "arxiv:2607.13087", "title": "GDM AI Control Roadmap", "url": "https://arxiv.org/abs/2607.13087", "pdf_url": "https://arxiv.org/pdf/2607.13087", "published": "2026-07-13", "updated": "2026-07-13", "authors": [ "Mary Phuong", "Erik Jenner", "Laurent Simon", "Lewis Ho", "Rohin Shah", "Sebastian Farquhar", "Scott Coull" ], "categories": [ "cs.CR" ], "topics": [ "agent-safety", "tool-use" ], "score": 6, "relevance": "low", "primary_query": "ai-agent", "matched_queries": [ "ai-agent" ] }, { "id": "2607.20775", "arxiv_id": "2607.20775", "source": "arxiv", "source_id": "arxiv:2607.20775", "title": "Flint: A Semantics-Driven Data Visualization Intermediate Language", "url": "https://arxiv.org/abs/2607.20775", "pdf_url": "https://arxiv.org/pdf/2607.20775", "published": "2026-07-22", "updated": "2026-07-22", "authors": [ "Yunhai Wang", "Kecheng Lu", "Junhao Chen", "Alper Sarikaya", "Chenglong Wang" ], "categories": [ "cs.HC" ], "topics": [ "agent" ], "score": 4, "relevance": "low", "primary_query": "ai-agent", "matched_queries": [ "ai-agent" ] }, { "id": "2607.21659", "arxiv_id": "2607.21659", "source": "arxiv", "source_id": "arxiv:2607.21659", "title": "Defining AI-Native Systems: Autonomy as Revision Authority", "url": "https://arxiv.org/abs/2607.21659", "pdf_url": "https://arxiv.org/pdf/2607.21659", "published": "2026-07-22", "updated": "2026-07-22", "authors": [ "Cheng Tan" ], "categories": [ "cs.AI" ], "topics": [ "computer-use" ], "score": 4, "relevance": "low", "primary_query": "coding-agent", "matched_queries": [ "coding-agent" ] }, { "id": "2607.19618", "arxiv_id": "2607.19618", "source": "arxiv", "source_id": "arxiv:2607.19618", "title": "Causal dictionary learning reveals and validates transcription-factor binding features in genomic language models", "url": "https://arxiv.org/abs/2607.19618", "pdf_url": "https://arxiv.org/pdf/2607.19618", "published": "2026-07-21", "updated": "2026-07-21", "authors": [ "Sarwan Ali" ], "categories": [ "q-bio.GN" ], "topics": [ "coding-agent" ], "score": 4, "relevance": "low", "primary_query": "web-gui-agent", "matched_queries": [ "web-gui-agent" ] }, { "id": "2607.21965", "arxiv_id": "2607.21965", "source": "arxiv", "source_id": "arxiv:2607.21965", "title": "Inertial Asynchronous Computation", "url": "https://arxiv.org/abs/2607.21965", "pdf_url": "https://arxiv.org/pdf/2607.21965", "published": "2026-07-24", "updated": "2026-07-24", "authors": [ "Doruk Efe Gökmen", "Michel Fruchart", "Dmitrii Zendrikov", "Giacomo Indiveri", "Giulio Biroli", "Vincenzo Vitelli" ], "categories": [ "cond-mat.stat-mech" ], "topics": [ "agent-evaluation" ], "score": 2, "relevance": "low", "primary_query": "web-gui-agent", "matched_queries": [ "web-gui-agent" ] }, { "id": "2607.16950", "arxiv_id": "2607.16950", "source": "arxiv", "source_id": "arxiv:2607.16950", "title": "Optimal Scheduling for Remote State Estimation over Hybrid Channels", "url": "https://arxiv.org/abs/2607.16950", "pdf_url": "https://arxiv.org/pdf/2607.16950", "published": "2026-07-18", "updated": "2026-07-18", "authors": [ "Manali Dutta", "Rahul Singh", "Shalabh Bhatnagar" ], "categories": [ "math.OC" ], "topics": [ "rag" ], "score": 2, "relevance": "low", "primary_query": "web-gui-agent", "matched_queries": [ "web-gui-agent" ] }, { "id": "2607.15808", "arxiv_id": "2607.15808", "source": "arxiv", "source_id": "arxiv:2607.15808", "title": "Examining the Associations between Visual and Non-Visual Elements and Cyclists' Route Choices for Various Trip Purposes", "url": "https://arxiv.org/abs/2607.15808", "pdf_url": "https://arxiv.org/pdf/2607.15808", "published": "2026-07-17", "updated": "2026-07-17", "authors": [ "Heyang Hua", "Koichi Ito", "Filip Biljecki" ], "categories": [ "cs.CV" ], "topics": [ "planning" ], "score": 2, "relevance": "low", "primary_query": "web-gui-agent", "matched_queries": [ "web-gui-agent" ] }, { "id": "2607.15729", "arxiv_id": "2607.15729", "source": "arxiv", "source_id": "arxiv:2607.15729", "title": "Stable X-ray reverberation lags in the black hole X-ray binary Swift J1727.8-1613", "url": "https://arxiv.org/abs/2607.15729", "pdf_url": "https://arxiv.org/pdf/2607.15729", "published": "2026-07-17", "updated": "2026-07-17", "authors": [ "Wei Yu", "Sinan Allak", "Zi-Xu Yang", "Xiao Fan", "Andrea Santangelo", "Tian-hao Xie" ], "categories": [ "astro-ph.HE" ], "topics": [ "tool-use" ], "score": 2, "relevance": "low", "primary_query": "web-gui-agent", "matched_queries": [ "web-gui-agent" ] }, { "id": "2607.14985", "arxiv_id": "2607.14985", "source": "arxiv", "source_id": "arxiv:2607.14985", "title": "Isomer-specific excitation of formic acid in collisions with helium atoms", "url": "https://arxiv.org/abs/2607.14985", "pdf_url": "https://arxiv.org/pdf/2607.14985", "published": "2026-07-16", "updated": "2026-07-16", "authors": [ "Karina Sogomonyan", "Anzhela Veselinova-Marinova", "François Lique", "Jérôme Loreau" ], "categories": [ "astro-ph.GA" ], "topics": [ "tool-use" ], "score": 2, "relevance": "low", "primary_query": "web-gui-agent", "matched_queries": [ "web-gui-agent" ] }, { "id": "2607.17885", "arxiv_id": "2607.17885", "source": "arxiv", "source_id": "arxiv:2607.17885", "title": "Hybrid-Dimensional Biot Problem with an Optimization Based Domain Decomposition Approach", "url": "https://arxiv.org/abs/2607.17885", "pdf_url": "https://arxiv.org/pdf/2607.17885", "published": "2026-07-20", "updated": "2026-07-20", "authors": [ "Francesca Marcon", "Stefano Scialò" ], "categories": [ "math.NA" ], "topics": [], "score": 1, "relevance": "low", "primary_query": "web-gui-agent", "matched_queries": [ "web-gui-agent" ] }, { "id": "2607.09293", "arxiv_id": "2607.09293", "source": "arxiv", "source_id": "arxiv:2607.09293", "title": "Fifth-Order Well-Balanced Path-Conservative A-WENO Scheme for the Ripa Model", "url": "https://arxiv.org/abs/2607.09293", "pdf_url": "https://arxiv.org/pdf/2607.09293", "published": "2026-07-10", "updated": "2026-07-10", "authors": [ "Yan-Ping Qiu", "Zhen Gao", "Alexander Kurganov", "Bao-Shan Wang", "Xiao Wen" ], "categories": [ "math.NA" ], "topics": [], "score": 1, "relevance": "low", "primary_query": "web-gui-agent", "matched_queries": [ "web-gui-agent" ] } ]