[ { "id": "2606.08340", "title": "Benchmarking Open-Ended Multi-Agent Coordination in Language Agents", "url": "https://arxiv.org/abs/2606.08340", "published": "2026-06-06", "updated": "2026-06-06", "authors": [ "Kale-ab Abebe Tessera", "Andras Szecsenyi", "Cameron Barker", "Alexander Rutherford", "Davide Paglieri", "Aidan Scannell", "Henry Gouk", "Elliot J. Crowley", "Tim Rocktäschel", "Amos Storkey" ], "categories": [ "cs.AI", "cs.LG", "cs.MA" ], "topics": [ "agent-evaluation", "memory", "multi-agent", "planning", "rag", "reasoning", "tool-use" ], "score": 28, "relevance": "high", "matched_queries": [ "autonomous-agent-llm", "language-agent", "planning-agent" ], "arxiv_id": "2606.08340", "source": "arxiv", "source_id": "arxiv:2606.08340", "pdf_url": "https://arxiv.org/pdf/2606.08340", "primary_query": "autonomous-agent-llm" }, { "id": "2605.20833", "title": "MemGym: a Long-Horizon Memory Environment for LLM Agents", "url": "https://arxiv.org/abs/2605.20833", "published": "2026-05-20", "updated": "2026-05-20", "authors": [ "Wujiang Xu", "Yu Wang", "Kai Mei", "Kaiqu Liang", "Zhenting Wang", "Mingyu Jin", "Han Zhang", "Shi-Xiong Zhang", "Wenyue Hua", "Sambit Sahu", "Dimitris N. Metaxas" ], "categories": [ "cs.CL" ], "topics": [ "agent-evaluation", "coding-agent", "computer-use", "embodied-agent", "memory", "planning", "rag", "reasoning", "tool-use" ], "score": 26, "relevance": "high", "matched_queries": [ "agent-memory" ], "arxiv_id": "2605.20833", "source": "arxiv", "source_id": "arxiv:2605.20833", "pdf_url": "https://arxiv.org/pdf/2605.20833", "primary_query": "agent-memory" }, { "id": "2607.06008", "title": "PolyWorkBench: Benchmarking Multilingual Long-Horizon LLM Agents", "url": "https://arxiv.org/abs/2607.06008", "published": "2026-07-07", "updated": "2026-07-07", "authors": [ "Hongliang Li", "Yijin Liu", "Zhiwei Zhang", "Zihe Liu", "Xinyue Lou", "Jinan Xu", "Fandong Meng", "Kaiyu Huang" ], "categories": [ "cs.AI", "cs.CL" ], "topics": [ "agent-evaluation", "computer-use", "planning", "reasoning", "tool-use", "workflow-agent" ], "score": 25, "relevance": "high", "matched_queries": [ "agent-evaluation", "llm-agent", "planning-agent", "tool-use" ], "arxiv_id": "2607.06008", "source": "arxiv", "source_id": "arxiv:2607.06008", "pdf_url": "https://arxiv.org/pdf/2607.06008", "primary_query": "agent-evaluation" }, { "id": "2606.28425", "title": "Tool Use Enables Undetectable Steganography in Multi-Agent LLM Systems", "url": "https://arxiv.org/abs/2606.28425", "published": "2026-06-25", "updated": "2026-06-25", "authors": [ "Jimmy Laurence Rippin", "Simon C. Marshall", "David Demitri Africa", "Christian Schroeder de Witt" ], "categories": [ "cs.CR", "cs.AI" ], "topics": [ "agent-evaluation", "agent-safety", "coding-agent", "computer-use", "multi-agent", "rag", "tool-use" ], "score": 25, "relevance": "high", "matched_queries": [ "agentic-ai", "ai-agent", "autonomous-agent-llm", "multi-agent-llm", "tool-use" ], "arxiv_id": "2606.28425", "source": "arxiv", "source_id": "arxiv:2606.28425", "pdf_url": "https://arxiv.org/pdf/2606.28425", "primary_query": "agentic-ai" }, { "id": "2606.24937", "title": "The Hitchhiker's Guide to Agentic AI: From Foundations to Systems", "url": "https://arxiv.org/abs/2606.24937", "published": "2026-06-22", "updated": "2026-06-22", "authors": [ "Haggai Roitman" ], "categories": [ "cs.AI", "cs.CL", "cs.IR", "cs.LG" ], "topics": [ "agent-evaluation", "agent-safety", "computer-use", "memory", "multi-agent", "rag", "reasoning", "tool-use" ], "score": 25, "relevance": "high", "matched_queries": [ "agentic-ai", "multi-agent-llm", "rag-agent", "tool-use" ], "arxiv_id": "2606.24937", "source": "arxiv", "source_id": "arxiv:2606.24937", "pdf_url": "https://arxiv.org/pdf/2606.24937", "primary_query": "agentic-ai" }, { "id": "2606.10749", "title": "Toward Secure LLM Agents: Threat Surfaces, Attacks, Defenses, and Evaluation", "url": "https://arxiv.org/abs/2606.10749", "published": "2026-06-09", "updated": "2026-06-09", "authors": [ "Yuchen Ling", "Shengcheng Yu", "Zhenyu Chen", "Chunrong Fang" ], "categories": [ "cs.CR", "cs.AI" ], "topics": [ "agent-evaluation", "agent-safety", "memory", "multi-agent", "planning", "rag", "tool-use" ], "score": 25, "relevance": "high", "matched_queries": [ "agent-safety", "planning-agent" ], "arxiv_id": "2606.10749", "source": "arxiv", "source_id": "arxiv:2606.10749", "pdf_url": "https://arxiv.org/pdf/2606.10749", "primary_query": "agent-safety" }, { "id": "2606.29824", "title": "Neural Procedural Memory: Empowering LLM Agents with Implicit Activation Steering", "url": "https://arxiv.org/abs/2606.29824", "published": "2026-06-29", "updated": "2026-06-29", "authors": [ "Chengfeng Zhao", "Yuqiao Tan", "Shizhu He", "Yequan Wang", "Jun Zhao", "Kang Liu" ], "categories": [ "cs.CL", "cs.AI" ], "topics": [ "agent-evaluation", "computer-use", "memory", "rag", "tool-use", "workflow-agent" ], "score": 24, "relevance": "high", "matched_queries": [ "agent-evaluation", "agent-memory", "autonomous-agent-llm", "llm-agent", "rag-agent" ], "arxiv_id": "2606.29824", "source": "arxiv", "source_id": "arxiv:2606.29824", "pdf_url": "https://arxiv.org/pdf/2606.29824", "primary_query": "agent-evaluation" }, { "id": "2606.28011", "title": "From Detection to Action: Using LLM Agents for Fault-Tolerant Control", "url": "https://arxiv.org/abs/2606.28011", "published": "2026-06-26", "updated": "2026-06-26", "authors": [ "Javal Vyas", "Milapji Singh Gill", "Artan Markaj", "Felix Gehlhoff", "Mehmet Mercangöz" ], "categories": [ "eess.SY", "cs.LG" ], "topics": [ "agent-evaluation", "agent-safety", "multi-agent", "planning", "rag", "tool-use", "workflow-agent", "world-model" ], "score": 24, "relevance": "high", "matched_queries": [ "agentic-ai", "llm-agent", "multi-agent-llm", "planning-agent", "rag-agent" ], "arxiv_id": "2606.28011", "source": "arxiv", "source_id": "arxiv:2606.28011", "pdf_url": "https://arxiv.org/pdf/2606.28011", "primary_query": "agentic-ai" }, { "id": "2606.20401", "title": "PowerAgentBench-Dyn: A Benchmark for Agentic AI in Power System Dynamic Studies", "url": "https://arxiv.org/abs/2606.20401", "published": "2026-06-18", "updated": "2026-06-18", "authors": [ "Qian Zhang", "Andrea Pomarico", "Costas Mylonas", "Magda Foti", "Alberto Berizzi", "Le Xie" ], "categories": [ "eess.SY" ], "topics": [ "agent-evaluation", "agent-safety", "coding-agent", "memory", "planning", "rag", "reasoning", "tool-use", "world-model" ], "score": 24, "relevance": "high", "matched_queries": [ "agentic-ai" ], "arxiv_id": "2606.20401", "source": "arxiv", "source_id": "arxiv:2606.20401", "pdf_url": "https://arxiv.org/pdf/2606.20401", "primary_query": "agentic-ai" }, { "id": "2606.18789", "title": "PowerAgentBench-SS: A Benchmark for Agentic AI in Power System Steady-State Studies", "url": "https://arxiv.org/abs/2606.18789", "published": "2026-06-17", "updated": "2026-06-17", "authors": [ "Costas Mylonas", "Magda Foti", "Andrea Pomarico", "Matheus Duarte", "Qian Zhang", "Emmanouel Varvarigos" ], "categories": [ "eess.SY" ], "topics": [ "agent-evaluation", "agent-safety", "computer-use", "planning", "rag", "tool-use", "workflow-agent" ], "score": 24, "relevance": "high", "matched_queries": [ "agentic-ai", "planning-agent", "tool-use" ], "arxiv_id": "2606.18789", "source": "arxiv", "source_id": "arxiv:2606.18789", "pdf_url": "https://arxiv.org/pdf/2606.18789", "primary_query": "agentic-ai" }, { "id": "2606.08274", "title": "Toward Human-Centered Multi-Agent Systems: Integrating Cognition, Culture, Values, and Cooperation in AI Agents", "url": "https://arxiv.org/abs/2606.08274", "published": "2026-06-06", "updated": "2026-06-06", "authors": [ "Safia Baloch", "Rahemeen Khan" ], "categories": [ "cs.MA" ], "topics": [ "agent-evaluation", "agent-safety", "computer-use", "multi-agent", "planning", "tool-use", "workflow-agent" ], "score": 24, "relevance": "high", "matched_queries": [ "autonomous-agent-llm", "planning-agent" ], "arxiv_id": "2606.08274", "source": "arxiv", "source_id": "arxiv:2606.08274", "pdf_url": "https://arxiv.org/pdf/2606.08274", "primary_query": "autonomous-agent-llm" }, { "id": "2607.05775", "title": "Beyond the Leaderboard: A Synthesis of Tool-Use, Planning, and Reasoning Failures in Large Language Model Agents", "url": "https://arxiv.org/abs/2607.05775", "published": "2026-07-07", "updated": "2026-07-07", "authors": [ "Wael Albayaydh", "Rui Zhao", "Ivan Flechais" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "agent-safety", "coding-agent", "embodied-agent", "multi-agent", "planning", "reasoning", "tool-use" ], "score": 23, "relevance": "high", "matched_queries": [ "llm-agent", "multi-agent-llm", "planning-agent", "tool-use" ], "arxiv_id": "2607.05775", "source": "arxiv", "source_id": "arxiv:2607.05775", "pdf_url": "https://arxiv.org/pdf/2607.05775", "primary_query": "llm-agent" }, { "id": "2607.02255", "title": "AgenticSTS: A Bounded-Memory Testbed for Long-Horizon LLM Agents", "url": "https://arxiv.org/abs/2607.02255", "published": "2026-07-02", "updated": "2026-07-02", "authors": [ "Xiangchen Cheng", "Yunwei Jiang", "Jianwen Sun", "Zizhen Li", "Chuanhao Li", "Xiangcheng Cao", "Yihao Liu", "Fanrui Zhang", "Li Jin", "Kaipeng Zhang" ], "categories": [ "cs.AI", "cs.CL" ], "topics": [ "agent-evaluation", "memory", "planning", "rag", "reasoning", "tool-use" ], "score": 23, "relevance": "high", "matched_queries": [ "llm-agent" ], "arxiv_id": "2607.02255", "source": "arxiv", "source_id": "arxiv:2607.02255", "pdf_url": "https://arxiv.org/pdf/2607.02255", "primary_query": "llm-agent" }, { "id": "2606.28791", "title": "From Determinism to Delegation: AI-Native Software Engineering and the Evolution of the Agentic Engineer", "url": "https://arxiv.org/abs/2606.28791", "published": "2026-06-27", "updated": "2026-06-27", "authors": [ "Mamdouh Alenezi" ], "categories": [ "cs.SE" ], "topics": [ "agent-evaluation", "agent-safety", "coding-agent", "memory", "multi-agent", "reasoning", "tool-use", "workflow-agent" ], "score": 23, "relevance": "high", "matched_queries": [ "agentic-ai", "autonomous-agent-llm", "tool-use" ], "arxiv_id": "2606.28791", "source": "arxiv", "source_id": "arxiv:2606.28791", "pdf_url": "https://arxiv.org/pdf/2606.28791", "primary_query": "agentic-ai" }, { "id": "2606.26614", "title": "HiLSVA: Design and Evaluation of a Human-in-the-Loop Agentic System for Scientific Visualization", "url": "https://arxiv.org/abs/2606.26614", "published": "2026-06-25", "updated": "2026-06-25", "authors": [ "Kuangshi Ai", "Patrick Phuoc Do", "Chaoli Wang" ], "categories": [ "cs.HC", "cs.AI", "cs.GR" ], "topics": [ "agent-evaluation", "computer-use", "multi-agent", "planning", "rag", "reasoning", "tool-use", "workflow-agent" ], "score": 23, "relevance": "high", "matched_queries": [ "multi-agent-llm", "planning-agent" ], "arxiv_id": "2606.26614", "source": "arxiv", "source_id": "arxiv:2606.26614", "pdf_url": "https://arxiv.org/pdf/2606.26614", "primary_query": "multi-agent-llm" }, { "id": "2605.08442", "title": "Defense effectiveness across architectural layers: a mechanistic evaluation of persistent memory attacks on stateful LLM agents", "url": "https://arxiv.org/abs/2605.08442", "published": "2026-05-08", "updated": "2026-07-03", "authors": [ "Jun Wen Leong" ], "categories": [ "cs.CR", "cs.AI", "cs.LG" ], "topics": [ "agent-evaluation", "agent-safety", "computer-use", "memory", "rag", "reasoning", "tool-use" ], "score": 23, "relevance": "high", "matched_queries": [ "agent-safety" ], "arxiv_id": "2605.08442", "source": "arxiv", "source_id": "arxiv:2605.08442", "pdf_url": "https://arxiv.org/pdf/2605.08442", "primary_query": "agent-safety" }, { "id": "2605.06869", "title": "Agentick: A Unified Benchmark for General Sequential Decision-Making Agents", "url": "https://arxiv.org/abs/2605.06869", "published": "2026-05-07", "updated": "2026-05-12", "authors": [ "Roger Creus Castanyer", "Pablo Samuel Castro", "Glen Berseth" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "coding-agent", "multi-agent", "planning", "rag", "reasoning", "tool-use" ], "score": 23, "relevance": "high", "matched_queries": [ "autonomous-agent-llm" ], "arxiv_id": "2605.06869", "source": "arxiv", "source_id": "arxiv:2605.06869", "pdf_url": "https://arxiv.org/pdf/2605.06869", "primary_query": "autonomous-agent-llm" }, { "id": "2607.03233", "title": "Agentic and Generative AI for Open-Source Intelligence and Cyber Investigations: Taxonomy, Evaluation, Challenges, and Future Directions", "url": "https://arxiv.org/abs/2607.03233", "published": "2026-07-03", "updated": "2026-07-03", "authors": [ "Eduardo Almeida Palmieri", "Mohamed Chahine Ghanem", "Dipo Dunsin", "Zubair Baig", "Ed de Quincey", "Kim-Kwang Raymond Choo" ], "categories": [ "cs.CR", "cs.AI", "cs.IR", "cs.SI" ], "topics": [ "agent-evaluation", "agent-safety", "rag", "reasoning", "tool-use" ], "score": 22, "relevance": "high", "matched_queries": [ "agentic-ai", "rag-agent", "tool-use" ], "arxiv_id": "2607.03233", "source": "arxiv", "source_id": "arxiv:2607.03233", "pdf_url": "https://arxiv.org/pdf/2607.03233", "primary_query": "agentic-ai" }, { "id": "2606.28061", "title": "ToolPrivacyBench: Benchmarking Purpose-Bound Privacy in Tool-Using LLM Agents", "url": "https://arxiv.org/abs/2606.28061", "published": "2026-06-26", "updated": "2026-06-26", "authors": [ "Shijing Hu", "Liang Liu", "Zhu Meng", "Zhicheng Zhao" ], "categories": [ "cs.CR", "cs.AI" ], "topics": [ "agent-evaluation", "rag", "tool-use", "workflow-agent" ], "score": 22, "relevance": "high", "matched_queries": [ "agent-evaluation", "function-calling", "llm-agent", "tool-use" ], "arxiv_id": "2606.28061", "source": "arxiv", "source_id": "arxiv:2606.28061", "pdf_url": "https://arxiv.org/pdf/2606.28061", "primary_query": "agent-evaluation" }, { "id": "2606.17459", "title": "Can LLMs Be CEOs? Benchmarking Strategic Resource Reallocation with Multi-Role Agent Simulation", "url": "https://arxiv.org/abs/2606.17459", "published": "2026-06-16", "updated": "2026-06-16", "authors": [ "Yuyang Dai", "Xueqing Peng", "Lingfei Qian", "Zhuohan Xie" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "computer-use", "multi-agent", "planning", "rag", "reasoning", "tool-use", "world-model" ], "score": 22, "relevance": "high", "matched_queries": [ "agent-evaluation", "planning-agent" ], "arxiv_id": "2606.17459", "source": "arxiv", "source_id": "arxiv:2606.17459", "pdf_url": "https://arxiv.org/pdf/2606.17459", "primary_query": "agent-evaluation" }, { "id": "2606.16613", "title": "CoffeeBench: Benchmarking Long-Horizon LLM Agents in Heterogeneous Multi-Agent Economies", "url": "https://arxiv.org/abs/2606.16613", "published": "2026-06-15", "updated": "2026-06-15", "authors": [ "Issa Sugiura", "Daichi Hattori", "Kazuo Araragi", "Keita Ogawa", "Shota Onose", "Taro Makino", "Teppei Usuki", "Takashi Ishida" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "multi-agent", "planning", "tool-use", "world-model" ], "score": 22, "relevance": "high", "matched_queries": [ "autonomous-agent-llm", "planning-agent" ], "arxiv_id": "2606.16613", "source": "arxiv", "source_id": "arxiv:2606.16613", "pdf_url": "https://arxiv.org/pdf/2606.16613", "primary_query": "autonomous-agent-llm" }, { "id": "2606.12945", "title": "Learning What to Remember: A Cognitively Grounded Multi-Factor Value Model for Agentic Memory", "url": "https://arxiv.org/abs/2606.12945", "published": "2026-06-11", "updated": "2026-06-20", "authors": [ "Zhibao Chen", "Qian Cheng" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "agent-safety", "coding-agent", "memory", "planning", "rag", "tool-use" ], "score": 22, "relevance": "high", "matched_queries": [ "agent-memory" ], "arxiv_id": "2606.12945", "source": "arxiv", "source_id": "arxiv:2606.12945", "pdf_url": "https://arxiv.org/pdf/2606.12945", "primary_query": "agent-memory" }, { "id": "2606.06399", "title": "CollabSim: A CSCW-Grounded Methodology for Investigating Collaborative Competence of LLM Agents through Controlled Multi-Agent Experiments", "url": "https://arxiv.org/abs/2606.06399", "published": "2026-06-04", "updated": "2026-06-06", "authors": [ "Jiaju Chen", "Bo Sun", "Yuxuan Lu", "Yun Wang", "Dakuo Wang", "Bingsheng Yao" ], "categories": [ "cs.CL" ], "topics": [ "agent-evaluation", "agent-safety", "multi-agent", "planning", "reasoning", "tool-use", "world-model" ], "score": 22, "relevance": "high", "matched_queries": [ "planning-agent" ], "arxiv_id": "2606.06399", "source": "arxiv", "source_id": "arxiv:2606.06399", "pdf_url": "https://arxiv.org/pdf/2606.06399", "primary_query": "planning-agent" }, { "id": "2606.01199", "title": "Can LLM Agents Sustain Long-Horizon Organizational Dynamics?", "url": "https://arxiv.org/abs/2606.01199", "published": "2026-05-31", "updated": "2026-05-31", "authors": [ "Xuancheng Zhu", "Yang Yue", "Shuaibing Wan", "Zihan Dou", "Xiaohan Zhang", "Yongrui Liu", "Guoshun Nan" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "memory", "multi-agent", "planning", "rag", "workflow-agent", "world-model" ], "score": 22, "relevance": "high", "matched_queries": [ "language-agent", "planning-agent" ], "arxiv_id": "2606.01199", "source": "arxiv", "source_id": "arxiv:2606.01199", "pdf_url": "https://arxiv.org/pdf/2606.01199", "primary_query": "language-agent" }, { "id": "2605.29861", "title": "Towards Verifiable Multimodal Deep Research: A Multi-Agent Harness for Interleaved Report Generation", "url": "https://arxiv.org/abs/2605.29861", "published": "2026-05-28", "updated": "2026-06-03", "authors": [ "Chenghao Zhang", "Guanting Dong", "Yufan Liu", "Tong Zhao", "Xiaoxi Li", "Zhicheng Dou" ], "categories": [ "cs.CL", "cs.AI" ], "topics": [ "agent-evaluation", "memory", "multi-agent", "planning", "rag", "reasoning", "tool-use", "workflow-agent" ], "score": 22, "relevance": "high", "matched_queries": [ "autonomous-agent-llm" ], "arxiv_id": "2605.29861", "source": "arxiv", "source_id": "arxiv:2605.29861", "pdf_url": "https://arxiv.org/pdf/2605.29861", "primary_query": "autonomous-agent-llm" }, { "id": "2605.18652", "title": "MementoGUI: Learning Agentic Multimodal Memory Control for Long-Horizon GUI Agents", "url": "https://arxiv.org/abs/2605.18652", "published": "2026-05-18", "updated": "2026-05-18", "authors": [ "Ziyun Zeng", "Hang Hua", "Bocheng Zou", "Mu Cai", "Rogerio Feris", "Jiebo Luo" ], "categories": [ "cs.CV" ], "topics": [ "agent-evaluation", "computer-use", "memory", "planning", "rag", "tool-use" ], "score": 22, "relevance": "high", "matched_queries": [ "agent-memory" ], "arxiv_id": "2605.18652", "source": "arxiv", "source_id": "arxiv:2605.18652", "pdf_url": "https://arxiv.org/pdf/2605.18652", "primary_query": "agent-memory" }, { "id": "2605.05704", "title": "SafeHarbor: Hierarchical Memory-Augmented Guardrail for LLM Agent Safety", "url": "https://arxiv.org/abs/2605.05704", "published": "2026-05-07", "updated": "2026-05-22", "authors": [ "Zhe Liu", "Zonghao Ying", "Wenxin Zhang", "Quanchen Zou", "Deyue Zhang", "Dongdong Yang", "Xiangzheng Zhang", "Hao Peng" ], "categories": [ "cs.CR", "cs.AI" ], "topics": [ "agent-safety", "computer-use", "memory", "reasoning", "tool-use" ], "score": 22, "relevance": "high", "matched_queries": [ "agent-safety", "autonomous-agent-llm" ], "arxiv_id": "2605.05704", "source": "arxiv", "source_id": "arxiv:2605.05704", "pdf_url": "https://arxiv.org/pdf/2605.05704", "primary_query": "agent-safety" }, { "id": "2605.03242", "title": "Enhancing Agent Safety Judgment: Controlled Benchmark Rewriting and Analogical Reasoning for Deceptive Out-of-Distribution Scenarios", "url": "https://arxiv.org/abs/2605.03242", "published": "2026-05-05", "updated": "2026-05-05", "authors": [ "Zuoyu Zhang", "Yancheng Zhu" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "agent-safety", "computer-use", "multi-agent", "rag", "reasoning", "tool-use" ], "score": 22, "relevance": "high", "matched_queries": [ "agent-safety" ], "arxiv_id": "2605.03242", "source": "arxiv", "source_id": "arxiv:2605.03242", "pdf_url": "https://arxiv.org/pdf/2605.03242", "primary_query": "agent-safety" }, { "id": "2607.06118", "title": "WebRetriever: A Large-Scale Comprehensive Benchmark for Efficient Web Agent Evaluation", "url": "https://arxiv.org/abs/2607.06118", "published": "2026-07-07", "updated": "2026-07-07", "authors": [ "Wei Dong", "Tianyu Fu", "Zhe Yu", "Hanning Wang", "Anyang Su", "Zhizhou Fang", "Yuyang Chen", "Shuo Wang", "Minghui Wu", "Ping Jiang", "Zhen Lei", "Chenxu Zhao" ], "categories": [ "cs.CV", "cs.MM" ], "topics": [ "agent-evaluation", "agent-safety", "computer-use", "embodied-agent", "rag", "tool-use", "workflow-agent" ], "score": 21, "relevance": "high", "matched_queries": [ "agent-evaluation", "web-gui-agent" ], "arxiv_id": "2607.06118", "source": "arxiv", "source_id": "arxiv:2607.06118", "pdf_url": "https://arxiv.org/pdf/2607.06118", "primary_query": "agent-evaluation" }, { "id": "2607.05773", "title": "Beyond Static Evaluation: Building Simulation Environments for Scalable Agentic Reinforcement Learning", "url": "https://arxiv.org/abs/2607.05773", "published": "2026-07-07", "updated": "2026-07-07", "authors": [ "Akshay Arora", "Ishan Nigam", "Ashutosh Aggarwal", "Shefali Bansal", "Krishna Singh", "Sweta Kumari", "Nikhil Mittal", "Shariq Farhan", "Siddarth Malreddy" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "computer-use", "tool-use", "world-model" ], "score": 21, "relevance": "high", "matched_queries": [ "autonomous-agent-llm", "tool-use", "web-gui-agent" ], "arxiv_id": "2607.05773", "source": "arxiv", "source_id": "arxiv:2607.05773", "pdf_url": "https://arxiv.org/pdf/2607.05773", "primary_query": "autonomous-agent-llm" }, { "id": "2607.05456", "title": "Prompt-to-Paper: Agentic AI System for Bioinformatics", "url": "https://arxiv.org/abs/2607.05456", "published": "2026-07-05", "updated": "2026-07-05", "authors": [ "Ramsha Kamran", "Maheera Amjad", "Zartasha Mustansar", "Arsalan Shaukat", "Salma Sherbaz", "Muhammad U. S. Khan" ], "categories": [ "cs.AI", "cs.CL", "q-bio.QM" ], "topics": [ "agent-evaluation", "coding-agent", "multi-agent", "rag", "tool-use" ], "score": 21, "relevance": "high", "matched_queries": [ "agentic-ai", "coding-agent", "multi-agent-llm" ], "arxiv_id": "2607.05456", "source": "arxiv", "source_id": "arxiv:2607.05456", "pdf_url": "https://arxiv.org/pdf/2607.05456", "primary_query": "agentic-ai" }, { "id": "2607.04433", "title": "Autonomous Information Seeking: A Roadmap for Agentic Recommender Systems", "url": "https://arxiv.org/abs/2607.04433", "published": "2026-07-05", "updated": "2026-07-05", "authors": [ "Xinyu Lin", "Yashar Deldjoo", "Sunhao Dai", "Honghui Bao", "Xiaopeng Ye", "Fatemeh Nazary", "Wenjie Wang", "Tommaso Di Noia", "Jun Xu", "Tat-Seng Chua" ], "categories": [ "cs.IR", "cs.CL" ], "topics": [ "agent-evaluation", "agent-safety", "memory", "planning", "reasoning", "tool-use", "workflow-agent", "world-model" ], "score": 21, "relevance": "high", "matched_queries": [ "tool-use" ], "arxiv_id": "2607.04433", "source": "arxiv", "source_id": "arxiv:2607.04433", "pdf_url": "https://arxiv.org/pdf/2607.04433", "primary_query": "tool-use" }, { "id": "2607.03601", "title": "ArchEval: Measuring AI Agents as Computer Architects", "url": "https://arxiv.org/abs/2607.03601", "published": "2026-07-03", "updated": "2026-07-03", "authors": [ "Chenyu Wang", "Zishen Wan", "Jeffrey Ma", "Shvetank Prakash", "Zhenting Qi", "Haebin Do", "Andy Cheng", "Arya Tschand", "Jiahe Shi", "Yilun Du", "Vijay Janapa Reddi" ], "categories": [ "cs.AR" ], "topics": [ "agent-evaluation", "memory", "reasoning", "tool-use", "workflow-agent" ], "score": 21, "relevance": "high", "matched_queries": [ "ai-agent", "llm-agent", "tool-use" ], "arxiv_id": "2607.03601", "source": "arxiv", "source_id": "arxiv:2607.03601", "pdf_url": "https://arxiv.org/pdf/2607.03601", "primary_query": "ai-agent" }, { "id": "2607.02032", "title": "PACE: A Proxy for Agentic Capability Evaluation", "url": "https://arxiv.org/abs/2607.02032", "published": "2026-07-02", "updated": "2026-07-06", "authors": [ "Yueqi Song", "Lintang Sutawika", "Jiarui Liu", "Lindia Tjuatja", "Jiayi Geng", "Yunze Xiao", "Daniel Lee", "Aditya Bharat Soni", "Vincent Lo", "Xiang Yue", "Graham Neubig" ], "categories": [ "cs.AI", "cs.CL" ], "topics": [ "agent-evaluation", "coding-agent", "reasoning" ], "score": 21, "relevance": "high", "matched_queries": [ "agent-evaluation", "coding-agent", "llm-agent" ], "arxiv_id": "2607.02032", "source": "arxiv", "source_id": "arxiv:2607.02032", "pdf_url": "https://arxiv.org/pdf/2607.02032", "primary_query": "agent-evaluation" }, { "id": "2607.01641", "title": "When Agents Do Not Stop: Uncovering Infinite Agentic Loops in LLM Agents", "url": "https://arxiv.org/abs/2607.01641", "published": "2026-07-02", "updated": "2026-07-02", "authors": [ "Xinyi Hou", "Shenao Wang", "Yanjie Zhao", "Haoyu Wang" ], "categories": [ "cs.SE" ], "topics": [ "agent-evaluation", "multi-agent", "planning", "tool-use", "workflow-agent" ], "score": 21, "relevance": "high", "matched_queries": [ "llm-agent", "planning-agent", "tool-use" ], "arxiv_id": "2607.01641", "source": "arxiv", "source_id": "arxiv:2607.01641", "pdf_url": "https://arxiv.org/pdf/2607.01641", "primary_query": "llm-agent" }, { "id": "2606.30524", "title": "The Illusion of Agentic Complexity in README.md Generation: Evaluating Single-Agent vs. Multi-Agent RAG Systems", "url": "https://arxiv.org/abs/2606.30524", "published": "2026-06-29", "updated": "2026-07-01", "authors": [ "Abu Saleh", "Tesfay Welegebreal Tesfay", "Phuong T. Nguyen", "Juri Di Rocco", "Muhammad Umar Zeshan", "Davide Di Ruscio" ], "categories": [ "cs.SE" ], "topics": [ "agent-evaluation", "coding-agent", "computer-use", "multi-agent", "planning", "rag" ], "score": 21, "relevance": "high", "matched_queries": [ "multi-agent-llm", "rag-agent" ], "arxiv_id": "2606.30524", "source": "arxiv", "source_id": "arxiv:2606.30524", "pdf_url": "https://arxiv.org/pdf/2606.30524", "primary_query": "multi-agent-llm" }, { "id": "2606.29537", "title": "OSWorld2.0: Benchmarking Computer Use Agents on Long-Horizon Real-World Tasks", "url": "https://arxiv.org/abs/2606.29537", "published": "2026-06-28", "updated": "2026-06-28", "authors": [ "Mengqi Yuan", "Zilong Zhou", "Xinzhuang Xiong", "Weiming Wu", "Jiayang Sun", "Jiamin Song", "Kaiqian Cui", "Bowen Wang", "Haoyuan Wu", "Yitong Li", "Dunjie Lu", "Haikong Lu", "Qi Zhen", "Xinyuan Wang", "Jiaqi Deng", "Yuhao Yang", "Cheng Chen", "Boyuan Zheng", "Alex Su", "Xiao Yu", "Hao Zou", "Saaket Agashe", "Xing Han Lu", "Manpreet Kaur", "Zhengyang Qi", "Vincent Sunn Chen", "Frederic Sala", "Dayiheng Liu", "Junyang Lin", "Zhou Yu", "Yu Su", "Siva Reddy", "Xin Eric Wang", "Peng Qi", "Tianbao Xie", "Tao Yu" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "agent-safety", "coding-agent", "computer-use", "memory", "planning", "rag", "reasoning", "tool-use", "workflow-agent" ], "score": 21, "relevance": "high", "matched_queries": [ "web-gui-agent" ], "arxiv_id": "2606.29537", "source": "arxiv", "source_id": "arxiv:2606.29537", "pdf_url": "https://arxiv.org/pdf/2606.29537", "primary_query": "web-gui-agent" }, { "id": "2606.28925", "title": "Multi-Agent Routing as Set-Valued Prediction: A WildChat Benchmark and Cost-Aware Evaluation", "url": "https://arxiv.org/abs/2606.28925", "published": "2026-06-27", "updated": "2026-06-27", "authors": [ "Ananto Nayan Bala", "Faisal Muhammad Shah" ], "categories": [ "cs.LG", "cs.AI", "cs.IR", "cs.MA" ], "topics": [ "agent-evaluation", "multi-agent", "rag", "tool-use", "world-model" ], "score": 21, "relevance": "high", "matched_queries": [ "multi-agent-llm" ], "arxiv_id": "2606.28925", "source": "arxiv", "source_id": "arxiv:2606.28925", "pdf_url": "https://arxiv.org/pdf/2606.28925", "primary_query": "multi-agent-llm" }, { "id": "2606.26511", "title": "Temporal Validity in Retrieval Memory: Eliminating Stale-Fact Errors for AI Agents over Evolving Knowledge", "url": "https://arxiv.org/abs/2606.26511", "published": "2026-06-25", "updated": "2026-06-25", "authors": [ "Neeraj Yadav" ], "categories": [ "cs.CL", "cs.AI", "cs.ET", "cs.LG" ], "topics": [ "agent-evaluation", "computer-use", "memory", "rag", "tool-use" ], "score": 21, "relevance": "high", "matched_queries": [ "ai-agent", "rag-agent" ], "arxiv_id": "2606.26511", "source": "arxiv", "source_id": "arxiv:2606.26511", "pdf_url": "https://arxiv.org/pdf/2606.26511", "primary_query": "ai-agent" }, { "id": "2606.22557", "title": "MacAgentBench: Benchmarking AI Agents on Real-World macOS Desktop", "url": "https://arxiv.org/abs/2606.22557", "published": "2026-06-21", "updated": "2026-06-21", "authors": [ "Yikun Fu", "Bowen Fu", "Zhenyu Wu", "Shuang Cheng", "Xiaowei Sun", "Bowen Yang", "Zehao Li", "Yibo Zhao", "Zichen Ding", "Zhoumianze Liu", "Shijie Wang", "Biqing Qi", "Bowen Zhou" ], "categories": [ "cs.AI", "cs.CL", "cs.HC" ], "topics": [ "agent-evaluation", "computer-use", "planning", "rag", "tool-use", "workflow-agent" ], "score": 21, "relevance": "high", "matched_queries": [ "agent-evaluation", "ai-agent", "web-gui-agent" ], "arxiv_id": "2606.22557", "source": "arxiv", "source_id": "arxiv:2606.22557", "pdf_url": "https://arxiv.org/pdf/2606.22557", "primary_query": "agent-evaluation" }, { "id": "2606.21877", "title": "AgentRiskBOM: A Risk-Scoping Security Bill of Materials for Agentic AI Systems", "url": "https://arxiv.org/abs/2606.21877", "published": "2026-06-20", "updated": "2026-06-20", "authors": [ "Srimonti Dutta", "Akshata Kishore Moharir" ], "categories": [ "cs.AI", "cs.CR", "cs.SE" ], "topics": [ "agent-evaluation", "agent-safety", "coding-agent", "memory", "multi-agent", "rag", "tool-use" ], "score": 21, "relevance": "high", "matched_queries": [ "agentic-ai", "ai-agent", "rag-agent", "tool-use" ], "arxiv_id": "2606.21877", "source": "arxiv", "source_id": "arxiv:2606.21877", "pdf_url": "https://arxiv.org/pdf/2606.21877", "primary_query": "agentic-ai" }, { "id": "2606.16802", "title": "LabOSBench: Benchmarking Computer Use Agents for Scientific Instrument Control", "url": "https://arxiv.org/abs/2606.16802", "published": "2026-06-15", "updated": "2026-06-15", "authors": [ "Anqi Zou", "Han Deng", "Chengyu Zhang", "Junquan Hu", "Yu Wang", "Yuxiang Xing", "Aokai Zhang", "Hanling Zhang", "Zhaoyang Liu", "Ben Fei", "Zhihui Wang", "Wanli Ouyang" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "agent-safety", "computer-use", "planning", "workflow-agent" ], "score": 21, "relevance": "high", "matched_queries": [ "web-gui-agent" ], "arxiv_id": "2606.16802", "source": "arxiv", "source_id": "arxiv:2606.16802", "pdf_url": "https://arxiv.org/pdf/2606.16802", "primary_query": "web-gui-agent" }, { "id": "2606.13994", "title": "Hidden in Plain Sight: Benchmarking Agent Safety Against Decomposition Attacks with DECOMPBENCH", "url": "https://arxiv.org/abs/2606.13994", "published": "2026-06-12", "updated": "2026-06-12", "authors": [ "Vikhyath Kothamasu", "Virginia Smith", "Chhavi Yadav" ], "categories": [ "cs.CR", "cs.AI", "cs.LG" ], "topics": [ "agent-evaluation", "agent-safety", "planning", "tool-use", "workflow-agent" ], "score": 21, "relevance": "high", "matched_queries": [ "agent-safety", "tool-use" ], "arxiv_id": "2606.13994", "source": "arxiv", "source_id": "arxiv:2606.13994", "pdf_url": "https://arxiv.org/pdf/2606.13994", "primary_query": "agent-safety" }, { "id": "2606.04990", "title": "From Agent Traces to Trust: A Survey of Evidence Tracing and Execution Provenance in LLM Agents", "url": "https://arxiv.org/abs/2606.04990", "published": "2026-06-03", "updated": "2026-06-28", "authors": [ "Yiqi Wang", "Jiaqi Zhang", "Taotao Cai", "Zirui Liu", "Qingqiang Sun", "Zequn Sun", "Zhangkai Wu", "Manqing Dong", "Mingkai Zheng", "Xuefei Yin", "Yanming Zhu" ], "categories": [ "cs.CR", "cs.AI" ], "topics": [ "agent-evaluation", "agent-safety", "memory", "multi-agent", "planning", "rag", "tool-use" ], "score": 21, "relevance": "high", "matched_queries": [ "planning-agent" ], "arxiv_id": "2606.04990", "source": "arxiv", "source_id": "arxiv:2606.04990", "pdf_url": "https://arxiv.org/pdf/2606.04990", "primary_query": "planning-agent" }, { "id": "2606.02461", "title": "AgentCL: Toward Rigorous Evaluation of Continual Learning in Language Agents", "url": "https://arxiv.org/abs/2606.02461", "published": "2026-06-01", "updated": "2026-06-02", "authors": [ "Yiheng Shu", "Bernal Jiménez Gutiérrez", "Saisri Padmaja Jonnalagedda", "Yuguang Yao", "Huan Sun", "Yu Su" ], "categories": [ "cs.AI", "cs.CL" ], "topics": [ "agent-evaluation", "coding-agent", "computer-use", "memory", "rag", "reasoning", "tool-use", "workflow-agent" ], "score": 21, "relevance": "high", "matched_queries": [ "language-agent" ], "arxiv_id": "2606.02461", "source": "arxiv", "source_id": "arxiv:2606.02461", "pdf_url": "https://arxiv.org/pdf/2606.02461", "primary_query": "language-agent" }, { "id": "2606.01385", "title": "Bridging Requirements and Architecture: Multi-Agent Orchestration with External Knowledge and Hierarchical Memory", "url": "https://arxiv.org/abs/2606.01385", "published": "2026-05-31", "updated": "2026-05-31", "authors": [ "Ruiyin Li", "Yiran Zhang", "Xiyu Zhou", "Yangxiao Cai", "Peng Liang", "Weisong Sun", "Jifeng Xuan", "Zhi Jin", "Yang Liu" ], "categories": [ "cs.SE", "cs.AI" ], "topics": [ "agent-evaluation", "coding-agent", "memory", "multi-agent", "rag", "reasoning", "workflow-agent" ], "score": 21, "relevance": "high", "matched_queries": [ "rag-agent" ], "arxiv_id": "2606.01385", "source": "arxiv", "source_id": "arxiv:2606.01385", "pdf_url": "https://arxiv.org/pdf/2606.01385", "primary_query": "rag-agent" }, { "id": "2606.00756", "title": "CoMIC: Collaborative Memory and Insights Circulation for Long-Horizon LLM Agents in Cloud-Edge Systems", "url": "https://arxiv.org/abs/2606.00756", "published": "2026-05-30", "updated": "2026-05-30", "authors": [ "Yannan Wang", "Longli Yang", "Zhen Liu", "Abhishek Kumar", "Carsten Maple" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "computer-use", "memory", "planning", "rag", "reasoning", "tool-use" ], "score": 21, "relevance": "high", "matched_queries": [ "planning-agent" ], "arxiv_id": "2606.00756", "source": "arxiv", "source_id": "arxiv:2606.00756", "pdf_url": "https://arxiv.org/pdf/2606.00756", "primary_query": "planning-agent" }, { "id": "2605.20315", "title": "Mix-Quant: Quantized Prefilling, Precise Decoding for Agentic LLMs", "url": "https://arxiv.org/abs/2605.20315", "published": "2026-05-19", "updated": "2026-05-19", "authors": [ "Haiquan Lu", "Zigeng Chen", "Gongfan Fang", "Xinyin Ma", "Xinchao Wang" ], "categories": [ "cs.CL" ], "topics": [ "agent-evaluation", "coding-agent", "memory", "planning", "rag", "tool-use", "workflow-agent" ], "score": 21, "relevance": "high", "matched_queries": [ "planning-agent" ], "arxiv_id": "2605.20315", "source": "arxiv", "source_id": "arxiv:2605.20315", "pdf_url": "https://arxiv.org/pdf/2605.20315", "primary_query": "planning-agent" }, { "id": "2605.14498", "title": "GroupMemBench: Benchmarking LLM Agent Memory in Multi-Party Conversations", "url": "https://arxiv.org/abs/2605.14498", "published": "2026-05-14", "updated": "2026-05-16", "authors": [ "Jingbo Yang", "Kwei-Herng Lai", "Xiaowen Wang", "Shiyu Chang", "Yaar Harari", "Evgeniy Gabrilovich" ], "categories": [ "cs.CL" ], "topics": [ "agent-evaluation", "computer-use", "memory", "rag", "reasoning" ], "score": 21, "relevance": "high", "matched_queries": [ "agent-memory" ], "arxiv_id": "2605.14498", "source": "arxiv", "source_id": "arxiv:2605.14498", "pdf_url": "https://arxiv.org/pdf/2605.14498", "primary_query": "agent-memory" }, { "id": "2605.11633", "title": "Can LLM Agents Respond to Disasters? Benchmarking Heterogeneous Geospatial Reasoning in Emergency Operations", "url": "https://arxiv.org/abs/2605.11633", "published": "2026-05-12", "updated": "2026-05-12", "authors": [ "Junjue Wang", "Weihao Xuan", "Heli Qi", "Pengyu Dai", "Kunyi Liu", "Hongruixuan Chen", "Zhuo Zheng", "Junshi Xia", "Stefano Ermon", "Naoto Yokoya" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "planning", "rag", "reasoning", "tool-use", "workflow-agent" ], "score": 21, "relevance": "high", "matched_queries": [ "planning-agent" ], "arxiv_id": "2605.11633", "source": "arxiv", "source_id": "arxiv:2605.11633", "pdf_url": "https://arxiv.org/pdf/2605.11633", "primary_query": "planning-agent" }, { "id": "2605.06812", "title": "Towards Security-Auditable LLM Agents: A Unified Graph Representation", "url": "https://arxiv.org/abs/2605.06812", "published": "2026-05-07", "updated": "2026-05-07", "authors": [ "Chaofan Li", "Lyuye Zhang", "Jintao Zhai", "Siyue Feng", "Xichun Yang", "Huahao Wang", "Shihan Dou", "Yu Ji", "Yutao Hu", "Yueming Wu", "Yang Liu", "Deqing Zou" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "agent-safety", "memory", "multi-agent", "rag", "reasoning", "tool-use" ], "score": 21, "relevance": "high", "matched_queries": [ "agent-safety" ], "arxiv_id": "2605.06812", "source": "arxiv", "source_id": "arxiv:2605.06812", "pdf_url": "https://arxiv.org/pdf/2605.06812", "primary_query": "agent-safety" }, { "id": "2602.08412", "title": "From Assistant to Double Agent: Formalizing and Benchmarking Attacks on OpenClaw for Personalized Local AI Agent", "url": "https://arxiv.org/abs/2602.08412", "published": "2026-02-09", "updated": "2026-02-11", "authors": [ "Yuhang Wang", "Feiming Xu", "Zheng Lin", "Guangyu He", "Yuzhe Huang", "Haichang Gao", "Zhenxing Niu", "Shiguo Lian", "Zhaoxiang Liu" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "agent-safety", "memory", "planning", "rag", "tool-use" ], "score": 21, "relevance": "high", "matched_queries": [ "agent-safety" ], "arxiv_id": "2602.08412", "source": "arxiv", "source_id": "arxiv:2602.08412", "pdf_url": "https://arxiv.org/pdf/2602.08412", "primary_query": "agent-safety" }, { "id": "2508.07575", "title": "MCPToolBench++: A Large Scale AI Agent Model Context Protocol MCP Tool Use Benchmark", "url": "https://arxiv.org/abs/2508.07575", "published": "2025-08-11", "updated": "2025-08-11", "authors": [ "Shiqing Fan", "Xichen Ding", "Liang Zhang", "Linjian Mo" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "computer-use", "tool-use" ], "score": 21, "relevance": "high", "matched_queries": [ "function-calling" ], "arxiv_id": "2508.07575", "source": "arxiv", "source_id": "arxiv:2508.07575", "pdf_url": "https://arxiv.org/pdf/2508.07575", "primary_query": "function-calling" }, { "id": "2607.05318", "title": "PiSAs: Benchmarking Contextual Integrity in Multi-User Agentic Systems", "url": "https://arxiv.org/abs/2607.05318", "published": "2026-07-06", "updated": "2026-07-06", "authors": [ "Shubham Gupta", "Nazanin Mohammadi Sepahvand", "Abhinav Kumar", "Cem Subakan", "Spandana Gella", "Pierre-André Noël", "Perouz Taslakian", "Eugene Bagdasarian", "Valentina Zantedeschi" ], "categories": [ "cs.MA", "cs.CR" ], "topics": [ "agent-evaluation", "agent-safety", "memory", "tool-use" ], "score": 20, "relevance": "high", "matched_queries": [ "llm-agent" ], "arxiv_id": "2607.05318", "source": "arxiv", "source_id": "arxiv:2607.05318", "pdf_url": "https://arxiv.org/pdf/2607.05318", "primary_query": "llm-agent" }, { "id": "2607.04391", "title": "Memory-Orchestrated Semantic System (MOSS): An Auditable Agentic Memory Architecture", "url": "https://arxiv.org/abs/2607.04391", "published": "2026-07-05", "updated": "2026-07-05", "authors": [ "Serge Lacasse", "Jérémie Hatier", "Alex Baker" ], "categories": [ "cs.CL" ], "topics": [ "agent-evaluation", "memory", "rag", "tool-use" ], "score": 20, "relevance": "high", "matched_queries": [ "agent-memory", "ai-agent", "rag-agent" ], "arxiv_id": "2607.04391", "source": "arxiv", "source_id": "arxiv:2607.04391", "pdf_url": "https://arxiv.org/pdf/2607.04391", "primary_query": "agent-memory" }, { "id": "2607.03953", "title": "The Remarkable Effectiveness of Providing AI Agents with Natural Language Tools: A Replication Study Validating NLT Performance Across 14 Models", "url": "https://arxiv.org/abs/2607.03953", "published": "2026-07-04", "updated": "2026-07-04", "authors": [ "Alexander Somma", "Isabelle Plante", "Fred Premji" ], "categories": [ "cs.CL", "cs.AI" ], "topics": [ "agent-evaluation", "rag", "reasoning", "tool-use", "workflow-agent" ], "score": 20, "relevance": "high", "matched_queries": [ "agentic-ai", "ai-agent", "llm-agent", "tool-use" ], "arxiv_id": "2607.03953", "source": "arxiv", "source_id": "arxiv:2607.03953", "pdf_url": "https://arxiv.org/pdf/2607.03953", "primary_query": "agentic-ai" }, { "id": "2607.03726", "title": "SelfMem: Self-Optimizing Memory for AI Agents", "url": "https://arxiv.org/abs/2607.03726", "published": "2026-07-04", "updated": "2026-07-04", "authors": [ "Shu Yang", "Junchao Wu", "Derek F. Wong", "Di Wang" ], "categories": [ "cs.CL" ], "topics": [ "agent-evaluation", "computer-use", "memory", "planning", "rag", "tool-use" ], "score": 20, "relevance": "high", "matched_queries": [ "agent-memory", "ai-agent", "tool-use" ], "arxiv_id": "2607.03726", "source": "arxiv", "source_id": "arxiv:2607.03726", "pdf_url": "https://arxiv.org/pdf/2607.03726", "primary_query": "agent-memory" }, { "id": "2607.01766", "title": "SimWorlds: A Multi-Agent System for Dynamic 3D Scene Creation", "url": "https://arxiv.org/abs/2607.01766", "published": "2026-07-02", "updated": "2026-07-02", "authors": [ "Chunjiang Liu", "Xiaoyuan Wang", "Haoyu Chen", "Yizhou Zhao", "Ming-Hsuan Yang", "László A. Jeni" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "embodied-agent", "multi-agent", "planning", "reasoning", "tool-use", "workflow-agent" ], "score": 20, "relevance": "high", "matched_queries": [ "llm-agent", "multi-agent-llm" ], "arxiv_id": "2607.01766", "source": "arxiv", "source_id": "arxiv:2607.01766", "pdf_url": "https://arxiv.org/pdf/2607.01766", "primary_query": "llm-agent" }, { "id": "2607.00454", "title": "Agri-SAGE: Simulation-Grounded Multi-Agent LLM for Context-Aware Agricultural Advisory Generation", "url": "https://arxiv.org/abs/2607.00454", "published": "2026-07-01", "updated": "2026-07-01", "authors": [ "Vedant Balasubramaniam", "Geetha Charan", "Manojkumar Patil", "Rohit P Suresh", "V Priyanka", "Kodur Sai Vinay Sathvik", "Y. Narahari" ], "categories": [ "cs.AI", "cs.MA" ], "topics": [ "agent-evaluation", "agent-safety", "computer-use", "memory", "multi-agent", "planning", "rag", "reasoning", "world-model" ], "score": 20, "relevance": "high", "matched_queries": [ "multi-agent-llm" ], "arxiv_id": "2607.00454", "source": "arxiv", "source_id": "arxiv:2607.00454", "pdf_url": "https://arxiv.org/pdf/2607.00454", "primary_query": "multi-agent-llm" }, { "id": "2606.31073", "title": "MultiUAV-Plat: An LLM-Oriented Platform, Benchmark and Framework for Multi-UAV Collaborative Task Planning", "url": "https://arxiv.org/abs/2606.31073", "published": "2026-06-30", "updated": "2026-06-30", "authors": [ "Sheng Zhang", "Qinglin Li", "Yuechao Zang", "Xueqin Huang", "Yijia Fu", "Cheng Zhu" ], "categories": [ "cs.AI", "cs.MA", "cs.RO" ], "topics": [ "agent-evaluation", "embodied-agent", "memory", "multi-agent", "planning", "rag", "tool-use", "world-model" ], "score": 20, "relevance": "high", "matched_queries": [ "agent-evaluation", "llm-agent", "planning-agent" ], "arxiv_id": "2606.31073", "source": "arxiv", "source_id": "arxiv:2606.31073", "pdf_url": "https://arxiv.org/pdf/2606.31073", "primary_query": "agent-evaluation" }, { "id": "2606.31179", "title": "HealthAgentBench: A Unified Benchmark Suite of Realistic Agentic Healthcare Environments for Challenging Frontier AI Agents", "url": "https://arxiv.org/abs/2606.31179", "published": "2026-06-30", "updated": "2026-06-30", "authors": [ "Qianchu Liu", "Sheng Zhang", "Guanghui Qin", "Jeya Maria Jose Valanarasu", "Maximilian Rokuss", "Mingyu Lu", "Timothy Ossowski", "Juan Manuel Zambrano Chaves", "Cliff Wong", "Peniel Argaw", "Yashna Hasija", "Mu Wei", "Wen-wai Yim", "Qin Liu", "Zilin Jing", "Jason Entenmann", "Naoto Usuyama", "Tristan Naumann", "Hoifung Poon" ], "categories": [ "cs.AI", "cs.CL", "cs.CV" ], "topics": [ "agent-evaluation", "planning", "reasoning", "workflow-agent" ], "score": 20, "relevance": "high", "matched_queries": [ "agent-evaluation", "ai-agent" ], "arxiv_id": "2606.31179", "source": "arxiv", "source_id": "arxiv:2606.31179", "pdf_url": "https://arxiv.org/pdf/2606.31179", "primary_query": "agent-evaluation" }, { "id": "2606.31612", "title": "What Memory Do GUI Agents Really Need? From Passive Records to Active Task-Driving States", "url": "https://arxiv.org/abs/2606.31612", "published": "2026-06-30", "updated": "2026-07-02", "authors": [ "Chen Liu", "Ling Chen", "Hanzhang Zhou", "Xu Zhang", "Quyu Kong", "Panrong Tong", "Wenhao Wang", "Xin Yu", "Steven Hoi", "Yue Wang" ], "categories": [ "cs.CV" ], "topics": [ "agent-evaluation", "computer-use", "memory", "planning", "rag", "tool-use", "workflow-agent" ], "score": 20, "relevance": "high", "matched_queries": [ "agent-memory", "web-gui-agent" ], "arxiv_id": "2606.31612", "source": "arxiv", "source_id": "arxiv:2606.31612", "pdf_url": "https://arxiv.org/pdf/2606.31612", "primary_query": "agent-memory" }, { "id": "2606.30906", "title": "Investigating Multi-Agent Deliberation in Law", "url": "https://arxiv.org/abs/2606.30906", "published": "2026-06-29", "updated": "2026-06-29", "authors": [ "Cor Steging", "Ludi van Leeuwen", "Tadeusz Zbiegień" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "multi-agent", "reasoning", "tool-use" ], "score": 20, "relevance": "high", "matched_queries": [ "agentic-ai", "ai-agent", "multi-agent-llm" ], "arxiv_id": "2606.30906", "source": "arxiv", "source_id": "arxiv:2606.30906", "pdf_url": "https://arxiv.org/pdf/2606.30906", "primary_query": "agentic-ai" }, { "id": "2606.30949", "title": "AgRefactor: Self-Evolving Agentic Workflow for HLS Compatibility and Performance", "url": "https://arxiv.org/abs/2606.30949", "published": "2026-06-29", "updated": "2026-06-29", "authors": [ "Yang Zou", "Zijian Ding", "Yizhou Sun", "Jason Cong" ], "categories": [ "cs.AI", "cs.AR" ], "topics": [ "agent-evaluation", "memory", "multi-agent", "rag", "tool-use", "workflow-agent" ], "score": 20, "relevance": "high", "matched_queries": [ "agentic-ai", "multi-agent-llm" ], "arxiv_id": "2606.30949", "source": "arxiv", "source_id": "arxiv:2606.30949", "pdf_url": "https://arxiv.org/pdf/2606.30949", "primary_query": "agentic-ai" }, { "id": "2606.29116", "title": "Characterizing Large Language Model Agentic Workflows: A Study on N8n Ecosystem", "url": "https://arxiv.org/abs/2606.29116", "published": "2026-06-27", "updated": "2026-06-27", "authors": [ "Yutian Tang", "Yuming Zhou", "Huaming Chen" ], "categories": [ "cs.AI" ], "topics": [ "agent-safety", "planning", "rag", "tool-use", "workflow-agent" ], "score": 20, "relevance": "high", "matched_queries": [ "agentic-ai", "llm-agent", "planning-agent", "tool-use" ], "arxiv_id": "2606.29116", "source": "arxiv", "source_id": "arxiv:2606.29116", "pdf_url": "https://arxiv.org/pdf/2606.29116", "primary_query": "agentic-ai" }, { "id": "2606.24535", "title": "Governed Shared Memory for Multi-Agent LLM Systems", "url": "https://arxiv.org/abs/2606.24535", "published": "2026-06-23", "updated": "2026-06-23", "authors": [ "Yanki Margalit", "Nurit Cohen-Inger", "Erni Avram", "Ran Taig", "Oded Margalit" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "memory", "multi-agent", "rag", "tool-use" ], "score": 20, "relevance": "high", "matched_queries": [ "agent-memory", "multi-agent-llm" ], "arxiv_id": "2606.24535", "source": "arxiv", "source_id": "arxiv:2606.24535", "pdf_url": "https://arxiv.org/pdf/2606.24535", "primary_query": "agent-memory" }, { "id": "2606.24820", "title": "SHERLOC: Structured Diagnostic Localization for Code Repair Agents", "url": "https://arxiv.org/abs/2606.24820", "published": "2026-06-23", "updated": "2026-06-23", "authors": [ "Hovhannes Tamoyan", "Sean Narenthiran", "Erik Arakelyan", "Mira Mezini", "Boris Ginsburg" ], "categories": [ "cs.CL" ], "topics": [ "agent-evaluation", "coding-agent", "multi-agent", "rag", "reasoning", "tool-use" ], "score": 20, "relevance": "high", "matched_queries": [ "coding-agent", "multi-agent-llm", "tool-use" ], "arxiv_id": "2606.24820", "source": "arxiv", "source_id": "arxiv:2606.24820", "pdf_url": "https://arxiv.org/pdf/2606.24820", "primary_query": "coding-agent" }, { "id": "2606.22263", "title": "Revelio: Cost-Efficient Agentic Memory Safety Vulnerability Detection For Repository-Scale Codebases", "url": "https://arxiv.org/abs/2606.22263", "published": "2026-06-20", "updated": "2026-06-20", "authors": [ "Yiwei Hou", "Hao Wang", "Muxi Lyu", "Marius Momeu", "Eric Nguyen", "Taige Yang", "Koushik Sen", "Dawn Song", "David Wagner" ], "categories": [ "cs.CR", "cs.AI", "cs.MA", "cs.SE" ], "topics": [ "agent-evaluation", "agent-safety", "coding-agent", "memory" ], "score": 20, "relevance": "high", "matched_queries": [ "agent-memory", "coding-agent" ], "arxiv_id": "2606.22263", "source": "arxiv", "source_id": "arxiv:2606.22263", "pdf_url": "https://arxiv.org/pdf/2606.22263", "primary_query": "agent-memory" }, { "id": "2606.21627", "title": "Counsel: A Meta-Evaluation Dataset for Agentic Tasks", "url": "https://arxiv.org/abs/2606.21627", "published": "2026-06-19", "updated": "2026-06-19", "authors": [ "Sashank Pisupati", "Henry Broomfield", "Eujeong Choi", "Antonia Calvi", "Charlie Wang", "Roman Engeler", "Max Bartolo", "Patrick Lewis" ], "categories": [ "cs.AI", "cs.LG" ], "topics": [ "agent-evaluation", "agent-safety", "coding-agent", "reasoning" ], "score": 20, "relevance": "high", "matched_queries": [ "agent-evaluation", "coding-agent" ], "arxiv_id": "2606.21627", "source": "arxiv", "source_id": "arxiv:2606.21627", "pdf_url": "https://arxiv.org/pdf/2606.21627", "primary_query": "agent-evaluation" }, { "id": "2606.16871", "title": "Human-on-the-Bridge: Scalable Evaluation for AI Agents", "url": "https://arxiv.org/abs/2606.16871", "published": "2026-06-15", "updated": "2026-06-15", "authors": [ "Fouad Bousetouane" ], "categories": [ "cs.MA" ], "topics": [ "agent-evaluation", "computer-use", "memory", "rag", "tool-use" ], "score": 20, "relevance": "high", "matched_queries": [ "agent-evaluation" ], "arxiv_id": "2606.16871", "source": "arxiv", "source_id": "arxiv:2606.16871", "pdf_url": "https://arxiv.org/pdf/2606.16871", "primary_query": "agent-evaluation" }, { "id": "2606.17246", "title": "GeoDisaster: Benchmarking Orchestrated Agents for Operational Disaster Geo-Intelligence", "url": "https://arxiv.org/abs/2606.17246", "published": "2026-06-15", "updated": "2026-06-15", "authors": [ "Maram Hasan", "Aman Verma", "Savitra Roy", "Hariseetharam Gunduboina", "Daksh Jain", "Muhammad Haris Khan", "Subhasis Chaudhuri", "Biplab Banerjee" ], "categories": [ "cs.CV", "cs.MA" ], "topics": [ "agent-evaluation", "agent-safety", "multi-agent", "rag", "reasoning", "tool-use", "workflow-agent" ], "score": 20, "relevance": "high", "matched_queries": [ "tool-use" ], "arxiv_id": "2606.17246", "source": "arxiv", "source_id": "arxiv:2606.17246", "pdf_url": "https://arxiv.org/pdf/2606.17246", "primary_query": "tool-use" }, { "id": "2606.17114", "title": "An Evaluation of Data Leakage Risks in Tool-Using LLM Agents in Realistic Scenarios", "url": "https://arxiv.org/abs/2606.17114", "published": "2026-06-15", "updated": "2026-06-15", "authors": [ "Hankyul Baek", "Jaewon Noh", "Sang Seo", "Yongsu Kim", "Gabriel Waikin Loh Matienzo", "Young Il Kim", "Ee Wei Seah", "Akriti Vij" ], "categories": [ "cs.CR", "cs.AI" ], "topics": [ "agent-evaluation", "agent-safety", "tool-use", "workflow-agent", "world-model" ], "score": 20, "relevance": "high", "matched_queries": [ "agent-safety", "tool-use" ], "arxiv_id": "2606.17114", "source": "arxiv", "source_id": "arxiv:2606.17114", "pdf_url": "https://arxiv.org/pdf/2606.17114", "primary_query": "agent-safety" }, { "id": "2606.16420", "title": "Transferable Self-Evolving Playbooks for Agentic Security Auditing", "url": "https://arxiv.org/abs/2606.16420", "published": "2026-06-15", "updated": "2026-06-15", "authors": [ "Ziyue Wang", "Cheuk Wang Maurice Ng", "Chenchen Yu", "Strick Sheng", "Kaihua Qin", "Liyi Zhou" ], "categories": [ "cs.CR" ], "topics": [ "agent-evaluation", "agent-safety", "computer-use", "embodied-agent", "rag", "tool-use", "workflow-agent" ], "score": 20, "relevance": "high", "matched_queries": [ "agent-safety", "tool-use" ], "arxiv_id": "2606.16420", "source": "arxiv", "source_id": "arxiv:2606.16420", "pdf_url": "https://arxiv.org/pdf/2606.16420", "primary_query": "agent-safety" }, { "id": "2606.14790", "title": "XFlow: An Executable Protocol Programming System for Reliable Multi-Agent Workflows", "url": "https://arxiv.org/abs/2606.14790", "published": "2026-06-11", "updated": "2026-06-11", "authors": [ "Hanqi Li", "Jing Peng", "Zijian Wang", "Lu Chen", "Kai Yu" ], "categories": [ "cs.PL", "cs.AI" ], "topics": [ "coding-agent", "memory", "multi-agent", "planning", "reasoning", "tool-use", "workflow-agent" ], "score": 20, "relevance": "high", "matched_queries": [ "tool-use" ], "arxiv_id": "2606.14790", "source": "arxiv", "source_id": "arxiv:2606.14790", "pdf_url": "https://arxiv.org/pdf/2606.14790", "primary_query": "tool-use" }, { "id": "2606.08531", "title": "VESTA: A Fully Automated Scenario Generation and Safety Evaluation Framework for LLM Agents", "url": "https://arxiv.org/abs/2606.08531", "published": "2026-06-07", "updated": "2026-06-07", "authors": [ "Lu Jia", "Haibo Tong", "Feifei Zhao", "Jindong Li", "Dongqi Liang", "Ping Wu", "Qian Zhang", "Yi Zeng" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "agent-safety", "memory", "rag", "tool-use" ], "score": 20, "relevance": "high", "matched_queries": [ "agent-safety" ], "arxiv_id": "2606.08531", "source": "arxiv", "source_id": "arxiv:2606.08531", "pdf_url": "https://arxiv.org/pdf/2606.08531", "primary_query": "agent-safety" }, { "id": "2606.07402", "title": "M$^3$Exam: Benchmarking Multimodal Memory for Realistic User-Agent Interactions", "url": "https://arxiv.org/abs/2606.07402", "published": "2026-06-05", "updated": "2026-06-05", "authors": [ "Zhengjun Huang", "Wenxuan Liu", "Zhoujin Tian", "Wei Chen", "Junle Chen", "Yuqian Wu", "Fangyuan Zhang", "Qintian Guo", "Xiaofang Zhou" ], "categories": [ "cs.CL" ], "topics": [ "agent-evaluation", "memory", "rag", "reasoning", "tool-use" ], "score": 20, "relevance": "high", "matched_queries": [ "language-agent" ], "arxiv_id": "2606.07402", "source": "arxiv", "source_id": "arxiv:2606.07402", "pdf_url": "https://arxiv.org/pdf/2606.07402", "primary_query": "language-agent" }, { "id": "2606.04780", "title": "PersonaTree: Structured Lifecycle Memory for Person Understanding in LLM Agents", "url": "https://arxiv.org/abs/2606.04780", "published": "2026-06-03", "updated": "2026-06-03", "authors": [ "Yubo Hou", "Jingwei Song", "Hongbo Zhang", "Zhisheng Chen", "Bang Xiao", "Tao Wan", "Zengchang Qin" ], "categories": [ "cs.CL" ], "topics": [ "agent-evaluation", "agent-safety", "computer-use", "memory", "rag", "tool-use" ], "score": 20, "relevance": "high", "matched_queries": [ "agent-memory" ], "arxiv_id": "2606.04780", "source": "arxiv", "source_id": "arxiv:2606.04780", "pdf_url": "https://arxiv.org/pdf/2606.04780", "primary_query": "agent-memory" }, { "id": "2606.03657", "title": "Diagnosing Knowledge Gaps in LLM Tool Use: An Agentic Benchmark for Novel API Acquisition", "url": "https://arxiv.org/abs/2606.03657", "published": "2026-06-02", "updated": "2026-06-02", "authors": [ "Jinnuo Liu", "Yue Peng", "Jinhan Niu", "Hongyi Wen" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "coding-agent", "rag", "tool-use" ], "score": 20, "relevance": "high", "matched_queries": [ "agent-evaluation" ], "arxiv_id": "2606.03657", "source": "arxiv", "source_id": "arxiv:2606.03657", "pdf_url": "https://arxiv.org/pdf/2606.03657", "primary_query": "agent-evaluation" }, { "id": "2606.02109", "title": "BADGER: Bridging Agentic and Deterministic Evaluation for Generative Enterprise Reasoning", "url": "https://arxiv.org/abs/2606.02109", "published": "2026-06-01", "updated": "2026-06-01", "authors": [ "Shannon Serrao", "Soumitra Chatterjee", "Dorina Strori", "Abhishek Sharma", "Nathan Miller" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "agent-safety", "rag", "reasoning", "tool-use", "workflow-agent" ], "score": 20, "relevance": "high", "matched_queries": [ "agent-evaluation" ], "arxiv_id": "2606.02109", "source": "arxiv", "source_id": "arxiv:2606.02109", "pdf_url": "https://arxiv.org/pdf/2606.02109", "primary_query": "agent-evaluation" }, { "id": "2606.01416", "title": "Self-Healing Agentic Orchestrators for Reliable Tool-Augmented Large Language Model Systems", "url": "https://arxiv.org/abs/2606.01416", "published": "2026-05-31", "updated": "2026-05-31", "authors": [ "Rahul Suresh Babu", "Adarsh Agrawal" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "computer-use", "memory", "planning", "rag", "reasoning", "tool-use", "workflow-agent" ], "score": 20, "relevance": "high", "matched_queries": [ "planning-agent" ], "arxiv_id": "2606.01416", "source": "arxiv", "source_id": "arxiv:2606.01416", "pdf_url": "https://arxiv.org/pdf/2606.01416", "primary_query": "planning-agent" }, { "id": "2606.00610", "title": "MemGraphRAG: Memory-based Multi-Agent System for Graph Retrieval-Augmented Generation", "url": "https://arxiv.org/abs/2606.00610", "published": "2026-05-30", "updated": "2026-05-30", "authors": [ "Chuanjie Wu", "Zhishang Xiang", "Yunbo Tang", "Zerui Chen", "Qinggang Zhang", "Jinsong Su" ], "categories": [ "cs.IR", "cs.AI", "cs.MA" ], "topics": [ "agent-evaluation", "memory", "multi-agent", "rag", "reasoning", "tool-use" ], "score": 20, "relevance": "high", "matched_queries": [ "rag-agent" ], "arxiv_id": "2606.00610", "source": "arxiv", "source_id": "arxiv:2606.00610", "pdf_url": "https://arxiv.org/pdf/2606.00610", "primary_query": "rag-agent" }, { "id": "2605.30883", "title": "TRACE: Task-Aware Adaptive Self-Evolving Agentic Jailbreaking", "url": "https://arxiv.org/abs/2605.30883", "published": "2026-05-29", "updated": "2026-05-29", "authors": [ "Churui Zeng", "Weiwei Qi", "Kedong Xiu", "Tianhang Zheng", "Chaochao Lu", "Liang He", "Zhan Qin", "Kui Ren" ], "categories": [ "cs.CR" ], "topics": [ "agent-evaluation", "agent-safety", "coding-agent", "computer-use", "planning", "rag", "tool-use", "workflow-agent" ], "score": 20, "relevance": "high", "matched_queries": [ "planning-agent" ], "arxiv_id": "2605.30883", "source": "arxiv", "source_id": "arxiv:2605.30883", "pdf_url": "https://arxiv.org/pdf/2605.30883", "primary_query": "planning-agent" }, { "id": "2605.30090", "title": "DirectorBench: Diagnosing Long-Form Video Generation with Personalized Multi-Agent Evaluation", "url": "https://arxiv.org/abs/2605.30090", "published": "2026-05-28", "updated": "2026-05-28", "authors": [ "Jiamin Chen", "Qianben Chen", "Jiawen Zhang", "Yidi Wu", "Yuchen Li", "Xiaokun Zhang", "Wangchunshu Zhou", "Chen Ma" ], "categories": [ "cs.CL", "cs.CV" ], "topics": [ "agent-evaluation", "agent-safety", "multi-agent", "rag", "tool-use", "workflow-agent" ], "score": 20, "relevance": "high", "matched_queries": [ "agent-evaluation" ], "arxiv_id": "2605.30090", "source": "arxiv", "source_id": "arxiv:2605.30090", "pdf_url": "https://arxiv.org/pdf/2605.30090", "primary_query": "agent-evaluation" }, { "id": "2605.27134", "title": "Scaling, Benchmarking, and Reasoning of Vision-Language Agents for Mobile GUI Navigation", "url": "https://arxiv.org/abs/2605.27134", "published": "2026-05-26", "updated": "2026-05-26", "authors": [ "Heng Qu", "Yike Liu", "Renren Jin", "Wenzong Zhang", "Pengzhi Gao", "Wei Liu", "Jian Luan" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "computer-use", "embodied-agent", "rag", "reasoning", "tool-use" ], "score": 20, "relevance": "high", "matched_queries": [ "language-agent" ], "arxiv_id": "2605.27134", "source": "arxiv", "source_id": "arxiv:2605.27134", "pdf_url": "https://arxiv.org/pdf/2605.27134", "primary_query": "language-agent" }, { "id": "2605.22643", "title": "Boiling the Frog: A Multi-Turn Benchmark for Agentic Safety", "url": "https://arxiv.org/abs/2605.22643", "published": "2026-05-21", "updated": "2026-05-22", "authors": [ "Piercosma Bisconti", "Matteo Prandi", "Federico Pierucci", "Federico Sartore", "Enrico Panai", "Laura Caroli", "Yue Zhu", "Adam Leon Smith", "Luca Nannini", "Marcello Galisai", "Susanna Cifani", "Francesco Giarrusso", "Marcantonio Bracale Syrnikov", "Daniele Nardi" ], "categories": [ "cs.CL" ], "topics": [ "agent-evaluation", "agent-safety", "memory", "rag", "tool-use", "workflow-agent" ], "score": 20, "relevance": "high", "matched_queries": [ "agent-safety" ], "arxiv_id": "2605.22643", "source": "arxiv", "source_id": "arxiv:2605.22643", "pdf_url": "https://arxiv.org/pdf/2605.22643", "primary_query": "agent-safety" }, { "id": "2605.15040", "title": "Orchard: An Open-Source Agentic Modeling Framework", "url": "https://arxiv.org/abs/2605.15040", "published": "2026-05-14", "updated": "2026-05-21", "authors": [ "Baolin Peng", "Wenlin Yao", "Qianhui Wu", "Hao Cheng", "Xiao Yu", "Rui Yang", "Tao Ge", "Alessandro Sordoni", "Xingdi Yuan", "Yelong Shen", "Pengcheng He", "Tong Zhang", "Zhou Yu", "Jianfeng Gao" ], "categories": [ "cs.AI", "cs.CL" ], "topics": [ "agent-evaluation", "coding-agent", "computer-use", "planning", "reasoning", "tool-use" ], "score": 20, "relevance": "high", "matched_queries": [ "autonomous-agent-llm" ], "arxiv_id": "2605.15040", "source": "arxiv", "source_id": "arxiv:2605.15040", "pdf_url": "https://arxiv.org/pdf/2605.15040", "primary_query": "autonomous-agent-llm" }, { "id": "2605.13542", "title": "RealICU: Do LLM Agents Understand Long-Context ICU Data? A Benchmark Beyond Behavior Imitation", "url": "https://arxiv.org/abs/2605.13542", "published": "2026-05-13", "updated": "2026-05-13", "authors": [ "Chengzhi Shen", "Weixiang Shen", "Tobias Susetzky", "Chen", "Chen", "Jun Li", "Yuyuan Liu", "Xuepeng Zhang", "Zhenyu Gong", "Daniel Rueckert", "Jiazhen Pan" ], "categories": [ "cs.AI", "cs.CL", "cs.LG", "cs.MA" ], "topics": [ "agent-evaluation", "agent-safety", "memory", "planning", "reasoning", "tool-use" ], "score": 20, "relevance": "high", "matched_queries": [ "agent-memory" ], "arxiv_id": "2605.13542", "source": "arxiv", "source_id": "arxiv:2605.13542", "pdf_url": "https://arxiv.org/pdf/2605.13542", "primary_query": "agent-memory" }, { "id": "2605.12015", "title": "SkillSafetyBench: Evaluating Agent Safety under Skill-Facing Attack Surfaces", "url": "https://arxiv.org/abs/2605.12015", "published": "2026-05-12", "updated": "2026-05-27", "authors": [ "Chang Jin", "An Wang", "Zeming Wei", "Kai Wang", "Biaojie Zeng", "Qiaosheng Zhang", "Chao Yang", "Jingjing Qu", "Xia Hu", "Xingcheng Xu" ], "categories": [ "cs.CR", "cs.AI", "cs.CL", "cs.LG", "cs.MA" ], "topics": [ "agent-evaluation", "agent-safety", "computer-use", "memory", "reasoning", "tool-use", "workflow-agent" ], "score": 20, "relevance": "high", "matched_queries": [ "agent-safety" ], "arxiv_id": "2605.12015", "source": "arxiv", "source_id": "arxiv:2605.12015", "pdf_url": "https://arxiv.org/pdf/2605.12015", "primary_query": "agent-safety" }, { "id": "2605.08374", "title": "MemQ: Integrating Q-Learning into Self-Evolving Memory Agents over Provenance DAGs", "url": "https://arxiv.org/abs/2605.08374", "published": "2026-05-08", "updated": "2026-05-14", "authors": [ "Junwei Liao", "Haoting Shi", "Ruiwen Zhou", "Jiaqian Wang", "Shengtao Zhang", "Wei Zhang", "Ying Wen", "Zhiyu Li", "Feiyu Xiong", "Bo Tang", "Weinan Zhang", "Muning Wen" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "computer-use", "embodied-agent", "memory", "rag", "reasoning", "tool-use" ], "score": 20, "relevance": "high", "matched_queries": [ "function-calling" ], "arxiv_id": "2605.08374", "source": "arxiv", "source_id": "arxiv:2605.08374", "pdf_url": "https://arxiv.org/pdf/2605.08374", "primary_query": "function-calling" }, { "id": "2605.15206", "title": "AgentStop: Terminating Local AI Agents Early to Save Energy in Consumer Devices", "url": "https://arxiv.org/abs/2605.15206", "published": "2026-05-01", "updated": "2026-05-01", "authors": [ "Dzung Pham", "Kleomenis Katevas", "Ali Shahin Shamsabadi", "Hamed Haddadi" ], "categories": [ "cs.LG", "cs.AI", "cs.DC" ], "topics": [ "agent-evaluation", "coding-agent", "rag", "reasoning", "tool-use", "workflow-agent" ], "score": 20, "relevance": "high", "matched_queries": [ "autonomous-agent-llm" ], "arxiv_id": "2605.15206", "source": "arxiv", "source_id": "arxiv:2605.15206", "pdf_url": "https://arxiv.org/pdf/2605.15206", "primary_query": "autonomous-agent-llm" }, { "id": "2605.16282", "title": "Taxonomy and Consistency Analysis of Safety Benchmarks for AI Agents", "url": "https://arxiv.org/abs/2605.16282", "published": "2026-04-11", "updated": "2026-04-11", "authors": [ "Miles Q. Li", "Benjamin C. M. Fung", "Boyang Li", "Heba Ismail", "Farkhund Iqbal" ], "categories": [ "cs.CY", "cs.AI" ], "topics": [ "agent-evaluation", "agent-safety", "coding-agent", "rag", "tool-use" ], "score": 20, "relevance": "high", "matched_queries": [ "agent-safety" ], "arxiv_id": "2605.16282", "source": "arxiv", "source_id": "arxiv:2605.16282", "pdf_url": "https://arxiv.org/pdf/2605.16282", "primary_query": "agent-safety" }, { "id": "2604.02022", "title": "ATBench: A Diverse and Realistic Agent Trajectory Benchmark for Safety Evaluation and Diagnosis", "url": "https://arxiv.org/abs/2604.02022", "published": "2026-04-02", "updated": "2026-05-13", "authors": [ "Yu Li", "Haoyu Luo", "Yuejin Xie", "Yuqian Fu", "Zhonghao Yang", "Shuai Shao", "Qihan Ren", "Wanying Qu", "Yanwei Fu", "Yujiu Yang", "Jing Shao", "Xia Hu", "Dongrui Liu" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "agent-safety", "planning", "rag", "tool-use" ], "score": 20, "relevance": "high", "matched_queries": [ "agent-safety" ], "arxiv_id": "2604.02022", "source": "arxiv", "source_id": "arxiv:2604.02022", "pdf_url": "https://arxiv.org/pdf/2604.02022", "primary_query": "agent-safety" }, { "id": "2603.16734", "title": "Differential Harm Propensity in Personalized LLM Agents: The Curious Case of Mental Health Disclosure", "url": "https://arxiv.org/abs/2603.16734", "published": "2026-03-17", "updated": "2026-03-17", "authors": [ "Caglar Yildirim" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "agent-safety", "memory", "rag", "tool-use" ], "score": 20, "relevance": "high", "matched_queries": [ "agent-safety" ], "arxiv_id": "2603.16734", "source": "arxiv", "source_id": "arxiv:2603.16734", "pdf_url": "https://arxiv.org/pdf/2603.16734", "primary_query": "agent-safety" }, { "id": "2509.20998", "title": "CORE: Full-Path Evaluation of LLM Agents Beyond Final State", "url": "https://arxiv.org/abs/2509.20998", "published": "2025-09-25", "updated": "2025-09-25", "authors": [ "Panagiotis Michelakis", "Yiannis Hadjiyiannis", "Dimitrios Stamoulis" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "agent-safety", "tool-use", "world-model" ], "score": 20, "relevance": "high", "matched_queries": [ "function-calling" ], "arxiv_id": "2509.20998", "source": "arxiv", "source_id": "arxiv:2509.20998", "pdf_url": "https://arxiv.org/pdf/2509.20998", "primary_query": "function-calling" }, { "id": "2607.05174", "title": "AgentGym2: Benchmarking Large Language Model Agents in De-Idealized Real-World Environments", "url": "https://arxiv.org/abs/2607.05174", "published": "2026-07-06", "updated": "2026-07-06", "authors": [ "Zhiheng Xi", "Dingwen Yang", "Jiaqi Liu", "Jixuan Huang", "Honglin Guo", "Baodai Huang", "Tinggang Chen", "Qi Zhang", "Zhonghang Lu", "Chenyu Liu", "Jiajun Sun", "Jiazheng Zhang", "Dingwei Zhu", "Xin Guo", "Junzhe Wang", "Zhihao Zhang", "Yuming Yang", "Junjie Ye", "Minghe Gao", "Dongrui Liu", "Jiaming Ji", "Guohao Li", "Tao Gui", "Qi Zhang", "Xuanjing Huang" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "planning", "reasoning", "tool-use" ], "score": 19, "relevance": "high", "matched_queries": [ "language-agent", "llm-agent", "planning-agent" ], "arxiv_id": "2607.05174", "source": "arxiv", "source_id": "arxiv:2607.05174", "pdf_url": "https://arxiv.org/pdf/2607.05174", "primary_query": "language-agent" }, { "id": "2607.05029", "title": "Your Agent's Memories Are Not Its Own: Forged Reasoning Attacks on LLM Agent Memory and Defenses", "url": "https://arxiv.org/abs/2607.05029", "published": "2026-07-06", "updated": "2026-07-06", "authors": [ "Neeraj Karamchandani", "Piyush Nagasubramaniam", "Sencun Zhu", "Dinghao Wu" ], "categories": [ "cs.CR", "cs.AI" ], "topics": [ "agent-evaluation", "memory", "reasoning", "tool-use" ], "score": 19, "relevance": "high", "matched_queries": [ "agent-memory", "llm-agent" ], "arxiv_id": "2607.05029", "source": "arxiv", "source_id": "arxiv:2607.05029", "pdf_url": "https://arxiv.org/pdf/2607.05029", "primary_query": "agent-memory" }, { "id": "2607.05120", "title": "Agent Data Injection Attacks are Realistic Threats to AI Agents", "url": "https://arxiv.org/abs/2607.05120", "published": "2026-07-06", "updated": "2026-07-06", "authors": [ "Woohyuk Choi", "Juhee Kim", "Taehyun Kang", "Jihyeon Jeong", "Luyi Xing", "Byoungyoung Lee" ], "categories": [ "cs.CR", "cs.AI" ], "topics": [ "agent-evaluation", "agent-safety", "coding-agent", "computer-use", "tool-use" ], "score": 19, "relevance": "high", "matched_queries": [ "agent-safety", "ai-agent", "coding-agent", "web-gui-agent" ], "arxiv_id": "2607.05120", "source": "arxiv", "source_id": "arxiv:2607.05120", "pdf_url": "https://arxiv.org/pdf/2607.05120", "primary_query": "agent-safety" }, { "id": "2607.05202", "title": "EvoAgentBench: Benchmarking Agent Self-Evolution via Ability Transfer", "url": "https://arxiv.org/abs/2607.05202", "published": "2026-07-06", "updated": "2026-07-06", "authors": [ "Xingze Gao", "Chuanrui Hu", "Hongda Chen", "Pengfei Yao", "Zhao Wang", "Yi Bai", "Zhengwei Wu", "Yunyun Han", "Xiaofeng Cong", "Jie Gui", "Yafeng Deng", "Teng Li" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "coding-agent", "computer-use", "memory", "planning", "reasoning" ], "score": 19, "relevance": "high", "matched_queries": [ "agent-evaluation" ], "arxiv_id": "2607.05202", "source": "arxiv", "source_id": "arxiv:2607.05202", "pdf_url": "https://arxiv.org/pdf/2607.05202", "primary_query": "agent-evaluation" }, { "id": "2607.04395", "title": "NKI-Agent: Domain-Specific Fine-Tuning and Agentic Tool Use for Neuron Kernel Generation", "url": "https://arxiv.org/abs/2607.04395", "published": "2026-07-05", "updated": "2026-07-05", "authors": [ "Junjie Tang", "Jun Huan", "Hao Zhou", "Yuhao Zhang", "Lin Wang" ], "categories": [ "cs.LG" ], "topics": [ "agent-evaluation", "computer-use", "memory", "tool-use" ], "score": 19, "relevance": "high", "matched_queries": [ "tool-use" ], "arxiv_id": "2607.04395", "source": "arxiv", "source_id": "arxiv:2607.04395", "pdf_url": "https://arxiv.org/pdf/2607.04395", "primary_query": "tool-use" }, { "id": "2607.03510", "title": "CAGE-1: Control, Assurance, and Governance Evaluation for Enterprise Agentic AI", "url": "https://arxiv.org/abs/2607.03510", "published": "2026-07-03", "updated": "2026-07-03", "authors": [ "Roopam W. Sure" ], "categories": [ "cs.SE", "cs.AI", "cs.CY" ], "topics": [ "agent-evaluation", "agent-safety", "memory", "planning", "rag", "tool-use", "workflow-agent" ], "score": 19, "relevance": "high", "matched_queries": [ "agentic-ai" ], "arxiv_id": "2607.03510", "source": "arxiv", "source_id": "arxiv:2607.03510", "pdf_url": "https://arxiv.org/pdf/2607.03510", "primary_query": "agentic-ai" }, { "id": "2607.02684", "title": "Can Coding Agents Implement Missed Compiler Optimizations? Evaluating LLM Agents on LLVM Peephole Optimizations", "url": "https://arxiv.org/abs/2607.02684", "published": "2026-07-02", "updated": "2026-07-02", "authors": [ "Hongxu Xu", "Chunhao Liao", "Xintong Zhou", "Chengnian Sun" ], "categories": [ "cs.SE" ], "topics": [ "agent-evaluation", "coding-agent", "reasoning" ], "score": 19, "relevance": "high", "matched_queries": [ "coding-agent", "llm-agent" ], "arxiv_id": "2607.02684", "source": "arxiv", "source_id": "arxiv:2607.02684", "pdf_url": "https://arxiv.org/pdf/2607.02684", "primary_query": "coding-agent" }, { "id": "2607.02507", "title": "What LLM Agents Say When No One Is Watching: Social Structure and Latent Objective Emergence in Multi-Agent Debates", "url": "https://arxiv.org/abs/2607.02507", "published": "2026-07-02", "updated": "2026-07-02", "authors": [ "Arman Ghaffarizadeh", "Danyal Mohaddes", "Aliakbar Izadkhah", "Shahriar Noroozizadeh" ], "categories": [ "cs.AI", "cs.CL", "cs.LG", "cs.MA" ], "topics": [ "agent-evaluation", "agent-safety", "multi-agent" ], "score": 19, "relevance": "high", "matched_queries": [ "agent-evaluation", "llm-agent", "multi-agent-llm" ], "arxiv_id": "2607.02507", "source": "arxiv", "source_id": "arxiv:2607.02507", "pdf_url": "https://arxiv.org/pdf/2607.02507", "primary_query": "agent-evaluation" }, { "id": "2606.30986", "title": "The Organizational Behavior of Agentic AI: Collective Intelligence in Human-Agent Workflows", "url": "https://arxiv.org/abs/2606.30986", "published": "2026-06-29", "updated": "2026-06-29", "authors": [ "Canhui Liu" ], "categories": [ "cs.CY", "cs.HC", "cs.MA", "econ.GN" ], "topics": [ "memory", "planning", "tool-use", "workflow-agent", "world-model" ], "score": 19, "relevance": "high", "matched_queries": [ "agentic-ai", "llm-agent" ], "arxiv_id": "2606.30986", "source": "arxiv", "source_id": "arxiv:2606.30986", "pdf_url": "https://arxiv.org/pdf/2606.30986", "primary_query": "agentic-ai" }, { "id": "2606.30639", "title": "Self-Evolving World Models for LLM Agent Planning", "url": "https://arxiv.org/abs/2606.30639", "published": "2026-06-29", "updated": "2026-06-29", "authors": [ "Xuan Zhang", "Wenxuan Zhang", "See-Kiong Ng", "Yang Deng" ], "categories": [ "cs.AI", "cs.CL" ], "topics": [ "agent-evaluation", "memory", "planning", "rag", "reasoning", "tool-use", "world-model" ], "score": 19, "relevance": "high", "matched_queries": [ "llm-agent", "planning-agent" ], "arxiv_id": "2606.30639", "source": "arxiv", "source_id": "arxiv:2606.30639", "pdf_url": "https://arxiv.org/pdf/2606.30639", "primary_query": "llm-agent" }, { "id": "2606.29771", "title": "CLQT: A Closed-Loop, Cost-Aware, Strategy-Consistent Benchmark for Diagnostic Evaluation of LLM Portfolio-Management Agents", "url": "https://arxiv.org/abs/2606.29771", "published": "2026-06-29", "updated": "2026-06-29", "authors": [ "Bo Qu", "Mingguang Chen" ], "categories": [ "cs.AI", "cs.LG", "q-fin.CP", "q-fin.PM" ], "topics": [ "agent-evaluation", "memory", "reasoning", "tool-use" ], "score": 19, "relevance": "high", "matched_queries": [ "llm-agent" ], "arxiv_id": "2606.29771", "source": "arxiv", "source_id": "arxiv:2606.29771", "pdf_url": "https://arxiv.org/pdf/2606.29771", "primary_query": "llm-agent" }, { "id": "2607.00041", "title": "ATM: CID-Brokered Pre-Write Admission for Multi-Agent Code Co-Synthesis", "url": "https://arxiv.org/abs/2607.00041", "published": "2026-06-29", "updated": "2026-06-29", "authors": [ "Eagl Huang" ], "categories": [ "cs.SE", "cs.AI" ], "topics": [ "agent-evaluation", "coding-agent", "computer-use", "multi-agent", "planning", "rag" ], "score": 19, "relevance": "high", "matched_queries": [ "agent-evaluation", "multi-agent-llm" ], "arxiv_id": "2607.00041", "source": "arxiv", "source_id": "arxiv:2607.00041", "pdf_url": "https://arxiv.org/pdf/2607.00041", "primary_query": "agent-evaluation" }, { "id": "2606.29774", "title": "Analytic Concept-Centric Memory for Agentic Embodied Manipulation", "url": "https://arxiv.org/abs/2606.29774", "published": "2026-06-29", "updated": "2026-06-29", "authors": [ "Mingyang Sun", "Xiujian Liang", "Jiude Wei", "Qichen He", "Donglin Wang", "Cewu Lu", "Jianhua Sun" ], "categories": [ "cs.RO" ], "topics": [ "agent-evaluation", "embodied-agent", "memory", "planning", "rag", "reasoning", "tool-use" ], "score": 19, "relevance": "high", "matched_queries": [ "agent-memory" ], "arxiv_id": "2606.29774", "source": "arxiv", "source_id": "arxiv:2606.29774", "pdf_url": "https://arxiv.org/pdf/2606.29774", "primary_query": "agent-memory" }, { "id": "2606.29193", "title": "A Multi-Dataset Benchmark for Evaluating LLM Agents in Microservice Failure Diagnosis", "url": "https://arxiv.org/abs/2606.29193", "published": "2026-06-28", "updated": "2026-06-28", "authors": [ "Yuanhong Cai", "Xiaohui Nie", "Kanglin Yin", "Changhua Pei", "Yongqian Sun", "Shenglin Zhang", "Haibin Liu", "Guiyang Liu", "Xidao Wen", "Fang Situ", "Dan Pei" ], "categories": [ "cs.SE", "cs.AI" ], "topics": [ "agent-evaluation", "rag", "reasoning", "tool-use" ], "score": 19, "relevance": "high", "matched_queries": [ "llm-agent" ], "arxiv_id": "2606.29193", "source": "arxiv", "source_id": "arxiv:2606.29193", "pdf_url": "https://arxiv.org/pdf/2606.29193", "primary_query": "llm-agent" }, { "id": "2606.29030", "title": "Memory as an Attack Surface in LLM Agents: A Study on Multiple-Choice Question Answering", "url": "https://arxiv.org/abs/2606.29030", "published": "2026-06-27", "updated": "2026-06-27", "authors": [ "Shahnewaz Karim Sakib", "Anindya Bijoy Das" ], "categories": [ "cs.AI", "cs.ET" ], "topics": [ "memory", "tool-use" ], "score": 19, "relevance": "high", "matched_queries": [ "ai-agent", "llm-agent", "tool-use" ], "arxiv_id": "2606.29030", "source": "arxiv", "source_id": "arxiv:2606.29030", "pdf_url": "https://arxiv.org/pdf/2606.29030", "primary_query": "ai-agent" }, { "id": "2606.28692", "title": "An AI agent for treatment reasoning over a biomedical tool universe", "url": "https://arxiv.org/abs/2606.28692", "published": "2026-06-27", "updated": "2026-06-27", "authors": [ "Shanghua Gao", "Ayush Noori", "Richard Zhu", "Curtis Ginder", "Zhenglun Kong", "Xiaorui Su", "Justin Kauffman", "Benjamin S. Glicksberg", "Joshua Lampert", "Ankit Sakhuja", "Ashwin Sawant", "ATHENA-R1 Evaluation Consortium", "David A. Clifton", "Noa Dagan", "Ran Balicer", "Marinka Zitnik" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "multi-agent", "reasoning", "tool-use" ], "score": 19, "relevance": "high", "matched_queries": [ "ai-agent", "tool-use" ], "arxiv_id": "2606.28692", "source": "arxiv", "source_id": "arxiv:2606.28692", "pdf_url": "https://arxiv.org/pdf/2606.28692", "primary_query": "ai-agent" }, { "id": "2606.27806", "title": "Agent vs. Parametric World Models: Hybrid Planning for Reliable Language Agents", "url": "https://arxiv.org/abs/2606.27806", "published": "2026-06-26", "updated": "2026-07-05", "authors": [ "Xinyuan Song", "Zekun Cai" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "agent-safety", "planning", "rag", "reasoning", "tool-use", "world-model" ], "score": 19, "relevance": "high", "matched_queries": [ "language-agent" ], "arxiv_id": "2606.27806", "source": "arxiv", "source_id": "arxiv:2606.27806", "pdf_url": "https://arxiv.org/pdf/2606.27806", "primary_query": "language-agent" }, { "id": "2606.28467", "title": "An Agentic AI Pipeline for Appliance-Level Energy Anomaly Detection and LLM-Driven Recommendations", "url": "https://arxiv.org/abs/2606.28467", "published": "2026-06-26", "updated": "2026-06-26", "authors": [ "Dihia Falouz", "Aida Douaibia", "Amine Bechar", "Youssef Elmir", "Abbes Amira", "Adel Oulefki" ], "categories": [ "cs.LG", "cs.AI", "cs.CL" ], "topics": [ "agent-evaluation", "memory", "rag", "reasoning", "tool-use", "workflow-agent" ], "score": 19, "relevance": "high", "matched_queries": [ "agentic-ai", "rag-agent" ], "arxiv_id": "2606.28467", "source": "arxiv", "source_id": "arxiv:2606.28467", "pdf_url": "https://arxiv.org/pdf/2606.28467", "primary_query": "agentic-ai" }, { "id": "2606.26960", "title": "Toward Agentic SysAdmin: Rethinking System Administration with AI Agents", "url": "https://arxiv.org/abs/2606.26960", "published": "2026-06-25", "updated": "2026-06-25", "authors": [ "Gianmaria Frigo", "Davide Saladino", "Alberto Castagnaro", "Francesco Marchiori", "Denis Donadel", "Luca Pajola", "Mauro Conti" ], "categories": [ "cs.NI" ], "topics": [ "agent-evaluation", "agent-safety", "rag", "tool-use" ], "score": 19, "relevance": "high", "matched_queries": [ "ai-agent" ], "arxiv_id": "2606.26960", "source": "arxiv", "source_id": "arxiv:2606.26960", "pdf_url": "https://arxiv.org/pdf/2606.26960", "primary_query": "ai-agent" }, { "id": "2606.26346", "title": "How Do Tool-Augmented LLM Agents Perform on Real-World Energy Analytics Tasks?", "url": "https://arxiv.org/abs/2606.26346", "published": "2026-06-24", "updated": "2026-06-24", "authors": [ "David Akinpelu", "Akintonde Abbas", "Rereloluwa Alimi", "Ayodeji Lana" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "agent-safety", "coding-agent", "rag", "reasoning", "tool-use" ], "score": 19, "relevance": "high", "matched_queries": [ "agent-evaluation" ], "arxiv_id": "2606.26346", "source": "arxiv", "source_id": "arxiv:2606.26346", "pdf_url": "https://arxiv.org/pdf/2606.26346", "primary_query": "agent-evaluation" }, { "id": "2606.26403", "title": "ProfileFoundry: A Synthetic Person-Object Substrate for Privacy, Memory, and Tool-Use Evaluation in LLM Agent", "url": "https://arxiv.org/abs/2606.26403", "published": "2026-06-24", "updated": "2026-06-24", "authors": [ "Sriram Selvam", "Anneswa Ghosh" ], "categories": [ "cs.CL" ], "topics": [ "agent-evaluation", "memory", "tool-use" ], "score": 19, "relevance": "high", "matched_queries": [ "tool-use" ], "arxiv_id": "2606.26403", "source": "arxiv", "source_id": "arxiv:2606.26403", "pdf_url": "https://arxiv.org/pdf/2606.26403", "primary_query": "tool-use" }, { "id": "2606.25161", "title": "TRUSTMEM: Learning Trustworthy Memory Consolidation for LLM Agents with Long-Term Memory", "url": "https://arxiv.org/abs/2606.25161", "published": "2026-06-23", "updated": "2026-06-23", "authors": [ "Tianyu Yang", "Sudipta Paul", "Vijay Srinivasan", "Vivek Kulkarni", "Srinivas Chappidi" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "computer-use", "memory", "rag", "reasoning", "tool-use" ], "score": 19, "relevance": "high", "matched_queries": [ "agent-memory" ], "arxiv_id": "2606.25161", "source": "arxiv", "source_id": "arxiv:2606.25161", "pdf_url": "https://arxiv.org/pdf/2606.25161", "primary_query": "agent-memory" }, { "id": "2606.24626", "title": "SAFARI: Scaling Long Horizon Agentic Fault Attribution via Active Investigation", "url": "https://arxiv.org/abs/2606.24626", "published": "2026-06-23", "updated": "2026-06-23", "authors": [ "Chenyang Zhu", "Jiayu Yao", "Kushal Chawla", "Youbing Yin", "Nathan Wolfe", "Pengshan Cai", "Jingyu Wu", "Spencer Hong", "Sangwoo Cho", "Shi-Xiong Zhang", "Daben Liu", "Sambit Sahu", "Erin Babinsky" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "memory", "multi-agent", "planning", "reasoning", "tool-use" ], "score": 19, "relevance": "high", "matched_queries": [ "autonomous-agent-llm", "multi-agent-llm" ], "arxiv_id": "2606.24626", "source": "arxiv", "source_id": "arxiv:2606.24626", "pdf_url": "https://arxiv.org/pdf/2606.24626", "primary_query": "autonomous-agent-llm" }, { "id": "2606.23991", "title": "Critique of Agent Model", "url": "https://arxiv.org/abs/2606.23991", "published": "2026-06-22", "updated": "2026-06-22", "authors": [ "Eric Xing", "Mingkai Deng", "Jinyu Hou" ], "categories": [ "cs.AI", "cs.LG", "cs.MA", "cs.RO" ], "topics": [ "agent-safety", "coding-agent", "rag", "reasoning", "tool-use", "workflow-agent", "world-model" ], "score": 19, "relevance": "high", "matched_queries": [ "agentic-ai", "ai-agent", "coding-agent" ], "arxiv_id": "2606.23991", "source": "arxiv", "source_id": "arxiv:2606.23991", "pdf_url": "https://arxiv.org/pdf/2606.23991", "primary_query": "agentic-ai" }, { "id": "2606.22844", "title": "RaMem: Contextual Reinstatement for Long-term Agentic Memory", "url": "https://arxiv.org/abs/2606.22844", "published": "2026-06-22", "updated": "2026-06-22", "authors": [ "Wei Yang", "Bryce Kan", "Shixuan Li", "Li Li", "Yuehan Qin", "Jiate Li", "Paul Bogdan", "Jesse Thomason" ], "categories": [ "cs.AI", "cs.MA" ], "topics": [ "agent-evaluation", "memory", "rag", "tool-use" ], "score": 19, "relevance": "high", "matched_queries": [ "agent-memory" ], "arxiv_id": "2606.22844", "source": "arxiv", "source_id": "arxiv:2606.22844", "pdf_url": "https://arxiv.org/pdf/2606.22844", "primary_query": "agent-memory" }, { "id": "2606.23565", "title": "HoloAgent-0: A Unified Embodied Agent Framework with 3D Spatial Memory", "url": "https://arxiv.org/abs/2606.23565", "published": "2026-06-22", "updated": "2026-06-22", "authors": [ "Xiaolin Zhou", "Liu Liu", "Tingyang Xiao", "Wei Feng", "Fa Fu", "Xinrui Meng", "Xinjie Wang", "Jialiang Han", "Boyang Yu", "Yun Du", "Wei Sui", "Zhizhong Su" ], "categories": [ "cs.RO", "cs.CV" ], "topics": [ "agent-evaluation", "agent-safety", "computer-use", "embodied-agent", "memory", "planning", "rag", "tool-use" ], "score": 19, "relevance": "high", "matched_queries": [ "planning-agent" ], "arxiv_id": "2606.23565", "source": "arxiv", "source_id": "arxiv:2606.23565", "pdf_url": "https://arxiv.org/pdf/2606.23565", "primary_query": "planning-agent" }, { "id": "2606.22678", "title": "RigorBench: Benchmarking Engineering Process Discipline in Autonomous AI Coding Agents", "url": "https://arxiv.org/abs/2606.22678", "published": "2026-06-21", "updated": "2026-06-29", "authors": [ "Meher Bhaskar Madiraju", "Meher Sai Preetam Madiraju" ], "categories": [ "cs.SE", "cs.AI" ], "topics": [ "agent-evaluation", "coding-agent", "planning", "rag", "tool-use" ], "score": 19, "relevance": "high", "matched_queries": [ "coding-agent" ], "arxiv_id": "2606.22678", "source": "arxiv", "source_id": "arxiv:2606.22678", "pdf_url": "https://arxiv.org/pdf/2606.22678", "primary_query": "coding-agent" }, { "id": "2606.22417", "title": "Code Isn't Memory: A Structural Codebase Index Inside a Coding Agent", "url": "https://arxiv.org/abs/2606.22417", "published": "2026-06-21", "updated": "2026-06-21", "authors": [ "Ishaan Bhola", "Adithyan Krishnan", "Sravanth Kurmala", "Mukunda NS" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "coding-agent", "memory", "rag" ], "score": 19, "relevance": "high", "matched_queries": [ "coding-agent" ], "arxiv_id": "2606.22417", "source": "arxiv", "source_id": "arxiv:2606.22417", "pdf_url": "https://arxiv.org/pdf/2606.22417", "primary_query": "coding-agent" }, { "id": "2606.21129", "title": "AgenticOS: An Intent-Oriented Secure Operating System Architecture for Autonomous AI Agents", "url": "https://arxiv.org/abs/2606.21129", "published": "2026-06-19", "updated": "2026-06-19", "authors": [ "Zhen Zhao", "Yu Zhang", "Yanpeng Zhu", "Jia Wang", "Songqiao Tao", "Xin Cheng", "Jiexin Gao" ], "categories": [ "cs.CR", "cs.OS" ], "topics": [ "agent-safety", "planning", "tool-use" ], "score": 19, "relevance": "high", "matched_queries": [ "ai-agent", "autonomous-agent-llm", "tool-use" ], "arxiv_id": "2606.21129", "source": "arxiv", "source_id": "arxiv:2606.21129", "pdf_url": "https://arxiv.org/pdf/2606.21129", "primary_query": "ai-agent" }, { "id": "2606.21649", "title": "EvoEmbedding: Evolvable Representations for Long-Context Retrieval and Agentic Memory", "url": "https://arxiv.org/abs/2606.21649", "published": "2026-06-19", "updated": "2026-06-25", "authors": [ "Chang Nie", "Chaoyou Fu", "Junlan Feng", "Caifeng Shan" ], "categories": [ "cs.CL" ], "topics": [ "agent-evaluation", "coding-agent", "memory", "rag", "workflow-agent" ], "score": 19, "relevance": "high", "matched_queries": [ "agent-memory", "agentic-ai", "rag-agent" ], "arxiv_id": "2606.21649", "source": "arxiv", "source_id": "arxiv:2606.21649", "pdf_url": "https://arxiv.org/pdf/2606.21649", "primary_query": "agent-memory" }, { "id": "2606.20950", "title": "Power Systems Agent Benchmark: Executable Evaluation of AI Agents in Electric Power Engineering", "url": "https://arxiv.org/abs/2606.20950", "published": "2026-06-18", "updated": "2026-07-02", "authors": [ "Sergei Trashchenkov" ], "categories": [ "cs.AI", "eess.SY" ], "topics": [ "agent-evaluation", "rag", "tool-use" ], "score": 19, "relevance": "high", "matched_queries": [ "agent-evaluation", "ai-agent", "tool-use" ], "arxiv_id": "2606.20950", "source": "arxiv", "source_id": "arxiv:2606.20950", "pdf_url": "https://arxiv.org/pdf/2606.20950", "primary_query": "agent-evaluation" }, { "id": "2606.19704", "title": "Beyond Static Leaderboards: Predictive Validity for the Evaluation of LLM Agents", "url": "https://arxiv.org/abs/2606.19704", "published": "2026-06-18", "updated": "2026-06-18", "authors": [ "Dhaval C. Patel", "Kaoutar El Maghraoui", "Shuxin Lin", "Yusheng Li", "Tianjun Feng", "Chun-Yi Tsai", "Yihan Sun", "Wei Alexander Xin", "Akshat Bhandari", "Tanisha Rathod", "Aaron Fan", "Sanskruti Vijay Shejwal", "Tomas Pasiecznik", "Sagar Chethan Kumar", "Tanmay Agarwal", "Rohith Kanathur", "Sam Colman", "Amaan Sheikh", "Dev Bahl", "Ann Li", "Krish Veera", "Alimurtaza Mustafa Merchant", "Shambhawi Baswaraj Bhure", "Sajal Kumar Goyla", "Chengrui Li", "Kirthana Natarajan", "Rui Li", "Thomas Ajai", "Rujing Li", "Vivek G. Iyer", "Sanjaii Vijayakumar", "Yitong Bai", "Ayal Yakobe", "Darief Maes", "Yassine Jebbouri", "Tianyang Xu", "Thai Quoc On", "Vera Mazeeva", "Winston Li", "Yuval Shemla", "Yeshitha Bhuvanesh", "Rushin Bhatt", "Siddharth Chethan Gowda", "Alisha Vinod", "Caroline Cahill", "Shriya Aishani Rachakonda", "Yunfeng Chen", "Aryaman Agrawal", "Aman Upganlawar", "Mao Le Jonathan Ang", "Yubin Sally Go", "Madhav Rajkondawar", "Yang-Jung Chen", "Trisha Maturi", "Ananya Kapoor", "Andrew Li", "Shrey Arora", "Mana Abbaszadeh", "Shen Li", "Charles Xu", "Byeolah Kwon" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "rag", "reasoning", "tool-use" ], "score": 19, "relevance": "high", "matched_queries": [ "agent-evaluation" ], "arxiv_id": "2606.19704", "source": "arxiv", "source_id": "arxiv:2606.19704", "pdf_url": "https://arxiv.org/pdf/2606.19704", "primary_query": "agent-evaluation" }, { "id": "2606.20512", "title": "Probe-and-Refine Tuning of Repository Guidance for Coding Agents", "url": "https://arxiv.org/abs/2606.20512", "published": "2026-06-18", "updated": "2026-06-19", "authors": [ "Asa Shepard", "Jeannie Albrecht" ], "categories": [ "cs.SE", "cs.LG" ], "topics": [ "agent-evaluation", "coding-agent", "computer-use", "rag", "tool-use", "workflow-agent" ], "score": 19, "relevance": "high", "matched_queries": [ "coding-agent", "tool-use" ], "arxiv_id": "2606.20512", "source": "arxiv", "source_id": "arxiv:2606.20512", "pdf_url": "https://arxiv.org/pdf/2606.20512", "primary_query": "coding-agent" }, { "id": "2606.18829", "title": "GateMem: Benchmarking Memory Governance in Multi-Principal Shared-Memory Agents", "url": "https://arxiv.org/abs/2606.18829", "published": "2026-06-17", "updated": "2026-06-17", "authors": [ "Zhe Ren", "Yibo Yang", "Yimeng Chen", "Zijun Zhao", "Benshuo Fu", "Zhihao Shu", "Bingjie Zhang", "Yangyang Xu", "Dandan Guo", "Shuicheng Yan" ], "categories": [ "cs.LG", "cs.CL" ], "topics": [ "agent-evaluation", "memory", "planning", "rag", "workflow-agent" ], "score": 19, "relevance": "high", "matched_queries": [ "agent-memory" ], "arxiv_id": "2606.18829", "source": "arxiv", "source_id": "arxiv:2606.18829", "pdf_url": "https://arxiv.org/pdf/2606.18829", "primary_query": "agent-memory" }, { "id": "2606.18356", "title": "SafeClawBench: Separating Semantic, Audit-Evidence, and Sandbox Harm in Tool-Using LLM Agents", "url": "https://arxiv.org/abs/2606.18356", "published": "2026-06-16", "updated": "2026-06-16", "authors": [ "Yuchuan Tian", "Mengyu Zheng", "Haocheng Mei", "Ye Yuan", "Chao Xu", "Xinghao Chen", "Hanting Chen", "Yu Wang" ], "categories": [ "cs.CR", "cs.AI" ], "topics": [ "agent-evaluation", "agent-safety", "computer-use", "memory", "tool-use" ], "score": 19, "relevance": "high", "matched_queries": [ "agent-safety", "tool-use" ], "arxiv_id": "2606.18356", "source": "arxiv", "source_id": "arxiv:2606.18356", "pdf_url": "https://arxiv.org/pdf/2606.18356", "primary_query": "agent-safety" }, { "id": "2606.16774", "title": "OpenClaw-Skill: Collective Skill Tree Search for Agentic Large Language Models", "url": "https://arxiv.org/abs/2606.16774", "published": "2026-06-15", "updated": "2026-06-15", "authors": [ "Tianyi Lin", "Chuanyu Sun", "Jingyi Zhang", "Changxu Wei", "Huanjin Yao", "Shunyu Liu", "Xikun Zhang", "Liu Liu", "Jiaxing Huang" ], "categories": [ "cs.AI", "cs.CL" ], "topics": [ "agent-evaluation", "planning", "rag", "reasoning", "tool-use" ], "score": 19, "relevance": "high", "matched_queries": [ "planning-agent", "tool-use" ], "arxiv_id": "2606.16774", "source": "arxiv", "source_id": "arxiv:2606.16774", "pdf_url": "https://arxiv.org/pdf/2606.16774", "primary_query": "planning-agent" }, { "id": "2606.15862", "title": "RetailBench: Benchmarking long horizon reasoning and coherent decision making of LLM agents in realistic retail environments", "url": "https://arxiv.org/abs/2606.15862", "published": "2026-06-14", "updated": "2026-06-19", "authors": [ "Linghua Zhang", "Jun Wang", "Jingtong Wu", "Zhisong Zhang" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "planning", "reasoning", "tool-use", "world-model" ], "score": 19, "relevance": "high", "matched_queries": [ "tool-use" ], "arxiv_id": "2606.15862", "source": "arxiv", "source_id": "arxiv:2606.15862", "pdf_url": "https://arxiv.org/pdf/2606.15862", "primary_query": "tool-use" }, { "id": "2606.12586", "title": "Beyond Attack Success Rate: Examining Trigger Leakage in Vision-Language Agentic Systems", "url": "https://arxiv.org/abs/2606.12586", "published": "2026-06-10", "updated": "2026-06-10", "authors": [ "Jiamin Chang", "Salil Kanhere", "Piotr Koniusz", "Jason", "Xue", "Hammond Pearce" ], "categories": [ "cs.CR" ], "topics": [ "agent-evaluation", "embodied-agent", "planning", "tool-use", "workflow-agent" ], "score": 19, "relevance": "high", "matched_queries": [ "language-agent", "tool-use" ], "arxiv_id": "2606.12586", "source": "arxiv", "source_id": "arxiv:2606.12586", "pdf_url": "https://arxiv.org/pdf/2606.12586", "primary_query": "language-agent" }, { "id": "2606.10507", "title": "HIPIF: Hierarchical Planning and Information Folding for Long-Horizon LLM Agent Learning", "url": "https://arxiv.org/abs/2606.10507", "published": "2026-06-09", "updated": "2026-06-09", "authors": [ "Juncheng Diao", "Zhicong Lu", "Peiguang Li", "Yongwei Zhou", "Changyuan Tian", "Qingbin Li", "Rongxiang Weng", "Jingang Wang", "Xunliang Cai" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "computer-use", "memory", "planning", "reasoning" ], "score": 19, "relevance": "high", "matched_queries": [ "agent-evaluation", "autonomous-agent-llm", "planning-agent" ], "arxiv_id": "2606.10507", "source": "arxiv", "source_id": "arxiv:2606.10507", "pdf_url": "https://arxiv.org/pdf/2606.10507", "primary_query": "agent-evaluation" }, { "id": "2606.11042", "title": "Workflow-GYM: Towards Long-Horizon Evaluation of Computer-use Agentic tasks in Real-World Professional Fields", "url": "https://arxiv.org/abs/2606.11042", "published": "2026-06-09", "updated": "2026-06-11", "authors": [ "Liya Zhu", "Jingzhe Ding", "Jian Zhang", "Jianbo Xue", "Shihao Liang", "Ge Zhang", "Yi Zhu", "Duju Zeng", "Xiang Gao", "Qingshui Gu", "Mailun Gao", "Huimin Che", "Yan Zhao", "Peiheng Zhou", "Haojun Wang", "Chaobo Xian", "Lili Le", "Chi Wu", "Yiwei Liu", "Shengda Long", "Jiale Yang", "Fangzhi Xu", "Sijin Wu", "Haodong Duan", "Chao He", "Zhaojian Li", "Minchao Wang", "Huan Zhou", "Jiani Hou", "Chuqian Yu", "Weiran Shi", "Hongwan Gao", "Jiamin Chen", "Guanhong Chen", "Tingqin Luo", "Kaiyuan Zhang", "Zhixin Yao", "Qing Hua", "Yuhao Jiang", "Jin Chen", "Pu Chen", "Zhenyu Hu", "Xingyu Li", "Zhengxuan Jiang", "Meng Cao", "Tianfeng Long", "Haozhe Wang", "Mingzhang Wang", "Yichen Zhang", "Yiming Dai", "Chenchen Zhang", "Jiaying Wang", "Xinying Liu", "Xingzu Liu", "Lingling Zhang", "Xinjie Chen", "Yujia Qin", "Wangchunshu Zhou", "Zhiyong Wu", "Yang Liu", "Jiaheng Liu", "Lei Zhang", "Shen Yan", "Wenhao Huang", "Zaiyuan Wang", "Xiaolong Chang" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "computer-use", "planning", "tool-use", "workflow-agent" ], "score": 19, "relevance": "high", "matched_queries": [ "web-gui-agent" ], "arxiv_id": "2606.11042", "source": "arxiv", "source_id": "arxiv:2606.11042", "pdf_url": "https://arxiv.org/pdf/2606.11042", "primary_query": "web-gui-agent" }, { "id": "2606.09483", "title": "Memory Beyond Recall: A Dual-Process Cognitive Memory System for Self-Evolving LLM Agents", "url": "https://arxiv.org/abs/2606.09483", "published": "2026-06-08", "updated": "2026-06-08", "authors": [ "Tianxiang Fei", "Mingyang Song", "Mao Zheng", "Xiang Yu" ], "categories": [ "cs.CL", "cs.AI" ], "topics": [ "agent-evaluation", "memory", "rag", "reasoning", "tool-use" ], "score": 19, "relevance": "high", "matched_queries": [ "agent-memory" ], "arxiv_id": "2606.09483", "source": "arxiv", "source_id": "arxiv:2606.09483", "pdf_url": "https://arxiv.org/pdf/2606.09483", "primary_query": "agent-memory" }, { "id": "2606.07314", "title": "QBugLM: An Agentic Benchmarking Framework for LLM-based Quantum Software Debugging", "url": "https://arxiv.org/abs/2606.07314", "published": "2026-06-05", "updated": "2026-06-05", "authors": [ "An B. B. Pham", "Hoa T. Nguyen", "Muhammad Usman" ], "categories": [ "cs.SE", "cs.ET", "quant-ph" ], "topics": [ "agent-evaluation", "coding-agent", "multi-agent", "reasoning", "world-model" ], "score": 19, "relevance": "high", "matched_queries": [ "agent-evaluation" ], "arxiv_id": "2606.07314", "source": "arxiv", "source_id": "arxiv:2606.07314", "pdf_url": "https://arxiv.org/pdf/2606.07314", "primary_query": "agent-evaluation" }, { "id": "2606.05684", "title": "AdaMEM: Test-Time Adaptive Memory for Language Agents", "url": "https://arxiv.org/abs/2606.05684", "published": "2026-06-04", "updated": "2026-06-04", "authors": [ "Yunxiang Zhang", "Yiheng Li", "Ali Payani", "Lu Wang" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "computer-use", "memory", "planning", "rag", "reasoning" ], "score": 19, "relevance": "high", "matched_queries": [ "agent-memory", "language-agent" ], "arxiv_id": "2606.05684", "source": "arxiv", "source_id": "arxiv:2606.05684", "pdf_url": "https://arxiv.org/pdf/2606.05684", "primary_query": "agent-memory" }, { "id": "2606.06448", "title": "Agent Memory: Characterization and System Implications of Stateful Long-Horizon Workloads", "url": "https://arxiv.org/abs/2606.06448", "published": "2026-06-04", "updated": "2026-06-04", "authors": [ "Yasmine Omri", "Ziyu Gan", "Zachary Broveak", "Robin Geens", "Zexue He", "Alex Pentland", "Marian Verhelst", "Tsachy Weissman", "Thierry Tambe" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "memory", "planning", "rag", "reasoning", "tool-use" ], "score": 19, "relevance": "high", "matched_queries": [ "agent-memory" ], "arxiv_id": "2606.06448", "source": "arxiv", "source_id": "arxiv:2606.06448", "pdf_url": "https://arxiv.org/pdf/2606.06448", "primary_query": "agent-memory" }, { "id": "2606.04874", "title": "Agent Planning Benchmark: A Diagnostic Framework for Planning Capabilities in LLM Agents", "url": "https://arxiv.org/abs/2606.04874", "published": "2026-06-03", "updated": "2026-06-05", "authors": [ "Haoyu Sun", "Wenxuan Wang", "Mingyang Song", "Jujie He", "Weinan Zhang", "Yang Liu", "Yang Yang", "Yu Cheng" ], "categories": [ "cs.CL" ], "topics": [ "agent-evaluation", "computer-use", "planning", "tool-use" ], "score": 19, "relevance": "high", "matched_queries": [ "agent-evaluation", "planning-agent" ], "arxiv_id": "2606.04874", "source": "arxiv", "source_id": "arxiv:2606.04874", "pdf_url": "https://arxiv.org/pdf/2606.04874", "primary_query": "agent-evaluation" }, { "id": "2606.28349", "title": "HMARS: A Hierarchical Multi-Agent Memory System for Long-Context Reasoning", "url": "https://arxiv.org/abs/2606.28349", "published": "2026-06-03", "updated": "2026-06-03", "authors": [ "Zeju Li", "Ziyang Zheng", "Yizhou Zhou", "Qiang Xu" ], "categories": [ "cs.IR", "cs.AI" ], "topics": [ "agent-evaluation", "memory", "multi-agent", "rag", "reasoning", "tool-use" ], "score": 19, "relevance": "high", "matched_queries": [ "agent-memory" ], "arxiv_id": "2606.28349", "source": "arxiv", "source_id": "arxiv:2606.28349", "pdf_url": "https://arxiv.org/pdf/2606.28349", "primary_query": "agent-memory" }, { "id": "2606.04315", "title": "Exploring Cross-Scenario Generality of Agentic Memory Systems: Diagnostics and a Strong Baseline", "url": "https://arxiv.org/abs/2606.04315", "published": "2026-06-03", "updated": "2026-06-03", "authors": [ "Zhikai Chen", "Jialiang Gu", "Junyu Yin", "Xianxuan Long", "Shenglai Zeng", "Xiaoze Liu", "Kai Guo", "Keren Zhou", "Jiliang Tang" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "memory", "planning", "rag", "tool-use" ], "score": 19, "relevance": "high", "matched_queries": [ "agent-memory" ], "arxiv_id": "2606.04315", "source": "arxiv", "source_id": "arxiv:2606.04315", "pdf_url": "https://arxiv.org/pdf/2606.04315", "primary_query": "agent-memory" }, { "id": "2606.03374", "title": "eMEM: A Hybrid Spatio-Temporal Memory System For Embodied Agents", "url": "https://arxiv.org/abs/2606.03374", "published": "2026-06-02", "updated": "2026-06-22", "authors": [ "A. Haroon Rasheed", "Maria Kabtoul" ], "categories": [ "cs.RO" ], "topics": [ "agent-evaluation", "embodied-agent", "memory", "planning", "rag", "tool-use" ], "score": 19, "relevance": "high", "matched_queries": [ "agent-memory", "rag-agent" ], "arxiv_id": "2606.03374", "source": "arxiv", "source_id": "arxiv:2606.03374", "pdf_url": "https://arxiv.org/pdf/2606.03374", "primary_query": "agent-memory" }, { "id": "2606.02372", "title": "COMAP: Co-Evolving World Models and Agent Policies for LLM Agents", "url": "https://arxiv.org/abs/2606.02372", "published": "2026-06-01", "updated": "2026-06-01", "authors": [ "Youwei Liu", "Jian Wang", "Hanlin Wang", "Wenjie Li" ], "categories": [ "cs.AI", "cs.CL" ], "topics": [ "agent-evaluation", "embodied-agent", "planning", "reasoning", "tool-use", "world-model" ], "score": 19, "relevance": "high", "matched_queries": [ "language-agent", "planning-agent" ], "arxiv_id": "2606.02372", "source": "arxiv", "source_id": "arxiv:2606.02372", "pdf_url": "https://arxiv.org/pdf/2606.02372", "primary_query": "language-agent" }, { "id": "2606.01613", "title": "TechRAG: Evidence-Gated Multimodal Agentic RAG for Technical Literature Reasoning", "url": "https://arxiv.org/abs/2606.01613", "published": "2026-06-01", "updated": "2026-06-13", "authors": [ "Kanwar Bharat Singh" ], "categories": [ "cs.IR", "cs.AI", "cs.MA" ], "topics": [ "agent-evaluation", "coding-agent", "computer-use", "multi-agent", "planning", "rag", "reasoning", "tool-use" ], "score": 19, "relevance": "high", "matched_queries": [ "rag-agent" ], "arxiv_id": "2606.01613", "source": "arxiv", "source_id": "arxiv:2606.01613", "pdf_url": "https://arxiv.org/pdf/2606.01613", "primary_query": "rag-agent" }, { "id": "2606.00939", "title": "FinCom: A Financial Multi-Agent Demo with Disagree-or-Commit Deliberation", "url": "https://arxiv.org/abs/2606.00939", "published": "2026-05-31", "updated": "2026-05-31", "authors": [ "Chao Peter Yang", "Zixiao Tan", "Kaisen Yao", "Ziyu Zhou", "Eleanor Jiang", "Michael Wu" ], "categories": [ "cs.MA" ], "topics": [ "agent-evaluation", "agent-safety", "multi-agent", "rag", "reasoning", "tool-use" ], "score": 19, "relevance": "high", "matched_queries": [ "agent-evaluation" ], "arxiv_id": "2606.00939", "source": "arxiv", "source_id": "arxiv:2606.00939", "pdf_url": "https://arxiv.org/pdf/2606.00939", "primary_query": "agent-evaluation" }, { "id": "2605.30690", "title": "ElasticMem: Latent Memory as a Learnable Resource for LLM Agents", "url": "https://arxiv.org/abs/2605.30690", "published": "2026-05-29", "updated": "2026-05-29", "authors": [ "Tao Feng", "Chongrui Ye", "Tianyang Luo", "Jingjun Xu", "Xueqiang Xu", "Haozhen Zhang", "Ge Liu", "Jiaxuan You" ], "categories": [ "cs.CL" ], "topics": [ "agent-evaluation", "embodied-agent", "memory", "planning", "rag", "tool-use" ], "score": 19, "relevance": "high", "matched_queries": [ "planning-agent" ], "arxiv_id": "2605.30690", "source": "arxiv", "source_id": "arxiv:2605.30690", "pdf_url": "https://arxiv.org/pdf/2605.30690", "primary_query": "planning-agent" }, { "id": "2606.20629", "title": "Specialize Roles, Mix Deployments: Pushing the Cost-Accuracy Frontier of LLM Agent Teams", "url": "https://arxiv.org/abs/2606.20629", "published": "2026-05-28", "updated": "2026-05-28", "authors": [ "Yinsicheng Jiang", "Liang Cheng", "Yeqi Huang", "Yufan Zhao", "Zhan Lu", "Li Dong", "Wenda Li", "Edoardo Ponti", "Luo Mai" ], "categories": [ "cs.MA", "cs.AI", "cs.LG" ], "topics": [ "agent-evaluation", "computer-use", "planning", "reasoning", "tool-use", "workflow-agent" ], "score": 19, "relevance": "high", "matched_queries": [ "agent-evaluation" ], "arxiv_id": "2606.20629", "source": "arxiv", "source_id": "arxiv:2606.20629", "pdf_url": "https://arxiv.org/pdf/2606.20629", "primary_query": "agent-evaluation" }, { "id": "2605.30604", "title": "An Organization-Scoped LLM Agent Runtime Architecture for Regulated Cybersecurity Operations", "url": "https://arxiv.org/abs/2605.30604", "published": "2026-05-28", "updated": "2026-05-28", "authors": [ "George Fatouros", "Georgios Makridis", "George Kousiouris", "John Soldatos", "Dimosthenis Kyriazis" ], "categories": [ "cs.CR", "cs.AI", "cs.CL", "cs.IR" ], "topics": [ "agent-evaluation", "agent-safety", "memory", "planning", "rag", "tool-use", "workflow-agent" ], "score": 19, "relevance": "high", "matched_queries": [ "planning-agent" ], "arxiv_id": "2605.30604", "source": "arxiv", "source_id": "arxiv:2605.30604", "pdf_url": "https://arxiv.org/pdf/2605.30604", "primary_query": "planning-agent" }, { "id": "2605.27240", "title": "ENPMR-Bench: Benchmarking Proactive Memory Retrieval for Emotional Support Agents", "url": "https://arxiv.org/abs/2605.27240", "published": "2026-05-26", "updated": "2026-05-26", "authors": [ "Xing Fu", "Yulin Hu", "Mengtong Ji", "Haozhen Li", "Yixin Sun", "Weixiang Zhao", "Yanyan Zhao", "Bing Qin" ], "categories": [ "cs.CL" ], "topics": [ "agent-evaluation", "agent-safety", "memory", "rag", "tool-use" ], "score": 19, "relevance": "high", "matched_queries": [ "language-agent" ], "arxiv_id": "2605.27240", "source": "arxiv", "source_id": "arxiv:2605.27240", "pdf_url": "https://arxiv.org/pdf/2605.27240", "primary_query": "language-agent" }, { "id": "2605.25200", "title": "GroupTravelBench: Benchmarking LLM Agents on Multi-Person Travel Planning", "url": "https://arxiv.org/abs/2605.25200", "published": "2026-05-24", "updated": "2026-06-03", "authors": [ "Xiang Cheng", "Yulan Hu", "Lulu Zheng", "Zheng Pan", "Xin Li", "Yong Liu" ], "categories": [ "cs.CL" ], "topics": [ "agent-evaluation", "planning", "reasoning", "tool-use" ], "score": 19, "relevance": "high", "matched_queries": [ "planning-agent" ], "arxiv_id": "2605.25200", "source": "arxiv", "source_id": "arxiv:2605.25200", "pdf_url": "https://arxiv.org/pdf/2605.25200", "primary_query": "planning-agent" }, { "id": "2605.24220", "title": "Polar: Agentic RL on Any Harness at Scale", "url": "https://arxiv.org/abs/2605.24220", "published": "2026-05-22", "updated": "2026-05-22", "authors": [ "Binfeng Xu", "Hao Zhang", "Shaokun Zhang", "Songyang Han", "Mingjie Liu", "Jian Hu", "Shizhe Diao", "Zhenghui Jin", "Yunheng Zou", "Michael Demoret", "Jan Kautz", "Yi Dong" ], "categories": [ "cs.DC" ], "topics": [ "agent-evaluation", "coding-agent", "multi-agent", "tool-use" ], "score": 19, "relevance": "high", "matched_queries": [ "language-agent" ], "arxiv_id": "2605.24220", "source": "arxiv", "source_id": "arxiv:2605.24220", "pdf_url": "https://arxiv.org/pdf/2605.24220", "primary_query": "language-agent" }, { "id": "2605.23574", "title": "Push Your Agent: Measuring and Enforcing Quantitative Goal Persistence in Long-Horizon LLM Agents", "url": "https://arxiv.org/abs/2605.23574", "published": "2026-05-22", "updated": "2026-05-22", "authors": [ "Yuandao Cai", "Yuzhang Zhu", "Liyou Gao", "Wensheng Tang", "Shengchao Qin" ], "categories": [ "cs.LG", "cs.SE" ], "topics": [ "agent-evaluation", "coding-agent", "planning", "rag", "reasoning", "tool-use" ], "score": 19, "relevance": "high", "matched_queries": [ "language-agent" ], "arxiv_id": "2605.23574", "source": "arxiv", "source_id": "arxiv:2605.23574", "pdf_url": "https://arxiv.org/pdf/2605.23574", "primary_query": "language-agent" }, { "id": "2605.24069", "title": "When the Manual Lies: A Realistic Benchmark to Evaluate MCP Poisoning Attacks for LLM Agents", "url": "https://arxiv.org/abs/2605.24069", "published": "2026-05-22", "updated": "2026-05-22", "authors": [ "Shi Liu", "Xuehai Tang", "Xikang Yang", "Liang Lin", "Biyu Zhou", "Wenjie Xiao", "Wantao Liu" ], "categories": [ "cs.CR", "cs.AI" ], "topics": [ "agent-evaluation", "agent-safety", "planning", "tool-use" ], "score": 19, "relevance": "high", "matched_queries": [ "planning-agent" ], "arxiv_id": "2605.24069", "source": "arxiv", "source_id": "arxiv:2605.24069", "pdf_url": "https://arxiv.org/pdf/2605.24069", "primary_query": "planning-agent" }, { "id": "2605.24216", "title": "Agent-ToM: Learning to Monitor Autonomous LLM Agents via Theory-of-Mind Reasoning", "url": "https://arxiv.org/abs/2605.24216", "published": "2026-05-22", "updated": "2026-05-22", "authors": [ "Nesreen K. Ahmed", "Nima Nafisi" ], "categories": [ "cs.LG", "cs.AI", "cs.CL", "cs.CR" ], "topics": [ "agent-evaluation", "agent-safety", "computer-use", "memory", "planning", "reasoning", "tool-use" ], "score": 19, "relevance": "high", "matched_queries": [ "autonomous-agent-llm" ], "arxiv_id": "2605.24216", "source": "arxiv", "source_id": "arxiv:2605.24216", "pdf_url": "https://arxiv.org/pdf/2605.24216", "primary_query": "autonomous-agent-llm" }, { "id": "2605.18421", "title": "EvoMemBench: Benchmarking Agent Memory from a Self-Evolving Perspective", "url": "https://arxiv.org/abs/2605.18421", "published": "2026-05-18", "updated": "2026-06-15", "authors": [ "Yuyao Wang", "Zhongjian Zhang", "Mo Chi", "Kaichi Yu", "Yuhan Li", "Miao Peng", "Bing Tong", "Chen Zhang", "Yan Zhou", "Jia Li" ], "categories": [ "cs.CL", "cs.AI", "cs.LG" ], "topics": [ "agent-evaluation", "memory", "planning", "rag", "reasoning" ], "score": 19, "relevance": "high", "matched_queries": [ "agent-memory", "planning-agent" ], "arxiv_id": "2605.18421", "source": "arxiv", "source_id": "arxiv:2605.18421", "pdf_url": "https://arxiv.org/pdf/2605.18421", "primary_query": "agent-memory" }, { "id": "2605.10779", "title": "LITMUS: Benchmarking Behavioral Jailbreaks of LLM Agents in Real OS Environments", "url": "https://arxiv.org/abs/2605.10779", "published": "2026-05-11", "updated": "2026-05-11", "authors": [ "Chiyu Zhang", "Huiqin Yang", "Bendong Jiang", "Xiaolei Zhang", "Yiran Zhao", "Ruyi Chen", "Lu Zhou", "Xiaogang Xu", "Jiafei Wu", "Liming Fang", "Zhe Liu" ], "categories": [ "cs.CR", "cs.CL" ], "topics": [ "agent-evaluation", "agent-safety", "multi-agent", "tool-use" ], "score": 19, "relevance": "high", "matched_queries": [ "autonomous-agent-llm" ], "arxiv_id": "2605.10779", "source": "arxiv", "source_id": "arxiv:2605.10779", "pdf_url": "https://arxiv.org/pdf/2605.10779", "primary_query": "autonomous-agent-llm" }, { "id": "2605.03312", "title": "MemFlow: Intent-Driven Memory Orchestration for Small Language Model Agents", "url": "https://arxiv.org/abs/2605.03312", "published": "2026-05-05", "updated": "2026-05-05", "authors": [ "Jiayi Chen", "Yingcong Li", "Guiling Wang" ], "categories": [ "cs.MA" ], "topics": [ "agent-evaluation", "memory", "planning", "rag", "reasoning", "tool-use" ], "score": 19, "relevance": "high", "matched_queries": [ "language-agent" ], "arxiv_id": "2605.03312", "source": "arxiv", "source_id": "arxiv:2605.03312", "pdf_url": "https://arxiv.org/pdf/2605.03312", "primary_query": "language-agent" }, { "id": "2605.01101", "title": "Virtual Speech Therapist: A Clinician-in-the-Loop AI Speech Therapy Agent for Personalized and Supervised Therapy", "url": "https://arxiv.org/abs/2605.01101", "published": "2026-05-01", "updated": "2026-06-15", "authors": [ "Shakeel Sheikh", "Patrick Marmaroli", "MD Sahidullah", "Slim Ouni", "Fabrice Hirsch", "Goncalo Leal", "Bjorn W Schuller" ], "categories": [ "cs.AI", "cs.CL", "cs.SD", "eess.AS" ], "topics": [ "agent-evaluation", "agent-safety", "computer-use", "multi-agent", "planning", "reasoning", "tool-use", "workflow-agent" ], "score": 19, "relevance": "high", "matched_queries": [ "planning-agent" ], "arxiv_id": "2605.01101", "source": "arxiv", "source_id": "arxiv:2605.01101", "pdf_url": "https://arxiv.org/pdf/2605.01101", "primary_query": "planning-agent" }, { "id": "2604.19844", "title": "If you're waiting for a sign... that might not be it! Mitigating Trust Boundary Confusion from Visual Injections on Vision-Language Agentic Systems", "url": "https://arxiv.org/abs/2604.19844", "published": "2026-04-21", "updated": "2026-04-21", "authors": [ "Jiamin Chang", "Minhui Xue", "Ruoxi Sun", "Shuchao Pang", "Salil S. Kanhere", "Hammond Pearce" ], "categories": [ "cs.CV", "cs.AI" ], "topics": [ "agent-evaluation", "agent-safety", "embodied-agent", "multi-agent" ], "score": 19, "relevance": "high", "matched_queries": [ "language-agent" ], "arxiv_id": "2604.19844", "source": "arxiv", "source_id": "arxiv:2604.19844", "pdf_url": "https://arxiv.org/pdf/2604.19844", "primary_query": "language-agent" }, { "id": "2604.18658", "title": "Owner-Harm: A Missing Threat Model for AI Agent Safety", "url": "https://arxiv.org/abs/2604.18658", "published": "2026-04-20", "updated": "2026-04-20", "authors": [ "Dongcheng Zhang", "Yiqing Jiang" ], "categories": [ "cs.CR", "cs.AI", "cs.CL" ], "topics": [ "agent-evaluation", "agent-safety", "rag", "reasoning", "tool-use" ], "score": 19, "relevance": "high", "matched_queries": [ "agent-safety" ], "arxiv_id": "2604.18658", "source": "arxiv", "source_id": "arxiv:2604.18658", "pdf_url": "https://arxiv.org/pdf/2604.18658", "primary_query": "agent-safety" }, { "id": "2604.17562", "title": "SafeAgent: A Runtime Protection Architecture for Agentic Systems", "url": "https://arxiv.org/abs/2604.17562", "published": "2026-04-19", "updated": "2026-04-19", "authors": [ "Hailin Liu", "Eugene Ilyushin", "Jie Ni", "Min Zhu" ], "categories": [ "cs.AI", "cs.MA" ], "topics": [ "agent-evaluation", "agent-safety", "coding-agent", "memory", "reasoning", "tool-use", "workflow-agent" ], "score": 19, "relevance": "high", "matched_queries": [ "agent-safety" ], "arxiv_id": "2604.17562", "source": "arxiv", "source_id": "arxiv:2604.17562", "pdf_url": "https://arxiv.org/pdf/2604.17562", "primary_query": "agent-safety" }, { "id": "2603.00623", "title": "TraceSIR: A Multi-Agent Framework for Structured Analysis and Reporting of Agentic Execution Traces", "url": "https://arxiv.org/abs/2603.00623", "published": "2026-02-28", "updated": "2026-02-28", "authors": [ "Shu-Xun Yang", "Cunxiang Wang", "Haoke Zhang", "Wenbo Yu", "Lindong Wu", "Jiayi Gui", "Dayong Yang", "Yukuo Cen", "Zhuoer Feng", "Bosi Wen", "Yidong Wang", "Lucen Zhong", "Jiamin Ren", "Linfeng Zhang", "Jie Tang" ], "categories": [ "cs.AI", "cs.CL" ], "topics": [ "agent-evaluation", "coding-agent", "multi-agent", "reasoning", "tool-use" ], "score": 19, "relevance": "high", "matched_queries": [ "function-calling" ], "arxiv_id": "2603.00623", "source": "arxiv", "source_id": "arxiv:2603.00623", "pdf_url": "https://arxiv.org/pdf/2603.00623", "primary_query": "function-calling" }, { "id": "2602.13530", "title": "REMem: Reasoning with Episodic Memory in Language Agent", "url": "https://arxiv.org/abs/2602.13530", "published": "2026-02-13", "updated": "2026-02-28", "authors": [ "Yiheng Shu", "Saisri Padmaja Jonnalagedda", "Xiang Gao", "Bernal Jiménez Gutiérrez", "Weijian Qi", "Kamalika Das", "Huan Sun", "Yu Su" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "memory", "rag", "reasoning", "tool-use" ], "score": 19, "relevance": "high", "matched_queries": [ "language-agent" ], "arxiv_id": "2602.13530", "source": "arxiv", "source_id": "arxiv:2602.13530", "pdf_url": "https://arxiv.org/pdf/2602.13530", "primary_query": "language-agent" }, { "id": "2510.03847", "title": "Small Language Models for Agentic Systems: A Survey of Architectures, Capabilities, and Deployment Trade offs", "url": "https://arxiv.org/abs/2510.03847", "published": "2025-10-04", "updated": "2025-10-04", "authors": [ "Raghav Sharma", "Manan Mehta" ], "categories": [ "cs.AI", "cs.LG" ], "topics": [ "agent-evaluation", "coding-agent", "computer-use", "planning", "rag", "reasoning", "tool-use" ], "score": 19, "relevance": "high", "matched_queries": [ "function-calling" ], "arxiv_id": "2510.03847", "source": "arxiv", "source_id": "arxiv:2510.03847", "pdf_url": "https://arxiv.org/pdf/2510.03847", "primary_query": "function-calling" }, { "id": "2607.06157", "title": "LLM Agents for Deliberative Collaboration: A Study on Joint Decision Making Under Partial Observability", "url": "https://arxiv.org/abs/2607.06157", "published": "2026-07-07", "updated": "2026-07-07", "authors": [ "Chenxu Wang", "Yongkun Yang", "Boyuan Du", "Shiwei Lin", "Huaping Liu" ], "categories": [ "cs.CL", "cs.AI" ], "topics": [ "agent-evaluation", "multi-agent", "reasoning", "tool-use" ], "score": 18, "relevance": "high", "matched_queries": [ "llm-agent", "multi-agent-llm" ], "arxiv_id": "2607.06157", "source": "arxiv", "source_id": "arxiv:2607.06157", "pdf_url": "https://arxiv.org/pdf/2607.06157", "primary_query": "llm-agent" }, { "id": "2607.04686", "title": "ToolFailBench: Diagnosing Tool-Use Failures in LLM Agents", "url": "https://arxiv.org/abs/2607.04686", "published": "2026-07-06", "updated": "2026-07-06", "authors": [ "Harsh Soni" ], "categories": [ "cs.CL", "cs.AI", "cs.SE" ], "topics": [ "agent-evaluation", "agent-safety", "tool-use" ], "score": 18, "relevance": "high", "matched_queries": [ "agentic-ai", "llm-agent", "tool-use" ], "arxiv_id": "2607.04686", "source": "arxiv", "source_id": "arxiv:2607.04686", "pdf_url": "https://arxiv.org/pdf/2607.04686", "primary_query": "agentic-ai" }, { "id": "2607.04240", "title": "Biological Motifs for Agentic Control", "url": "https://arxiv.org/abs/2607.04240", "published": "2026-07-05", "updated": "2026-07-05", "authors": [ "Bogdan Banu" ], "categories": [ "cs.AI", "q-bio.CB" ], "topics": [ "agent-evaluation", "agent-safety", "multi-agent", "tool-use" ], "score": 18, "relevance": "high", "matched_queries": [ "agent-evaluation", "autonomous-agent-llm", "multi-agent-llm" ], "arxiv_id": "2607.04240", "source": "arxiv", "source_id": "arxiv:2607.04240", "pdf_url": "https://arxiv.org/pdf/2607.04240", "primary_query": "agent-evaluation" }, { "id": "2607.03441", "title": "No Time Like the Present: Agentic Test-Time Training for LLM Agents", "url": "https://arxiv.org/abs/2607.03441", "published": "2026-07-03", "updated": "2026-07-03", "authors": [ "Yanbo Wang", "Jinhua Hao", "Yuze Shi", "Kun Yuan", "Ming Sun" ], "categories": [ "cs.LG", "cs.AI" ], "topics": [ "coding-agent", "computer-use", "tool-use" ], "score": 18, "relevance": "high", "matched_queries": [ "coding-agent", "llm-agent" ], "arxiv_id": "2607.03441", "source": "arxiv", "source_id": "arxiv:2607.03441", "pdf_url": "https://arxiv.org/pdf/2607.03441", "primary_query": "coding-agent" }, { "id": "2607.01874", "title": "SkillCoach: Self-Evolving Rubrics for Evaluating and Enhancing Agentic Skill-Use", "url": "https://arxiv.org/abs/2607.01874", "published": "2026-07-02", "updated": "2026-07-02", "authors": [ "Jiayin Zhu", "Kelong Mao", "Yudong Guo", "Dengbo He", "Sulong Xu", "Simiu Gu", "Yutao Yue" ], "categories": [ "cs.AI", "cs.CL" ], "topics": [ "agent-evaluation", "coding-agent", "computer-use", "reasoning", "tool-use", "workflow-agent" ], "score": 18, "relevance": "high", "matched_queries": [ "llm-agent" ], "arxiv_id": "2607.01874", "source": "arxiv", "source_id": "arxiv:2607.01874", "pdf_url": "https://arxiv.org/pdf/2607.01874", "primary_query": "llm-agent" }, { "id": "2607.01793", "title": "Safety Testing LLM Agents at Scale: From Risk Discovery to Evidence-Grounded Verification", "url": "https://arxiv.org/abs/2607.01793", "published": "2026-07-02", "updated": "2026-07-04", "authors": [ "Yunhao Feng", "Ruixiao Lin", "Ming Wen", "Qinqin He", "Yanming Guo", "Yifan Ding", "Yutao Wu", "Jialuo Chen", "Zhuoer Xu", "Xiaohu Du", "Jianan Ma", "Zixing Chen", "Xingjun Ma", "Yunhao Chen", "Xinhao Deng" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "agent-safety", "coding-agent", "rag", "reasoning", "tool-use" ], "score": 18, "relevance": "high", "matched_queries": [ "llm-agent" ], "arxiv_id": "2607.01793", "source": "arxiv", "source_id": "arxiv:2607.01793", "pdf_url": "https://arxiv.org/pdf/2607.01793", "primary_query": "llm-agent" }, { "id": "2607.02703", "title": "LLMoxie: Exploring Agentic AI for Scientific Software Development", "url": "https://arxiv.org/abs/2607.02703", "published": "2026-07-02", "updated": "2026-07-02", "authors": [ "Landung Setiawan", "Anant Mittal", "Cordero Core", "Anshul Tambay", "Carlos Garcia Jurado Suarez", "David A. C. Beck", "Andrew J. Connolly", "Vani Mandava" ], "categories": [ "cs.SE", "cs.AI", "cs.DC", "cs.MA" ], "topics": [ "agent-evaluation", "coding-agent", "planning", "reasoning", "workflow-agent" ], "score": 18, "relevance": "high", "matched_queries": [ "agentic-ai", "coding-agent" ], "arxiv_id": "2607.02703", "source": "arxiv", "source_id": "arxiv:2607.02703", "pdf_url": "https://arxiv.org/pdf/2607.02703", "primary_query": "agentic-ai" }, { "id": "2607.00627", "title": "AGI Maze as a Benchmark Framework for World-Modeling Agents", "url": "https://arxiv.org/abs/2607.00627", "published": "2026-07-01", "updated": "2026-07-01", "authors": [ "Alexey Potapov" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "memory", "reasoning", "tool-use" ], "score": 18, "relevance": "high", "matched_queries": [ "llm-agent" ], "arxiv_id": "2607.00627", "source": "arxiv", "source_id": "arxiv:2607.00627", "pdf_url": "https://arxiv.org/pdf/2607.00627", "primary_query": "llm-agent" }, { "id": "2607.02606", "title": "ChainSWE: Benchmarking Coding Agents on Multi-Bug Software Maintenance", "url": "https://arxiv.org/abs/2607.02606", "published": "2026-07-01", "updated": "2026-07-01", "authors": [ "Qirui Jin", "Lingching Tung", "Kenan Li", "Qiyang Shi", "Yushi She", "Huanzhong Jia", "Harrison Zhao", "Kejing Xia", "Zhenbang Du", "Yikai Zhang", "Jiaxin Pei", "Zhenyu Zhang", "Zhen Qi", "Yuyan Duan", "Wenke Lee", "Zijian Jin" ], "categories": [ "cs.SE" ], "topics": [ "agent-evaluation", "coding-agent", "workflow-agent" ], "score": 18, "relevance": "high", "matched_queries": [ "coding-agent" ], "arxiv_id": "2607.02606", "source": "arxiv", "source_id": "arxiv:2607.02606", "pdf_url": "https://arxiv.org/pdf/2607.02606", "primary_query": "coding-agent" }, { "id": "2606.32025", "title": "Generative Skill Composition for LLM Agents", "url": "https://arxiv.org/abs/2606.32025", "published": "2026-06-30", "updated": "2026-06-30", "authors": [ "Xinyu Zhao", "Zhen Tan", "Vaishnav Tadiparthi", "Nakul Agarwal", "Kwonjoon Lee", "Ehsan Moradi Pari", "Hossein Nourkhiz Mahjoub", "Tianlong Chen" ], "categories": [ "cs.CL" ], "topics": [ "agent-evaluation", "coding-agent", "planning", "rag", "reasoning" ], "score": 18, "relevance": "high", "matched_queries": [ "coding-agent", "llm-agent", "planning-agent" ], "arxiv_id": "2606.32025", "source": "arxiv", "source_id": "arxiv:2606.32025", "pdf_url": "https://arxiv.org/pdf/2606.32025", "primary_query": "coding-agent" }, { "id": "2606.31229", "title": "Agentic-Ideation: Sample Efficient Agentic Trajectories Synthesis for Scientific Ideation Agents", "url": "https://arxiv.org/abs/2606.31229", "published": "2026-06-30", "updated": "2026-06-30", "authors": [ "Keyu Zhao", "Lingyan Kong", "Fengli Xu", "Yong Li" ], "categories": [ "cs.AI" ], "topics": [ "computer-use", "multi-agent", "rag", "reasoning", "tool-use", "workflow-agent" ], "score": 18, "relevance": "high", "matched_queries": [ "agentic-ai", "multi-agent-llm" ], "arxiv_id": "2606.31229", "source": "arxiv", "source_id": "arxiv:2606.31229", "pdf_url": "https://arxiv.org/pdf/2606.31229", "primary_query": "agentic-ai" }, { "id": "2606.31410", "title": "Xiaomi-GUI-0 Technical Report", "url": "https://arxiv.org/abs/2606.31410", "published": "2026-06-30", "updated": "2026-07-01", "authors": [ "Wanxia Cao", "Chengzhen Duan", "Pei Fu", "Pengzhi Gao", "Niu Lian", "Fazhan Liu", "Hui Liu", "Heng Qu", "Qinzhuo Wu", "Zhehao Yu", "Tongbo Chen", "Shiqi Cui", "Anan Du", "Shukai Jia", "Yuanfa Li", "Wei Liu", "Yike Liu", "Wenchao Lu", "Zhenbo Luo", "Haoyuan Sun", "Jiatong Sun", "Cheng Tan", "Yajie Wang", "Changqiao Wu", "Tao Xiong", "Jiahui Yang", "Yuxuan Yuan", "Ruoceng Zhang", "Shaojie Zhang", "Jian Zhu", "Jian Luan", "Cong Zou" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "agent-safety", "computer-use", "embodied-agent", "memory", "planning", "reasoning", "tool-use" ], "score": 18, "relevance": "high", "matched_queries": [ "web-gui-agent" ], "arxiv_id": "2606.31410", "source": "arxiv", "source_id": "arxiv:2606.31410", "pdf_url": "https://arxiv.org/pdf/2606.31410", "primary_query": "web-gui-agent" }, { "id": "2606.29178", "title": "Selective Memory Retention for Long-Horizon LLM Agents", "url": "https://arxiv.org/abs/2606.29178", "published": "2026-06-28", "updated": "2026-06-28", "authors": [ "Pranath Reddy" ], "categories": [ "cs.AI", "cs.CL", "cs.LG" ], "topics": [ "agent-evaluation", "memory", "planning" ], "score": 18, "relevance": "high", "matched_queries": [ "llm-agent" ], "arxiv_id": "2606.29178", "source": "arxiv", "source_id": "arxiv:2606.29178", "pdf_url": "https://arxiv.org/pdf/2606.29178", "primary_query": "llm-agent" }, { "id": "2606.28456", "title": "Is Lying an Emergent Behaviour in LLMs? Evidence from Gaslighting AI agents in a Sustainability Game", "url": "https://arxiv.org/abs/2606.28456", "published": "2026-06-26", "updated": "2026-06-26", "authors": [ "Subhendu Bhandary", "Federico Carucci", "Christos Charalambous", "Francesca Dilisante", "Ksenia Dvorkina", "Anna Garbo", "Jiaqi Liang", "Riccardo Vasellini", "Francesco Bertolotti" ], "categories": [ "cs.MA", "cs.AI" ], "topics": [ "agent-safety", "memory", "multi-agent" ], "score": 18, "relevance": "high", "matched_queries": [ "ai-agent", "llm-agent", "multi-agent-llm" ], "arxiv_id": "2606.28456", "source": "arxiv", "source_id": "arxiv:2606.28456", "pdf_url": "https://arxiv.org/pdf/2606.28456", "primary_query": "ai-agent" }, { "id": "2606.27406", "title": "Towards Evaluation of Implicit Software World Models in Coding LLMs", "url": "https://arxiv.org/abs/2606.27406", "published": "2026-06-25", "updated": "2026-06-25", "authors": [ "Egor Bogomolov", "Yaroslav Zharov" ], "categories": [ "cs.SE", "cs.AI" ], "topics": [ "agent-evaluation", "coding-agent", "memory", "reasoning", "world-model" ], "score": 18, "relevance": "high", "matched_queries": [ "ai-agent", "coding-agent" ], "arxiv_id": "2606.27406", "source": "arxiv", "source_id": "arxiv:2606.27406", "pdf_url": "https://arxiv.org/pdf/2606.27406", "primary_query": "ai-agent" }, { "id": "2606.26627", "title": "Agents That Know Too Much: A Data-Centric Survey of Privacy in LLM Agents", "url": "https://arxiv.org/abs/2606.26627", "published": "2026-06-25", "updated": "2026-06-25", "authors": [ "Nada Lahjouji", "Ashwin Gerard Colaco" ], "categories": [ "cs.CR", "cs.AI" ], "topics": [ "agent-evaluation", "agent-safety", "memory", "rag", "tool-use", "workflow-agent" ], "score": 18, "relevance": "high", "matched_queries": [ "agent-memory" ], "arxiv_id": "2606.26627", "source": "arxiv", "source_id": "arxiv:2606.26627", "pdf_url": "https://arxiv.org/pdf/2606.26627", "primary_query": "agent-memory" }, { "id": "2606.26479", "title": "Adaptive Evaluation of Out-of-Band Defenses Against Prompt Injection in LLM Agents", "url": "https://arxiv.org/abs/2606.26479", "published": "2026-06-25", "updated": "2026-06-25", "authors": [ "Praneeth Narisetty", "Shiva Nagendra Babu Kore", "Uday Kumar Reddy Kattamanchi", "Jayaram Kumarapu" ], "categories": [ "cs.CR", "cs.AI", "cs.CL", "cs.LG" ], "topics": [ "agent-evaluation", "agent-safety", "rag", "tool-use" ], "score": 18, "relevance": "high", "matched_queries": [ "tool-use" ], "arxiv_id": "2606.26479", "source": "arxiv", "source_id": "arxiv:2606.26479", "pdf_url": "https://arxiv.org/pdf/2606.26479", "primary_query": "tool-use" }, { "id": "2606.26793", "title": "MIRROR: Novelty-Constrained Memory-Guided MCTS Red-Teaming for Agentic RAG", "url": "https://arxiv.org/abs/2606.26793", "published": "2026-06-25", "updated": "2026-06-25", "authors": [ "Inderjeet Singh", "Andrés Murillo", "Motoyoshi Sekiya", "Yuki Unno", "Junichi Suga" ], "categories": [ "cs.CR", "cs.AI", "cs.LG" ], "topics": [ "agent-evaluation", "agent-safety", "computer-use", "memory", "rag", "tool-use" ], "score": 18, "relevance": "high", "matched_queries": [ "rag-agent" ], "arxiv_id": "2606.26793", "source": "arxiv", "source_id": "arxiv:2606.26793", "pdf_url": "https://arxiv.org/pdf/2606.26793", "primary_query": "rag-agent" }, { "id": "2606.27472", "title": "Supersede: Diagnosing and Training the Memory-Update Gap in LLM Agents", "url": "https://arxiv.org/abs/2606.27472", "published": "2026-06-25", "updated": "2026-06-25", "authors": [ "Vedant Patel" ], "categories": [ "cs.CL", "cs.AI", "cs.LG" ], "topics": [ "agent-evaluation", "memory", "planning", "reasoning", "tool-use" ], "score": 18, "relevance": "high", "matched_queries": [ "planning-agent" ], "arxiv_id": "2606.27472", "source": "arxiv", "source_id": "arxiv:2606.27472", "pdf_url": "https://arxiv.org/pdf/2606.27472", "primary_query": "planning-agent" }, { "id": "2606.25622", "title": "Probabilistic Agents in Deterministic Audits: Evaluating Multi-Agent Systems for Automated Audits Based on the German IT-Grundschutz", "url": "https://arxiv.org/abs/2606.25622", "published": "2026-06-24", "updated": "2026-06-24", "authors": [ "Lea Roxanne Muth", "Marian Margraf" ], "categories": [ "cs.CR", "cs.AI" ], "topics": [ "agent-evaluation", "agent-safety", "multi-agent", "rag", "reasoning", "tool-use", "workflow-agent" ], "score": 18, "relevance": "high", "matched_queries": [ "multi-agent-llm" ], "arxiv_id": "2606.25622", "source": "arxiv", "source_id": "arxiv:2606.25622", "pdf_url": "https://arxiv.org/pdf/2606.25622", "primary_query": "multi-agent-llm" }, { "id": "2606.25358", "title": "Agentic Knowledge Tracing: A Multi-Agent LLM Architecture for Stealth Assessment of Financial Literacy in Serious Games", "url": "https://arxiv.org/abs/2606.25358", "published": "2026-06-24", "updated": "2026-06-24", "authors": [ "Gabriel Santos", "Rita Julia", "Marcelo Nascimento" ], "categories": [ "cs.AI", "cs.MA" ], "topics": [ "agent-evaluation", "agent-safety", "multi-agent", "reasoning", "tool-use" ], "score": 18, "relevance": "high", "matched_queries": [ "multi-agent-llm" ], "arxiv_id": "2606.25358", "source": "arxiv", "source_id": "arxiv:2606.25358", "pdf_url": "https://arxiv.org/pdf/2606.25358", "primary_query": "multi-agent-llm" }, { "id": "2606.25334", "title": "Bridging the Post-discharge Gap: A Traceable Multi-agent Framework for Safe and Continuous Care", "url": "https://arxiv.org/abs/2606.25334", "published": "2026-06-24", "updated": "2026-06-24", "authors": [ "Runwei Guan", "Yi Zhou", "Heyi Lin", "Jinjing Zhu", "Mingyuan Hou", "Yang Yang", "Fang Yuan", "Xiaohong Lin", "Shaofeng Liang", "Xuming Hu", "Tao Li", "Tianbin Zhao", "Yutao Yue", "Zhiyuan Wang", "Hui Xiong" ], "categories": [ "cs.MA" ], "topics": [ "agent-evaluation", "agent-safety", "memory", "multi-agent", "rag" ], "score": 18, "relevance": "high", "matched_queries": [ "multi-agent-llm" ], "arxiv_id": "2606.25334", "source": "arxiv", "source_id": "arxiv:2606.25334", "pdf_url": "https://arxiv.org/pdf/2606.25334", "primary_query": "multi-agent-llm" }, { "id": "2606.22673", "title": "AgentLens: Interpretable Safety Steering via Mechanistic Subspaces for Multi-Turn Coding Agent", "url": "https://arxiv.org/abs/2606.22673", "published": "2026-06-21", "updated": "2026-06-21", "authors": [ "Weidi Luo", "Qiming Zhang", "Yihao Quan", "Mingyu Jin", "Jie Cai", "Chaowei Xiao", "Jingcheng Niu", "Zhen Xiang" ], "categories": [ "cs.AI", "cs.SE" ], "topics": [ "agent-evaluation", "agent-safety", "coding-agent", "tool-use" ], "score": 18, "relevance": "high", "matched_queries": [ "agent-safety", "coding-agent" ], "arxiv_id": "2606.22673", "source": "arxiv", "source_id": "arxiv:2606.22673", "pdf_url": "https://arxiv.org/pdf/2606.22673", "primary_query": "agent-safety" }, { "id": "2606.21710", "title": "PrivacyAlign: Contextual Privacy Alignment for LLM Agents", "url": "https://arxiv.org/abs/2606.21710", "published": "2026-06-19", "updated": "2026-06-19", "authors": [ "Manveer Singh Tamber", "Abhay Puri", "Marc-Etienne Brunet", "Perouz Taslakian", "Jimmy Lin", "Spandana Gella" ], "categories": [ "cs.CL", "cs.AI", "cs.IR" ], "topics": [ "agent-evaluation", "agent-safety", "planning", "tool-use" ], "score": 18, "relevance": "high", "matched_queries": [ "ai-agent" ], "arxiv_id": "2606.21710", "source": "arxiv", "source_id": "arxiv:2606.21710", "pdf_url": "https://arxiv.org/pdf/2606.21710", "primary_query": "ai-agent" }, { "id": "2606.21013", "title": "Agentic Time Machine as an Infrastructure for Future-Event Forecasting", "url": "https://arxiv.org/abs/2606.21013", "published": "2026-06-19", "updated": "2026-06-19", "authors": [ "Jingyi Chai", "Bingyang Zheng", "Xiangrui Liu", "Hao Lu", "Zihang Zhou", "Tianchen Wang", "Kemeng Zhang", "Siheng Chen" ], "categories": [ "cs.AI", "cs.LG" ], "topics": [ "agent-evaluation", "multi-agent", "planning", "rag", "tool-use" ], "score": 18, "relevance": "high", "matched_queries": [ "agent-evaluation" ], "arxiv_id": "2606.21013", "source": "arxiv", "source_id": "arxiv:2606.21013", "pdf_url": "https://arxiv.org/pdf/2606.21013", "primary_query": "agent-evaluation" }, { "id": "2606.21123", "title": "A Multi-Agent Audit Framework for High-Stakes Reasoning: Evaluation and Interpretability in Clinical Mental Health Screening", "url": "https://arxiv.org/abs/2606.21123", "published": "2026-06-19", "updated": "2026-06-19", "authors": [ "Jingchen Ye", "Yanpei Yu", "Luyao Zhang" ], "categories": [ "cs.CL" ], "topics": [ "agent-evaluation", "multi-agent", "rag", "reasoning", "workflow-agent" ], "score": 18, "relevance": "high", "matched_queries": [ "rag-agent" ], "arxiv_id": "2606.21123", "source": "arxiv", "source_id": "arxiv:2606.21123", "pdf_url": "https://arxiv.org/pdf/2606.21123", "primary_query": "rag-agent" }, { "id": "2606.19899", "title": "Measuring Biological Capabilities and Risks of AI Agents", "url": "https://arxiv.org/abs/2606.19899", "published": "2026-06-18", "updated": "2026-06-18", "authors": [ "Patricia Paskov", "Jeffrey Lee", "Kyle Brady", "Alyssa Worland" ], "categories": [ "cs.CY", "cs.AI" ], "topics": [ "agent-evaluation", "agent-safety", "computer-use", "rag", "tool-use", "workflow-agent" ], "score": 18, "relevance": "high", "matched_queries": [ "agent-evaluation", "agentic-ai", "ai-agent" ], "arxiv_id": "2606.19899", "source": "arxiv", "source_id": "arxiv:2606.19899", "pdf_url": "https://arxiv.org/pdf/2606.19899", "primary_query": "agent-evaluation" }, { "id": "2606.20243", "title": "Phoenix: Safe GitHub Issue Resolution via Multi-Agent LLMs", "url": "https://arxiv.org/abs/2606.20243", "published": "2026-06-18", "updated": "2026-06-22", "authors": [ "Kipngeno Koech", "Muhammad Adam", "Baimam Boukar Jean Jacques", "Joao Barros" ], "categories": [ "cs.SE", "cs.MA" ], "topics": [ "agent-evaluation", "agent-safety", "coding-agent", "multi-agent", "planning", "rag" ], "score": 18, "relevance": "high", "matched_queries": [ "coding-agent" ], "arxiv_id": "2606.20243", "source": "arxiv", "source_id": "arxiv:2606.20243", "pdf_url": "https://arxiv.org/pdf/2606.20243", "primary_query": "coding-agent" }, { "id": "2606.18950", "title": "RTSGameBench: An RTS Benchmark for Strategic Reasoning by Vision-Language Models", "url": "https://arxiv.org/abs/2606.18950", "published": "2026-06-17", "updated": "2026-06-18", "authors": [ "San Kim", "Daechul Ahn", "Reokyoung Kim", "Hyeonbeom Choi", "Seungyeon Jwa", "Jonghyun Choi" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "memory", "multi-agent", "planning", "rag", "reasoning", "tool-use" ], "score": 18, "relevance": "high", "matched_queries": [ "agent-memory" ], "arxiv_id": "2606.18950", "source": "arxiv", "source_id": "arxiv:2606.18950", "pdf_url": "https://arxiv.org/pdf/2606.18950", "primary_query": "agent-memory" }, { "id": "2606.16659", "title": "FraudSMSWalker: Benchmarking Agentic Large Language Models for SMS-to-Webpage Fraud Detection", "url": "https://arxiv.org/abs/2606.16659", "published": "2026-06-15", "updated": "2026-06-15", "authors": [ "Y. H. Zhou", "Z. M. Ma", "Y. J. Zhou", "Y. T. Li", "H. X. Xiang", "Y. M. Cheng", "T. L. Chen", "K. J. Zhang", "Z. H. Nan", "J. H. Ni", "Z. Wu", "Q. Y. Pan", "S. Zhang", "S. Cheng", "M. Y. Luo" ], "categories": [ "cs.CL" ], "topics": [ "agent-evaluation", "agent-safety", "computer-use", "tool-use" ], "score": 18, "relevance": "high", "matched_queries": [ "web-gui-agent" ], "arxiv_id": "2606.16659", "source": "arxiv", "source_id": "arxiv:2606.16659", "pdf_url": "https://arxiv.org/pdf/2606.16659", "primary_query": "web-gui-agent" }, { "id": "2606.17041", "title": "Benchmarking LLM Agents on Meta-Analysis Articles from Nature Portfolio", "url": "https://arxiv.org/abs/2606.17041", "published": "2026-06-15", "updated": "2026-07-01", "authors": [ "Anzhe Xie", "Weihang Su", "Yujia Zhou", "Yiqun Liu", "Qingyao Ai" ], "categories": [ "cs.CL", "cs.IR" ], "topics": [ "agent-evaluation", "computer-use", "rag", "reasoning", "workflow-agent" ], "score": 18, "relevance": "high", "matched_queries": [ "rag-agent" ], "arxiv_id": "2606.17041", "source": "arxiv", "source_id": "arxiv:2606.17041", "pdf_url": "https://arxiv.org/pdf/2606.17041", "primary_query": "rag-agent" }, { "id": "2606.16576", "title": "Can LLM Agents Infer World Models? Evidence from Agentic Automata Learning", "url": "https://arxiv.org/abs/2606.16576", "published": "2026-06-15", "updated": "2026-06-15", "authors": [ "Reef Menaged", "Gili Lior", "Shauli Ravfogel", "Roee Aharoni", "Gabriel Stanovsky" ], "categories": [ "cs.CL" ], "topics": [ "agent-evaluation", "planning", "reasoning", "tool-use", "world-model" ], "score": 18, "relevance": "high", "matched_queries": [ "planning-agent" ], "arxiv_id": "2606.16576", "source": "arxiv", "source_id": "arxiv:2606.16576", "pdf_url": "https://arxiv.org/pdf/2606.16576", "primary_query": "planning-agent" }, { "id": "2606.12780", "title": "ProPlay: Procedural World Models for Self-Evolving LLM Agents", "url": "https://arxiv.org/abs/2606.12780", "published": "2026-06-11", "updated": "2026-06-11", "authors": [ "Yijun Ma", "Zehong Wang", "Yiyang Li", "Ziming Li", "Xiaoguang Guo", "Weixiang Sun", "Chuxu Zhang", "Yanfang Ye" ], "categories": [ "cs.LG", "cs.CL" ], "topics": [ "agent-evaluation", "computer-use", "memory", "planning", "tool-use", "world-model" ], "score": 18, "relevance": "high", "matched_queries": [ "planning-agent" ], "arxiv_id": "2606.12780", "source": "arxiv", "source_id": "arxiv:2606.12780", "pdf_url": "https://arxiv.org/pdf/2606.12780", "primary_query": "planning-agent" }, { "id": "2606.12320", "title": "A Five-Plane Reference Architecture for Runtime Governance of Production AI Agents", "url": "https://arxiv.org/abs/2606.12320", "published": "2026-06-10", "updated": "2026-06-10", "authors": [ "Krti Tallam" ], "categories": [ "cs.AI", "cs.CC", "cs.CR", "cs.SE" ], "topics": [ "agent-evaluation", "agent-safety", "memory", "planning", "reasoning", "tool-use", "workflow-agent" ], "score": 18, "relevance": "high", "matched_queries": [ "agent-evaluation" ], "arxiv_id": "2606.12320", "source": "arxiv", "source_id": "arxiv:2606.12320", "pdf_url": "https://arxiv.org/pdf/2606.12320", "primary_query": "agent-evaluation" }, { "id": "2606.11680", "title": "Organize then Retrieve: Hierarchical Memory Navigation for Efficient Agents", "url": "https://arxiv.org/abs/2606.11680", "published": "2026-06-10", "updated": "2026-06-10", "authors": [ "Hao-Lun Hsu", "Nikki Lijing Kuang", "Boyi Liu", "Zhewei Yao", "Yuxiong He" ], "categories": [ "cs.AI", "cs.CL", "cs.LG" ], "topics": [ "agent-evaluation", "computer-use", "embodied-agent", "memory", "planning", "rag", "reasoning" ], "score": 18, "relevance": "high", "matched_queries": [ "agent-memory" ], "arxiv_id": "2606.11680", "source": "arxiv", "source_id": "arxiv:2606.11680", "pdf_url": "https://arxiv.org/pdf/2606.11680", "primary_query": "agent-memory" }, { "id": "2606.12657", "title": "TrajGenAgent: A Hierarchical LLM Agent for Human Mobility Trajectory Generation", "url": "https://arxiv.org/abs/2606.12657", "published": "2026-06-10", "updated": "2026-06-10", "authors": [ "Siyu Li", "Toan Tran", "Lingyi Zhao", "Khurram Shafique", "Li Xiong" ], "categories": [ "cs.AI", "cs.DB", "cs.RO" ], "topics": [ "agent-evaluation", "planning", "rag", "reasoning", "workflow-agent", "world-model" ], "score": 18, "relevance": "high", "matched_queries": [ "planning-agent" ], "arxiv_id": "2606.12657", "source": "arxiv", "source_id": "arxiv:2606.12657", "pdf_url": "https://arxiv.org/pdf/2606.12657", "primary_query": "planning-agent" }, { "id": "2606.10677", "title": "Infini Memory: Maintainable Topic Documents for Long-Term LLM Agent Memory", "url": "https://arxiv.org/abs/2606.10677", "published": "2026-06-09", "updated": "2026-06-09", "authors": [ "Suozhao Ji", "Baodong Wu", "Zehao Wang", "Lei Xia", "Qingping Li", "Ruisong Wang", "Wenbo Ding", "Zhenhua Zhu", "Boxun Li", "Guohao Dai", "Yu Wang" ], "categories": [ "cs.AI", "cs.CL" ], "topics": [ "agent-evaluation", "memory", "rag", "tool-use" ], "score": 18, "relevance": "high", "matched_queries": [ "agent-memory" ], "arxiv_id": "2606.10677", "source": "arxiv", "source_id": "arxiv:2606.10677", "pdf_url": "https://arxiv.org/pdf/2606.10677", "primary_query": "agent-memory" }, { "id": "2606.11354", "title": "A Zero-Shot Multi-Agent Framework for Human-Building Interaction via Programmatic Reasoning", "url": "https://arxiv.org/abs/2606.11354", "published": "2026-06-09", "updated": "2026-06-09", "authors": [ "Yuqi Wang", "Gulai Shen", "Ali Mehmani" ], "categories": [ "cs.ET" ], "topics": [ "agent-safety", "coding-agent", "multi-agent", "planning", "rag", "reasoning", "tool-use" ], "score": 18, "relevance": "high", "matched_queries": [ "rag-agent" ], "arxiv_id": "2606.11354", "source": "arxiv", "source_id": "arxiv:2606.11354", "pdf_url": "https://arxiv.org/pdf/2606.11354", "primary_query": "rag-agent" }, { "id": "2606.05558", "title": "Autoregressive Diffusion World Models for Off-Policy Evaluation of LLM Agents", "url": "https://arxiv.org/abs/2606.05558", "published": "2026-06-04", "updated": "2026-06-04", "authors": [ "Kaixuan Liu", "Guojun Xiong", "Weinan Zhang", "Shengpu Tang" ], "categories": [ "cs.LG" ], "topics": [ "agent-evaluation", "agent-safety", "computer-use", "tool-use", "world-model" ], "score": 18, "relevance": "high", "matched_queries": [ "agent-evaluation" ], "arxiv_id": "2606.05558", "source": "arxiv", "source_id": "arxiv:2606.05558", "pdf_url": "https://arxiv.org/pdf/2606.05558", "primary_query": "agent-evaluation" }, { "id": "2606.04555", "title": "Temporal Order Matters for Agentic Memory: Segment Trees for Long-Horizon Agents", "url": "https://arxiv.org/abs/2606.04555", "published": "2026-06-03", "updated": "2026-06-03", "authors": [ "Yifan Simon Liu", "Liam Gallagher", "Faeze Moradi Kalarde", "Jiazhou Liang", "Armin Toroghi", "Scott Sanner" ], "categories": [ "cs.CL", "cs.AI" ], "topics": [ "agent-evaluation", "memory", "planning", "rag" ], "score": 18, "relevance": "high", "matched_queries": [ "agent-memory" ], "arxiv_id": "2606.04555", "source": "arxiv", "source_id": "arxiv:2606.04555", "pdf_url": "https://arxiv.org/pdf/2606.04555", "primary_query": "agent-memory" }, { "id": "2606.03895", "title": "Agent libOS: A Runtime Substrate for Capability-Controlled Self-Evolving LLM Agents", "url": "https://arxiv.org/abs/2606.03895", "published": "2026-06-02", "updated": "2026-06-29", "authors": [ "Yingqi Zhang" ], "categories": [ "cs.OS", "cs.AI", "cs.CR" ], "topics": [ "agent-evaluation", "agent-safety", "memory", "planning", "tool-use" ], "score": 18, "relevance": "high", "matched_queries": [ "planning-agent" ], "arxiv_id": "2606.03895", "source": "arxiv", "source_id": "arxiv:2606.03895", "pdf_url": "https://arxiv.org/pdf/2606.03895", "primary_query": "planning-agent" }, { "id": "2606.02302", "title": "SeClaw: Spec-Driven Security Task Synthesis for Evaluating Autonomous Agents", "url": "https://arxiv.org/abs/2606.02302", "published": "2026-06-01", "updated": "2026-06-01", "authors": [ "Hao Cheng", "Changtao Miao", "Tianle Song", "Yin Wu", "He Liu", "Erjia Xiao", "Junchi Chen", "Xiaoyu Shi", "Yichi Wang", "Jing Yang", "Taowen Wang", "Jinhao Duan", "Mengshu Sun", "Peiyan Dong", "Xuan Shen", "Yang Cao", "Renjing Xu", "Kaidi Xu", "Jindong Gu", "Bo Zhang", "Jize Zhang", "Chenhao Lin", "Philip Torr", "Chao Shen" ], "categories": [ "cs.CR", "cs.AI" ], "topics": [ "agent-evaluation", "agent-safety", "memory", "rag", "tool-use", "workflow-agent" ], "score": 18, "relevance": "high", "matched_queries": [ "agent-safety", "autonomous-agent-llm" ], "arxiv_id": "2606.02302", "source": "arxiv", "source_id": "arxiv:2606.02302", "pdf_url": "https://arxiv.org/pdf/2606.02302", "primary_query": "agent-safety" }, { "id": "2605.30711", "title": "SAGE: A Novelty Gate for Efficient Memory Evolution in Agentic LLMs", "url": "https://arxiv.org/abs/2605.30711", "published": "2026-05-29", "updated": "2026-06-18", "authors": [ "Sijia Wang", "Dhanajit Brahma", "Ricardo Henao" ], "categories": [ "cs.CL", "cs.AI", "cs.LG", "stat.ML" ], "topics": [ "agent-evaluation", "memory", "rag", "reasoning", "tool-use" ], "score": 18, "relevance": "high", "matched_queries": [ "agent-memory" ], "arxiv_id": "2605.30711", "source": "arxiv", "source_id": "arxiv:2605.30711", "pdf_url": "https://arxiv.org/pdf/2605.30711", "primary_query": "agent-memory" }, { "id": "2605.29341", "title": "WorldMemArena: Evaluating Multimodal Agent Memory Through Action-World Interaction", "url": "https://arxiv.org/abs/2605.29341", "published": "2026-05-28", "updated": "2026-06-01", "authors": [ "Chengzhi Liu", "Yuzhe Yang", "Sophia Xiao Pu", "Yepeng Liu", "Lin Long", "Yichen Guo", "Nuo Chen", "Zhaotian Weng", "Elena Kochkina", "Simerjot Kaur", "Charese Smiley", "Xiaomo Liu", "James Zou", "Sheng Liu", "Yuheng Bu", "Songyou Peng", "Xin Eric Wang" ], "categories": [ "cs.CV", "cs.CL" ], "topics": [ "agent-evaluation", "memory", "planning", "rag", "tool-use" ], "score": 18, "relevance": "high", "matched_queries": [ "agent-memory", "rag-agent" ], "arxiv_id": "2605.29341", "source": "arxiv", "source_id": "arxiv:2605.29341", "pdf_url": "https://arxiv.org/pdf/2605.29341", "primary_query": "agent-memory" }, { "id": "2605.27690", "title": "TRACES: Proactive Safety Auditing for Multi-Turn LLM Agents via Trajectory-State Modeling", "url": "https://arxiv.org/abs/2605.27690", "published": "2026-05-26", "updated": "2026-05-26", "authors": [ "Jiaqian Li", "Yanshu Li", "Boxuan Zhang", "Ruixiang Tang", "Kuan-Hao Huang" ], "categories": [ "cs.CL", "cs.LG" ], "topics": [ "agent-evaluation", "agent-safety", "computer-use", "planning", "tool-use" ], "score": 18, "relevance": "high", "matched_queries": [ "agent-safety" ], "arxiv_id": "2605.27690", "source": "arxiv", "source_id": "arxiv:2605.27690", "pdf_url": "https://arxiv.org/pdf/2605.27690", "primary_query": "agent-safety" }, { "id": "2605.25141", "title": "LLM Agent Based Renewable Energy Forecasting Using Edge and IoT Data A Review of Solar Wind Weather and Grid Aware Decision Support", "url": "https://arxiv.org/abs/2605.25141", "published": "2026-05-24", "updated": "2026-05-24", "authors": [ "Pavan Manjunath", "Thomas Pruefer" ], "categories": [ "cs.CL", "cs.AI" ], "topics": [ "agent-evaluation", "computer-use", "planning", "rag", "reasoning", "tool-use", "workflow-agent" ], "score": 18, "relevance": "high", "matched_queries": [ "planning-agent" ], "arxiv_id": "2605.25141", "source": "arxiv", "source_id": "arxiv:2605.25141", "pdf_url": "https://arxiv.org/pdf/2605.25141", "primary_query": "planning-agent" }, { "id": "2605.19952", "title": "Rethinking How to Remember: Beyond Atomic Facts in Lifelong LLM Agent Memory", "url": "https://arxiv.org/abs/2605.19952", "published": "2026-05-19", "updated": "2026-05-19", "authors": [ "Jingwei Sun", "Jianing Zhu", "Jiangchao Yao", "Tongliang Liu", "Bo Han" ], "categories": [ "cs.CL" ], "topics": [ "agent-evaluation", "memory", "rag", "reasoning", "tool-use" ], "score": 18, "relevance": "high", "matched_queries": [ "agent-memory" ], "arxiv_id": "2605.19952", "source": "arxiv", "source_id": "arxiv:2605.19952", "pdf_url": "https://arxiv.org/pdf/2605.19952", "primary_query": "agent-memory" }, { "id": "2605.17625", "title": "Episodic-Semantic Memory Architecture for Long-Horizon Scientific Agents", "url": "https://arxiv.org/abs/2605.17625", "published": "2026-05-17", "updated": "2026-05-17", "authors": [ "Nikola Milosevic" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "memory", "planning", "rag", "reasoning", "tool-use", "workflow-agent" ], "score": 18, "relevance": "high", "matched_queries": [ "agent-memory" ], "arxiv_id": "2605.17625", "source": "arxiv", "source_id": "arxiv:2605.17625", "pdf_url": "https://arxiv.org/pdf/2605.17625", "primary_query": "agent-memory" }, { "id": "2605.16821", "title": "Multi-Paradigm Agent Interaction in Practice:A Systematic Analysis of Generator-Evaluator, ReAct Loop,and Adversarial Evaluation in the buddyMe Framework", "url": "https://arxiv.org/abs/2605.16821", "published": "2026-05-16", "updated": "2026-05-16", "authors": [ "Xiaohua Wang", "Chao Han", "Kai Yu", "XiaoLiang Xu", "Liang Wang" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "computer-use", "memory", "multi-agent", "planning", "tool-use" ], "score": 18, "relevance": "high", "matched_queries": [ "planning-agent" ], "arxiv_id": "2605.16821", "source": "arxiv", "source_id": "arxiv:2605.16821", "pdf_url": "https://arxiv.org/pdf/2605.16821", "primary_query": "planning-agent" }, { "id": "2605.16481", "title": "Visual Agentic Memory: Enabling Online Long Video Understanding via Online Indexing, Hierarchical Memory, and Agentic Retrieval", "url": "https://arxiv.org/abs/2605.16481", "published": "2026-05-15", "updated": "2026-05-15", "authors": [ "Aiden Yiliu Li", "Nels Numan", "Anthony Steed" ], "categories": [ "cs.CV", "cs.AI" ], "topics": [ "agent-evaluation", "memory", "planning", "rag", "reasoning" ], "score": 18, "relevance": "high", "matched_queries": [ "agent-memory" ], "arxiv_id": "2605.16481", "source": "arxiv", "source_id": "arxiv:2605.16481", "pdf_url": "https://arxiv.org/pdf/2605.16481", "primary_query": "agent-memory" }, { "id": "2605.12061", "title": "SAGE: A Self-Evolving Agentic Graph-Memory Engine for Structure-Aware Associative Memory", "url": "https://arxiv.org/abs/2605.12061", "published": "2026-05-12", "updated": "2026-05-12", "authors": [ "Juntong Wang", "Haoyue Zhao", "guanghui Pan", "Xiyuan Wang", "Yanbo Wang", "Qiyan Deng", "Muhan Zhang" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "memory", "planning", "rag", "tool-use" ], "score": 18, "relevance": "high", "matched_queries": [ "language-agent" ], "arxiv_id": "2605.12061", "source": "arxiv", "source_id": "arxiv:2605.12061", "pdf_url": "https://arxiv.org/pdf/2605.12061", "primary_query": "language-agent" }, { "id": "2605.06890", "title": "Beyond the Black Box: Interpretability of Agentic AI Tool Use", "url": "https://arxiv.org/abs/2605.06890", "published": "2026-05-07", "updated": "2026-07-05", "authors": [ "Hariom Tatsat", "Ariye Shater" ], "categories": [ "cs.AI", "cs.MA" ], "topics": [ "agent-evaluation", "agent-safety", "planning", "tool-use", "workflow-agent" ], "score": 18, "relevance": "high", "matched_queries": [ "function-calling" ], "arxiv_id": "2605.06890", "source": "arxiv", "source_id": "arxiv:2605.06890", "pdf_url": "https://arxiv.org/pdf/2605.06890", "primary_query": "function-calling" }, { "id": "2605.05716", "title": "More Is Not Always Better: Cross-Component Interference in LLM Agent Scaffolding", "url": "https://arxiv.org/abs/2605.05716", "published": "2026-05-07", "updated": "2026-05-07", "authors": [ "Ming Liu" ], "categories": [ "cs.AI", "cs.CL" ], "topics": [ "agent-evaluation", "memory", "planning", "rag", "reasoning", "tool-use" ], "score": 18, "relevance": "high", "matched_queries": [ "planning-agent" ], "arxiv_id": "2605.05716", "source": "arxiv", "source_id": "arxiv:2605.05716", "pdf_url": "https://arxiv.org/pdf/2605.05716", "primary_query": "planning-agent" }, { "id": "2605.06716", "title": "From Storage to Experience: A Survey on the Evolution of LLM Agent Memory Mechanisms", "url": "https://arxiv.org/abs/2605.06716", "published": "2026-05-07", "updated": "2026-05-07", "authors": [ "Jinghao Luo", "Yuchen Tian", "Chuxue Cao", "Ziyang Luo", "Hongzhan Lin", "Kaixin Li", "Chuyi Kong", "Ruichao Yang", "Jing Ma" ], "categories": [ "cs.AI", "cs.CL" ], "topics": [ "memory", "planning", "rag", "reasoning", "tool-use" ], "score": 18, "relevance": "high", "matched_queries": [ "planning-agent" ], "arxiv_id": "2605.06716", "source": "arxiv", "source_id": "arxiv:2605.06716", "pdf_url": "https://arxiv.org/pdf/2605.06716", "primary_query": "planning-agent" }, { "id": "2604.25318", "title": "Cutscene Agent: An LLM Agent Framework for Automated 3D Cutscene Generation", "url": "https://arxiv.org/abs/2604.25318", "published": "2026-04-28", "updated": "2026-04-28", "authors": [ "Lanshan He", "Haozhou Pang", "Qi Gan", "Xin Shen", "Ziwei Zhang", "Yibo Liu", "Gang Fang", "Bo Liu", "Kai Sheng", "Shengfeng Zeng", "Chaofan Li", "Zhen Hui", "Keer Zhou", "Lan Zhou", "Shujun Dai" ], "categories": [ "cs.GR", "cs.AI", "cs.CL" ], "topics": [ "agent-evaluation", "multi-agent", "planning", "reasoning", "tool-use" ], "score": 18, "relevance": "high", "matched_queries": [ "function-calling" ], "arxiv_id": "2604.25318", "source": "arxiv", "source_id": "arxiv:2604.25318", "pdf_url": "https://arxiv.org/pdf/2604.25318", "primary_query": "function-calling" }, { "id": "2604.25135", "title": "FAMA: Failure-Aware Meta-Agentic Framework for Open-Source LLMs in Interactive Tool Use Environments", "url": "https://arxiv.org/abs/2604.25135", "published": "2026-04-28", "updated": "2026-04-28", "authors": [ "Amir Saeidi", "Venkatesh Mishra", "Souradeep Mukhopadhyay", "Gaowen Liu", "Ali Payani", "Jayanth Srinivasa", "Chitta Baral" ], "categories": [ "cs.CL" ], "topics": [ "agent-evaluation", "tool-use" ], "score": 18, "relevance": "high", "matched_queries": [ "autonomous-agent-llm" ], "arxiv_id": "2604.25135", "source": "arxiv", "source_id": "arxiv:2604.25135", "pdf_url": "https://arxiv.org/pdf/2604.25135", "primary_query": "autonomous-agent-llm" }, { "id": "2604.23459", "title": "Architecture Matters for Multi-Agent Security", "url": "https://arxiv.org/abs/2604.23459", "published": "2026-04-25", "updated": "2026-04-25", "authors": [ "Ben Hagag", "William L. Anderson", "Christian Schroeder de Witt", "Sarah Scheffler" ], "categories": [ "cs.MA", "cs.CR", "cs.LG" ], "topics": [ "agent-evaluation", "agent-safety", "computer-use", "memory", "multi-agent", "planning" ], "score": 18, "relevance": "high", "matched_queries": [ "agent-safety" ], "arxiv_id": "2604.23459", "source": "arxiv", "source_id": "arxiv:2604.23459", "pdf_url": "https://arxiv.org/pdf/2604.23459", "primary_query": "agent-safety" }, { "id": "2604.23374", "title": "Ghost in the Agent: Redefining Information Flow Tracking for LLM Agents", "url": "https://arxiv.org/abs/2604.23374", "published": "2026-04-25", "updated": "2026-04-25", "authors": [ "Yuandao Cai", "Wensheng Tang", "Cheng Wen", "Shengchao Qin" ], "categories": [ "cs.CR" ], "topics": [ "agent-evaluation", "agent-safety", "memory", "reasoning", "tool-use" ], "score": 18, "relevance": "high", "matched_queries": [ "agent-safety" ], "arxiv_id": "2604.23374", "source": "arxiv", "source_id": "arxiv:2604.23374", "pdf_url": "https://arxiv.org/pdf/2604.23374", "primary_query": "agent-safety" }, { "id": "2604.22879", "title": "Beyond Single-Agent Alignment: Preventing Context-Fragmented Violations in Multi-Agent Systems", "url": "https://arxiv.org/abs/2604.22879", "published": "2026-04-24", "updated": "2026-04-24", "authors": [ "Jie Wu", "Ming Gong" ], "categories": [ "cs.MA", "cs.AI", "cs.CR", "cs.LG" ], "topics": [ "agent-evaluation", "agent-safety", "multi-agent", "rag", "tool-use", "workflow-agent", "world-model" ], "score": 18, "relevance": "high", "matched_queries": [ "agent-safety" ], "arxiv_id": "2604.22879", "source": "arxiv", "source_id": "arxiv:2604.22879", "pdf_url": "https://arxiv.org/pdf/2604.22879", "primary_query": "agent-safety" }, { "id": "2604.18847", "title": "Human-Guided Harm Recovery for Computer Use Agents", "url": "https://arxiv.org/abs/2604.18847", "published": "2026-04-20", "updated": "2026-05-28", "authors": [ "Christy Li", "Sky CH-Wang", "Andi Peng", "Andreea Bobu" ], "categories": [ "cs.AI", "cs.CL" ], "topics": [ "agent-evaluation", "agent-safety", "computer-use", "memory", "planning", "rag", "tool-use" ], "score": 18, "relevance": "high", "matched_queries": [ "agent-safety" ], "arxiv_id": "2604.18847", "source": "arxiv", "source_id": "arxiv:2604.18847", "pdf_url": "https://arxiv.org/pdf/2604.18847", "primary_query": "agent-safety" }, { "id": "2603.18245", "title": "Who Tests the Testers? Systematic Enumeration and Coverage Audit of LLM Agent Tool Call Safety", "url": "https://arxiv.org/abs/2603.18245", "published": "2026-03-18", "updated": "2026-03-18", "authors": [ "Xuan Chen", "Lu Yan", "Ruqi Zhang", "Xiangyu Zhang" ], "categories": [ "cs.SE", "cs.CR" ], "topics": [ "agent-evaluation", "agent-safety", "rag", "tool-use", "workflow-agent" ], "score": 18, "relevance": "high", "matched_queries": [ "agent-safety" ], "arxiv_id": "2603.18245", "source": "arxiv", "source_id": "arxiv:2603.18245", "pdf_url": "https://arxiv.org/pdf/2603.18245", "primary_query": "agent-safety" }, { "id": "2603.07980", "title": "\\$OneMillion-Bench: How Far are Language Agents from Human Experts?", "url": "https://arxiv.org/abs/2603.07980", "published": "2026-03-09", "updated": "2026-03-09", "authors": [ "Qianyu Yang", "Yang Liu", "Jiaqi Li", "Jun Bai", "Hao Chen", "Kaiyuan Chen", "Tiliang Duan", "Jiayun Dong", "Xiaobo Hu", "Zixia Jia", "Yang Liu", "Tao Peng", "Yixin Ren", "Ran Tian", "Zaiyuan Wang", "Yanglihong Xiao", "Gang Yao", "Lingyue Yin", "Ge Zhang", "Chun Zhang", "Jianpeng Jiao", "Zilong Zheng", "Yuan Gong" ], "categories": [ "cs.LG", "cs.AI", "cs.CL" ], "topics": [ "agent-evaluation", "planning", "reasoning", "tool-use" ], "score": 18, "relevance": "high", "matched_queries": [ "language-agent" ], "arxiv_id": "2603.07980", "source": "arxiv", "source_id": "arxiv:2603.07980", "pdf_url": "https://arxiv.org/pdf/2603.07980", "primary_query": "language-agent" }, { "id": "2603.09002", "title": "Security Considerations for Multi-agent Systems", "url": "https://arxiv.org/abs/2603.09002", "published": "2026-03-09", "updated": "2026-04-26", "authors": [ "Tam Nguyen", "Moses Ndebugre", "Dheeraj Arremsetty" ], "categories": [ "cs.CR", "cs.AI" ], "topics": [ "agent-evaluation", "agent-safety", "computer-use", "memory", "multi-agent", "planning", "rag", "tool-use" ], "score": 18, "relevance": "high", "matched_queries": [ "agent-safety" ], "arxiv_id": "2603.09002", "source": "arxiv", "source_id": "arxiv:2603.09002", "pdf_url": "https://arxiv.org/pdf/2603.09002", "primary_query": "agent-safety" }, { "id": "2602.07962", "title": "LOCA-bench: Benchmarking Language Agents Under Controllable and Extreme Context Growth", "url": "https://arxiv.org/abs/2602.07962", "published": "2026-02-08", "updated": "2026-02-08", "authors": [ "Weihao Zeng", "Yuzhen Huang", "Junxian He" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "planning", "rag", "tool-use" ], "score": 18, "relevance": "high", "matched_queries": [ "language-agent" ], "arxiv_id": "2602.07962", "source": "arxiv", "source_id": "arxiv:2602.07962", "pdf_url": "https://arxiv.org/pdf/2602.07962", "primary_query": "language-agent" }, { "id": "2602.05302", "title": "PieArena: Ranking and Profiling Language Agents in Realistic Negotiation Scenarios", "url": "https://arxiv.org/abs/2602.05302", "published": "2026-02-05", "updated": "2026-06-01", "authors": [ "Chris Zhu", "Sasha Cui", "Will Sanok Dufallo", "Runzhi Jin", "Zhen Xu", "Linjun Zhang", "Daylian Cain" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "multi-agent", "reasoning", "tool-use" ], "score": 18, "relevance": "high", "matched_queries": [ "language-agent" ], "arxiv_id": "2602.05302", "source": "arxiv", "source_id": "arxiv:2602.05302", "pdf_url": "https://arxiv.org/pdf/2602.05302", "primary_query": "language-agent" }, { "id": "2601.06007", "title": "Don't Break the Cache: An Evaluation of Prompt Caching for Long-Horizon Agentic Tasks", "url": "https://arxiv.org/abs/2601.06007", "published": "2026-01-09", "updated": "2026-01-31", "authors": [ "Elias Lumer", "Faheem Nizar", "Akshaya Jangiti", "Kevin Frank", "Anmol Gulati", "Mandar Phadate", "Vamse Kumar Subbiah" ], "categories": [ "cs.CL" ], "topics": [ "agent-evaluation", "computer-use", "planning", "tool-use" ], "score": 18, "relevance": "high", "matched_queries": [ "function-calling" ], "arxiv_id": "2601.06007", "source": "arxiv", "source_id": "arxiv:2601.06007", "pdf_url": "https://arxiv.org/pdf/2601.06007", "primary_query": "function-calling" }, { "id": "2511.04847", "title": "Test-Time Adaptation for LLM Agents via Environment Interaction", "url": "https://arxiv.org/abs/2511.04847", "published": "2025-11-06", "updated": "2026-02-22", "authors": [ "Arthur Chen", "Zuxin Liu", "Jianguo Zhang", "Akshara Prabhakar", "Zhiwei Liu", "Shelby Heinecke", "Silvio Savarese", "Victor Zhong", "Caiming Xiong" ], "categories": [ "cs.LG" ], "topics": [ "agent-evaluation", "agent-safety", "embodied-agent", "rag", "tool-use", "world-model" ], "score": 18, "relevance": "high", "matched_queries": [ "function-calling" ], "arxiv_id": "2511.04847", "source": "arxiv", "source_id": "arxiv:2511.04847", "pdf_url": "https://arxiv.org/pdf/2511.04847", "primary_query": "function-calling" }, { "id": "2509.10769", "title": "AgentArch: A Comprehensive Benchmark to Evaluate Agent Architectures in Enterprise", "url": "https://arxiv.org/abs/2509.10769", "published": "2025-09-13", "updated": "2026-01-06", "authors": [ "Tara Bogavelli", "Roshnee Sharma", "Hari Subramani" ], "categories": [ "cs.AI", "cs.CL", "cs.MA" ], "topics": [ "agent-evaluation", "memory", "multi-agent", "tool-use", "workflow-agent" ], "score": 18, "relevance": "high", "matched_queries": [ "function-calling" ], "arxiv_id": "2509.10769", "source": "arxiv", "source_id": "arxiv:2509.10769", "pdf_url": "https://arxiv.org/pdf/2509.10769", "primary_query": "function-calling" }, { "id": "2607.06273", "title": "AgentTether: Graph-Guided Diagnosis and Runtime Intervention for Reliable LLM Agent Operation", "url": "https://arxiv.org/abs/2607.06273", "published": "2026-07-07", "updated": "2026-07-07", "authors": [ "Chenyu Zhao", "Shenglin Zhang", "Wenwei Gu", "Yongqian Sun", "Dan Pei", "Chetan Bansal", "Saravan Rajmohan", "Minghua Ma" ], "categories": [ "cs.SE" ], "topics": [ "agent-evaluation", "computer-use", "memory", "reasoning", "tool-use" ], "score": 17, "relevance": "high", "matched_queries": [ "llm-agent", "tool-use" ], "arxiv_id": "2607.06273", "source": "arxiv", "source_id": "arxiv:2607.06273", "pdf_url": "https://arxiv.org/pdf/2607.06273", "primary_query": "llm-agent" }, { "id": "2607.06195", "title": "LogicHunter: Testing LLM Agent Frameworks with an Agentic Oracle", "url": "https://arxiv.org/abs/2607.06195", "published": "2026-07-07", "updated": "2026-07-07", "authors": [ "Minghui Long", "Yanjie Zhao", "Haoyu Wang" ], "categories": [ "cs.SE" ], "topics": [ "agent-evaluation", "computer-use", "memory" ], "score": 17, "relevance": "high", "matched_queries": [ "llm-agent" ], "arxiv_id": "2607.06195", "source": "arxiv", "source_id": "arxiv:2607.06195", "pdf_url": "https://arxiv.org/pdf/2607.06195", "primary_query": "llm-agent" }, { "id": "2607.06080", "title": "From Blueprint to Reality: Modeling and Applying Putnam's Social Capital Theory with LLM-based Multi-agent Simulations", "url": "https://arxiv.org/abs/2607.06080", "published": "2026-07-07", "updated": "2026-07-07", "authors": [ "Shiyi Ling", "Zhi Zheng", "Hui Zheng", "Wenjun Xue", "Feng Ye", "Tong Xu" ], "categories": [ "cs.CL", "cs.AI", "cs.SI" ], "topics": [ "agent-safety", "multi-agent", "rag", "tool-use", "world-model" ], "score": 17, "relevance": "high", "matched_queries": [ "llm-agent", "multi-agent-llm" ], "arxiv_id": "2607.06080", "source": "arxiv", "source_id": "arxiv:2607.06080", "pdf_url": "https://arxiv.org/pdf/2607.06080", "primary_query": "llm-agent" }, { "id": "2607.05297", "title": "MetaSkill-Evolve: Recursive Self-Improvement of LLM Agents via Two-Timescale Meta-Skill Evolution", "url": "https://arxiv.org/abs/2607.05297", "published": "2026-07-06", "updated": "2026-07-06", "authors": [ "Zefeng Wang", "Minxi Yan", "Jinhe Bi", "Sikuan Yan", "Volker Tresp", "Yunpu Ma" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "planning", "reasoning", "workflow-agent" ], "score": 17, "relevance": "high", "matched_queries": [ "agent-evaluation", "llm-agent" ], "arxiv_id": "2607.05297", "source": "arxiv", "source_id": "arxiv:2607.05297", "pdf_url": "https://arxiv.org/pdf/2607.05297", "primary_query": "agent-evaluation" }, { "id": "2607.04528", "title": "Measuring Harness-Induced Belief Divergence in Multi-Step LLM Agents", "url": "https://arxiv.org/abs/2607.04528", "published": "2026-07-05", "updated": "2026-07-05", "authors": [ "Haiwen Yi", "Xinyuan Song" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "agent-safety", "coding-agent", "tool-use" ], "score": 17, "relevance": "high", "matched_queries": [ "agent-evaluation", "llm-agent" ], "arxiv_id": "2607.04528", "source": "arxiv", "source_id": "arxiv:2607.04528", "pdf_url": "https://arxiv.org/pdf/2607.04528", "primary_query": "agent-evaluation" }, { "id": "2607.04293", "title": "CausalGame: Benchmarking Causal Thinking of LLM Agents in Games", "url": "https://arxiv.org/abs/2607.04293", "published": "2026-07-05", "updated": "2026-07-05", "authors": [ "Zhenhao Chen", "Yongqiang Chen", "Chenxi Liu", "Junchi Yu", "Xiangchen Song", "Zijian Li", "Jialin Li", "Philip Torr", "Bo Han", "Kun Zhang" ], "categories": [ "cs.CL", "cs.AI", "cs.LG", "stat.ML" ], "topics": [ "agent-evaluation", "computer-use", "planning", "reasoning" ], "score": 17, "relevance": "high", "matched_queries": [ "llm-agent" ], "arxiv_id": "2607.04293", "source": "arxiv", "source_id": "arxiv:2607.04293", "pdf_url": "https://arxiv.org/pdf/2607.04293", "primary_query": "llm-agent" }, { "id": "2607.04426", "title": "ACE-Brain-0.5: A Unified Embodied Foundational Model for Physical Agentic AI", "url": "https://arxiv.org/abs/2607.04426", "published": "2026-07-05", "updated": "2026-07-05", "authors": [ "ACE-Brain Team", ":", "Ziyang Gong", "Haoming Gu", "Zehang Luo", "Tianyi Zhang", "Tao Tao", "Yixiao Chi", "Zhe Liu", "Lingsi Zhu", "Jingyuan Liu", "Anke Tang", "Songze Li", "Yilun Kong", "Ningjing Liu", "Tianyu Zhu", "Yunpeng Qing", "Shuang Luo", "Xiang Liu", "Shi Fu", "Dawei Nie", "Sixiang Liu", "Zhexi Wen", "Feng Pan", "Xiaofeng Wang", "Zhi Hou", "Chunxiao Liu", "Xue Yang", "Junchi Yan", "Hengshuang Zhao", "Dacheng Tao", "Xiaogang Wang" ], "categories": [ "cs.RO" ], "topics": [ "agent-evaluation", "embodied-agent", "memory", "planning", "rag", "reasoning", "tool-use" ], "score": 17, "relevance": "high", "matched_queries": [ "agentic-ai" ], "arxiv_id": "2607.04426", "source": "arxiv", "source_id": "arxiv:2607.04426", "pdf_url": "https://arxiv.org/pdf/2607.04426", "primary_query": "agentic-ai" }, { "id": "2607.04162", "title": "ACE: Agentic Control for Embodied Manipulation via Zero-shot Workflow Reasoning", "url": "https://arxiv.org/abs/2607.04162", "published": "2026-07-05", "updated": "2026-07-05", "authors": [ "Iok Tong Lei", "QianZhi Li", "Ying Jie Yap", "Yujie Zhang", "Rui Zhong", "Haichao Gui", "Xiaolong Liu", "Zhidong Deng" ], "categories": [ "cs.RO", "cs.LG" ], "topics": [ "agent-evaluation", "embodied-agent", "memory", "planning", "rag", "reasoning", "tool-use", "workflow-agent" ], "score": 17, "relevance": "high", "matched_queries": [ "agentic-ai" ], "arxiv_id": "2607.04162", "source": "arxiv", "source_id": "arxiv:2607.04162", "pdf_url": "https://arxiv.org/pdf/2607.04162", "primary_query": "agentic-ai" }, { "id": "2607.03333", "title": "SPORK: Self-Speculative Forking to Accelerate Agentic LLM Inference", "url": "https://arxiv.org/abs/2607.03333", "published": "2026-07-03", "updated": "2026-07-03", "authors": [ "Huajun Bai", "Weiwei Lv", "Huichuan Zheng", "Youyou Lu", "Jiwu Shu" ], "categories": [ "cs.DC", "cs.AI", "cs.LG" ], "topics": [ "agent-evaluation", "coding-agent", "reasoning", "tool-use", "workflow-agent" ], "score": 17, "relevance": "high", "matched_queries": [ "llm-agent" ], "arxiv_id": "2607.03333", "source": "arxiv", "source_id": "arxiv:2607.03333", "pdf_url": "https://arxiv.org/pdf/2607.03333", "primary_query": "llm-agent" }, { "id": "2607.02857", "title": "MOSAIC: Knowledge-Guided CLI Command Composition Attack in LLM Coding Agents", "url": "https://arxiv.org/abs/2607.02857", "published": "2026-07-03", "updated": "2026-07-03", "authors": [ "Jiangrong Wu", "Huaijin Wang", "Yihao Zhang", "Yuhong Nan", "Shuai Wang" ], "categories": [ "cs.CR", "cs.SE" ], "topics": [ "agent-evaluation", "agent-safety", "coding-agent", "computer-use", "workflow-agent" ], "score": 17, "relevance": "high", "matched_queries": [ "agent-evaluation", "coding-agent" ], "arxiv_id": "2607.02857", "source": "arxiv", "source_id": "arxiv:2607.02857", "pdf_url": "https://arxiv.org/pdf/2607.02857", "primary_query": "agent-evaluation" }, { "id": "2607.01935", "title": "A-TMA: Decoupling State-Aware Memory Failures in Long-Term Agent Memory", "url": "https://arxiv.org/abs/2607.01935", "published": "2026-07-02", "updated": "2026-07-02", "authors": [ "Zitong Shi", "Yixuan Tang", "Anthony Kum Hoe Tung" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "memory", "rag" ], "score": 17, "relevance": "high", "matched_queries": [ "agent-memory", "llm-agent" ], "arxiv_id": "2607.01935", "source": "arxiv", "source_id": "arxiv:2607.01935", "pdf_url": "https://arxiv.org/pdf/2607.01935", "primary_query": "agent-memory" }, { "id": "2607.01640", "title": "AgentFlow: Building Agent Dependency Graphs for Static Analysis of Agent Programs", "url": "https://arxiv.org/abs/2607.01640", "published": "2026-07-02", "updated": "2026-07-02", "authors": [ "Shenao Wang", "Xinyi Hou", "Yanjie Zhao", "Xiao Cheng", "Haoyu Wang" ], "categories": [ "cs.SE", "cs.CR" ], "topics": [ "agent-evaluation", "agent-safety", "memory", "multi-agent", "tool-use" ], "score": 17, "relevance": "high", "matched_queries": [ "llm-agent", "multi-agent-llm" ], "arxiv_id": "2607.01640", "source": "arxiv", "source_id": "arxiv:2607.01640", "pdf_url": "https://arxiv.org/pdf/2607.01640", "primary_query": "llm-agent" }, { "id": "2607.01668", "title": "VeriChat: An Agentic Conversational AI Assistant for Hardware Security Verification", "url": "https://arxiv.org/abs/2607.01668", "published": "2026-07-02", "updated": "2026-07-02", "authors": [ "Dipayan Saha", "Khan Thamid Hasan", "Shams Tarek", "Sujan Kumar Saha", "Mark Tehranipoor", "Farimah Farahmandi" ], "categories": [ "cs.CR" ], "topics": [ "agent-evaluation", "agent-safety", "computer-use", "multi-agent", "rag", "tool-use", "workflow-agent", "world-model" ], "score": 17, "relevance": "high", "matched_queries": [ "agentic-ai" ], "arxiv_id": "2607.01668", "source": "arxiv", "source_id": "arxiv:2607.01668", "pdf_url": "https://arxiv.org/pdf/2607.01668", "primary_query": "agentic-ai" }, { "id": "2607.02294", "title": "Coding Agents Are Guessing: Measuring Action-Boundary Violations in Underspecified DevOps Instructions", "url": "https://arxiv.org/abs/2607.02294", "published": "2026-07-02", "updated": "2026-07-02", "authors": [ "Zimo Ji", "Zekai Zhang", "Congying Xu", "Zongjie Li", "Yudong Gao", "Shuai Wang", "Shing-Chi Cheung" ], "categories": [ "cs.SE" ], "topics": [ "agent-evaluation", "agent-safety", "coding-agent", "tool-use" ], "score": 17, "relevance": "high", "matched_queries": [ "agent-evaluation", "coding-agent" ], "arxiv_id": "2607.02294", "source": "arxiv", "source_id": "arxiv:2607.02294", "pdf_url": "https://arxiv.org/pdf/2607.02294", "primary_query": "agent-evaluation" }, { "id": "2607.01916", "title": "ContextSniper: AntTrail's Token-Efficient Code Memory for Repository-Level Program Repair", "url": "https://arxiv.org/abs/2607.01916", "published": "2026-07-02", "updated": "2026-07-06", "authors": [ "Chiwang Luk", "Matin Mohammad Najafi", "Zhifeng Jia", "Wei Yang", "Xiuchang Li", "Jinwei Zhu", "Yang Ren", "Lei Chen", "Gao Cong" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "coding-agent", "memory", "rag", "tool-use" ], "score": 17, "relevance": "high", "matched_queries": [ "agent-memory", "coding-agent", "rag-agent" ], "arxiv_id": "2607.01916", "source": "arxiv", "source_id": "arxiv:2607.01916", "pdf_url": "https://arxiv.org/pdf/2607.01916", "primary_query": "agent-memory" }, { "id": "2607.01929", "title": "Beyond Textual Repository Exploration: Dual-Modal Structural Reasoning for Agentic Issue Resolution", "url": "https://arxiv.org/abs/2607.01929", "published": "2026-07-02", "updated": "2026-07-02", "authors": [ "Jiayi Zhang", "Kai Huang", "Yang Liu", "Chunyang Chen" ], "categories": [ "cs.SE" ], "topics": [ "agent-evaluation", "coding-agent", "embodied-agent", "planning", "rag", "reasoning", "tool-use" ], "score": 17, "relevance": "high", "matched_queries": [ "coding-agent", "function-calling" ], "arxiv_id": "2607.01929", "source": "arxiv", "source_id": "arxiv:2607.01929", "pdf_url": "https://arxiv.org/pdf/2607.01929", "primary_query": "coding-agent" }, { "id": "2607.01071", "title": "MemSyco-Bench: Benchmarking Sycophancy in Agent Memory", "url": "https://arxiv.org/abs/2607.01071", "published": "2026-07-01", "updated": "2026-07-02", "authors": [ "Zhishang Xiang", "Zerui Chen", "Yunbo Tang", "Zhimin Wei", "Ruqin Ning", "Yujie Lin", "Qinggang Zhang", "Jinsong Su" ], "categories": [ "cs.IR", "cs.AI" ], "topics": [ "agent-evaluation", "agent-safety", "memory", "reasoning" ], "score": 17, "relevance": "high", "matched_queries": [ "agent-memory" ], "arxiv_id": "2607.01071", "source": "arxiv", "source_id": "arxiv:2607.01071", "pdf_url": "https://arxiv.org/pdf/2607.01071", "primary_query": "agent-memory" }, { "id": "2607.00939", "title": "Leveraging LLM-Based Agentic Systems to Generate Quantum Applications for Test Optimization", "url": "https://arxiv.org/abs/2607.00939", "published": "2026-07-01", "updated": "2026-07-01", "authors": [ "Ming Tao", "Yuechen Li", "Tao Yue", "Man Zhang", "Aitor Arrieta Marcos" ], "categories": [ "cs.SE", "quant-ph" ], "topics": [ "agent-evaluation", "coding-agent", "multi-agent", "rag", "workflow-agent" ], "score": 17, "relevance": "high", "matched_queries": [ "multi-agent-llm" ], "arxiv_id": "2607.00939", "source": "arxiv", "source_id": "arxiv:2607.00939", "pdf_url": "https://arxiv.org/pdf/2607.00939", "primary_query": "multi-agent-llm" }, { "id": "2607.00334", "title": "Managed Autonomy at Runtime: Gear-Based Safety and Governance for Single- and Multi-Agent Cyber-Physical Systems", "url": "https://arxiv.org/abs/2607.00334", "published": "2026-07-01", "updated": "2026-07-01", "authors": [ "Srini Ramaswamy", "Wang Miaosheng" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "agent-safety", "embodied-agent", "multi-agent", "planning", "tool-use" ], "score": 17, "relevance": "high", "matched_queries": [ "autonomous-agent-llm", "multi-agent-llm" ], "arxiv_id": "2607.00334", "source": "arxiv", "source_id": "arxiv:2607.00334", "pdf_url": "https://arxiv.org/pdf/2607.00334", "primary_query": "autonomous-agent-llm" }, { "id": "2607.05428", "title": "CHARLIE: An On-Premise Multi-Agent Retrieval-Augmented Generation System for Evidential Reasoning in Forensic Science", "url": "https://arxiv.org/abs/2607.05428", "published": "2026-07-01", "updated": "2026-07-01", "authors": [ "Leandro D. Carneiro", "Andre L. S. Meirelles", "Juliano de A. Gomes", "Rafael C. A. Cabral" ], "categories": [ "cs.DL", "cs.AI" ], "topics": [ "agent-evaluation", "memory", "multi-agent", "planning", "rag", "reasoning", "tool-use", "workflow-agent" ], "score": 17, "relevance": "high", "matched_queries": [ "rag-agent" ], "arxiv_id": "2607.05428", "source": "arxiv", "source_id": "arxiv:2607.05428", "pdf_url": "https://arxiv.org/pdf/2607.05428", "primary_query": "rag-agent" }, { "id": "2606.31693", "title": "ShopX: A Foundation Model for Intent-to-Item Fulfillment in Agentic Shopping", "url": "https://arxiv.org/abs/2606.31693", "published": "2026-06-30", "updated": "2026-06-30", "authors": [ "Jiacheng Chen", "Tao Zhang", "Manxi Lin", "Dunxian Huang", "Teng Shi", "Honghao Fu", "Mengyan Li", "Xinming Zhang", "Chenchi Zhang", "Xuan Lu", "Xiaoxiong Du", "Haibin Chen", "Shaolin Ye", "Hao Chang", "Xiaoqi Li", "Shuwen Xiao", "Yujin Yuan", "Jingxuan Feng", "Shaopan Xiong", "Huimin Yi", "Ju Huang", "Qiu Shen", "Ying Chen", "Junjun Zheng", "Xiangheng Kong", "Yuning Jiang" ], "categories": [ "cs.IR", "cs.AI", "cs.CL" ], "topics": [ "agent-evaluation", "planning", "rag", "tool-use", "workflow-agent" ], "score": 17, "relevance": "high", "matched_queries": [ "llm-agent", "planning-agent" ], "arxiv_id": "2606.31693", "source": "arxiv", "source_id": "arxiv:2606.31693", "pdf_url": "https://arxiv.org/pdf/2606.31693", "primary_query": "llm-agent" }, { "id": "2606.31252", "title": "Embodied CAD: Solver-Grounded LLM Agents for Parametric B-Rep Assembly Modeling", "url": "https://arxiv.org/abs/2606.31252", "published": "2026-06-30", "updated": "2026-06-30", "authors": [ "Fumin Liu", "Haoyu Zhou", "Fei Hao", "Lin Yang" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "embodied-agent", "planning", "tool-use", "workflow-agent" ], "score": 17, "relevance": "high", "matched_queries": [ "llm-agent", "planning-agent" ], "arxiv_id": "2606.31252", "source": "arxiv", "source_id": "arxiv:2606.31252", "pdf_url": "https://arxiv.org/pdf/2606.31252", "primary_query": "llm-agent" }, { "id": "2606.31174", "title": "ClawArena-Team: Benchmarking Subagent Orchestration and Dynamic Workflows in Language-Model Agents", "url": "https://arxiv.org/abs/2606.31174", "published": "2026-06-30", "updated": "2026-07-02", "authors": [ "Kaiwen Xiong", "Haonian Ji", "Shi Qiu", "Zeyu Zheng", "Cihang Xie", "Xinyu Ye", "Huaxiu Yao" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "multi-agent", "tool-use", "workflow-agent" ], "score": 17, "relevance": "high", "matched_queries": [ "llm-agent", "multi-agent-llm" ], "arxiv_id": "2606.31174", "source": "arxiv", "source_id": "arxiv:2606.31174", "pdf_url": "https://arxiv.org/pdf/2606.31174", "primary_query": "llm-agent" }, { "id": "2606.31046", "title": "OpenLife: Toward Open-World Artificial Life with Autonomous LLM Agents", "url": "https://arxiv.org/abs/2606.31046", "published": "2026-06-30", "updated": "2026-06-30", "authors": [ "Atsushi Masumori", "Itsuki Doi", "Norihiro Maruyama", "Ryosuke Takata", "Takashi Ikegami" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "memory", "tool-use" ], "score": 17, "relevance": "high", "matched_queries": [ "llm-agent", "tool-use" ], "arxiv_id": "2606.31046", "source": "arxiv", "source_id": "arxiv:2606.31046", "pdf_url": "https://arxiv.org/pdf/2606.31046", "primary_query": "llm-agent" }, { "id": "2606.31650", "title": "ECHO: Prune to act, trace to learn with selective turn memory in agentic RL", "url": "https://arxiv.org/abs/2606.31650", "published": "2026-06-30", "updated": "2026-06-30", "authors": [ "Zijun Xie", "Binbin Zheng", "Enlei Gong", "Jihua Liu", "Yuyang You", "Lingfeng Liu", "Jiayao Tang", "Guanqun Zhao", "Aoqi Hu", "Zeyu Chen" ], "categories": [ "cs.LG", "cs.AI" ], "topics": [ "agent-evaluation", "memory", "planning", "tool-use" ], "score": 17, "relevance": "high", "matched_queries": [ "language-agent" ], "arxiv_id": "2606.31650", "source": "arxiv", "source_id": "arxiv:2606.31650", "pdf_url": "https://arxiv.org/pdf/2606.31650", "primary_query": "language-agent" }, { "id": "2606.31639", "title": "A Lifecycle and Application-Stack Survey of Large Language Model Vulnerabilities: Attacks, Risks, Defenses, and Open Problems", "url": "https://arxiv.org/abs/2606.31639", "published": "2026-06-30", "updated": "2026-06-30", "authors": [ "Seyed Bagher Hashemi Natanzi", "Bo Tang" ], "categories": [ "cs.CR", "cs.AI", "cs.GT", "cs.LO" ], "topics": [ "agent-evaluation", "agent-safety", "coding-agent", "embodied-agent", "memory", "planning", "rag", "tool-use", "workflow-agent" ], "score": 17, "relevance": "high", "matched_queries": [ "agent-evaluation", "autonomous-agent-llm" ], "arxiv_id": "2606.31639", "source": "arxiv", "source_id": "arxiv:2606.31639", "pdf_url": "https://arxiv.org/pdf/2606.31639", "primary_query": "agent-evaluation" }, { "id": "2606.30566", "title": "Forensic Trajectory Signatures for Agent Memory Poisoning Detection", "url": "https://arxiv.org/abs/2606.30566", "published": "2026-06-29", "updated": "2026-06-29", "authors": [ "Jun Wen Leong" ], "categories": [ "cs.CR", "cs.LG" ], "topics": [ "agent-evaluation", "computer-use", "memory", "rag", "tool-use" ], "score": 17, "relevance": "high", "matched_queries": [ "agent-memory", "llm-agent" ], "arxiv_id": "2606.30566", "source": "arxiv", "source_id": "arxiv:2606.30566", "pdf_url": "https://arxiv.org/pdf/2606.30566", "primary_query": "agent-memory" }, { "id": "2606.29914", "title": "MemDelta: Controlled Baselines and Hidden Confounds in Agent Memory Evaluation", "url": "https://arxiv.org/abs/2606.29914", "published": "2026-06-29", "updated": "2026-06-29", "authors": [ "Kuan Wang" ], "categories": [ "cs.CL", "cs.LG" ], "topics": [ "agent-evaluation", "memory", "rag" ], "score": 17, "relevance": "high", "matched_queries": [ "agent-memory", "rag-agent" ], "arxiv_id": "2606.29914", "source": "arxiv", "source_id": "arxiv:2606.29914", "pdf_url": "https://arxiv.org/pdf/2606.29914", "primary_query": "agent-memory" }, { "id": "2606.30555", "title": "Linguistic Firewall: Geometry as Defense in Multi-Agent Systems Routing", "url": "https://arxiv.org/abs/2606.30555", "published": "2026-06-29", "updated": "2026-07-05", "authors": [ "Dvir Alsheich", "Adar Peleg", "Ben Hagag", "Rom Himelstein", "Amit Levi", "Avi Mendelson" ], "categories": [ "cs.AI", "cs.MA" ], "topics": [ "agent-evaluation", "agent-safety", "computer-use", "multi-agent", "tool-use", "workflow-agent" ], "score": 17, "relevance": "high", "matched_queries": [ "multi-agent-llm" ], "arxiv_id": "2606.30555", "source": "arxiv", "source_id": "arxiv:2606.30555", "pdf_url": "https://arxiv.org/pdf/2606.30555", "primary_query": "multi-agent-llm" }, { "id": "2606.30546", "title": "MAS-Lab: A Specification-Driven Validation Framework for Reliable Multi-Agent Systems", "url": "https://arxiv.org/abs/2606.30546", "published": "2026-06-29", "updated": "2026-06-29", "authors": [ "Jordan Augé", "Giovanna Carofiglio", "Giulio Grassi", "Jacques Samain" ], "categories": [ "cs.MA" ], "topics": [ "agent-evaluation", "memory", "multi-agent", "tool-use", "workflow-agent" ], "score": 17, "relevance": "high", "matched_queries": [ "multi-agent-llm" ], "arxiv_id": "2606.30546", "source": "arxiv", "source_id": "arxiv:2606.30546", "pdf_url": "https://arxiv.org/pdf/2606.30546", "primary_query": "multi-agent-llm" }, { "id": "2606.30259", "title": "Multi-Agentic System Leveraging Open-Source LLMs to Mitigate Disinformation Threats", "url": "https://arxiv.org/abs/2606.30259", "published": "2026-06-29", "updated": "2026-06-29", "authors": [ "Sebastian Kula", "Martin Tamajka" ], "categories": [ "cs.CL" ], "topics": [ "agent-evaluation", "multi-agent", "rag" ], "score": 17, "relevance": "high", "matched_queries": [ "multi-agent-llm" ], "arxiv_id": "2606.30259", "source": "arxiv", "source_id": "arxiv:2606.30259", "pdf_url": "https://arxiv.org/pdf/2606.30259", "primary_query": "multi-agent-llm" }, { "id": "2606.29742", "title": "MicroAgent: Context-Augmented Multi-Agent Framework for Automatic Microservice Decomposition", "url": "https://arxiv.org/abs/2606.29742", "published": "2026-06-29", "updated": "2026-06-29", "authors": [ "Zishan Su", "Junjie Huang", "Shiwen Shan", "Xingyan Chen", "Hui Zeng", "Yuxin Su", "Yanlin Wang", "Michael R. Lyu" ], "categories": [ "cs.SE" ], "topics": [ "agent-evaluation", "coding-agent", "computer-use", "multi-agent", "rag", "tool-use" ], "score": 17, "relevance": "high", "matched_queries": [ "multi-agent-llm" ], "arxiv_id": "2606.29742", "source": "arxiv", "source_id": "arxiv:2606.29742", "pdf_url": "https://arxiv.org/pdf/2606.29742", "primary_query": "multi-agent-llm" }, { "id": "2606.29270", "title": "Minority Sentinel: When to Overturn Majority Voting in Multi-Agent LLM Debates", "url": "https://arxiv.org/abs/2606.29270", "published": "2026-06-28", "updated": "2026-06-28", "authors": [ "Chuan He", "Zebin Chen", "Zhengyi Yang", "Shaobo Qiao", "Mingchen Ju", "Jiate Liu", "Dong Wen", "Guanfeng Liu" ], "categories": [ "cs.MA" ], "topics": [ "agent-evaluation", "agent-safety", "multi-agent", "reasoning" ], "score": 17, "relevance": "high", "matched_queries": [ "llm-agent", "multi-agent-llm" ], "arxiv_id": "2606.29270", "source": "arxiv", "source_id": "arxiv:2606.29270", "pdf_url": "https://arxiv.org/pdf/2606.29270", "primary_query": "llm-agent" }, { "id": "2606.29654", "title": "Budgeted Act-or-Defer Multi-Agent LLM Deliberation with Local Reliability Bounds", "url": "https://arxiv.org/abs/2606.29654", "published": "2026-06-28", "updated": "2026-06-28", "authors": [ "Mengdie Flora Wang", "Haochen Xie", "Guanghui Wang", "Devin Zhang", "Jae Oh Woo" ], "categories": [ "cs.AI", "cs.MA" ], "topics": [ "agent-evaluation", "agent-safety", "multi-agent", "reasoning", "tool-use", "workflow-agent" ], "score": 17, "relevance": "high", "matched_queries": [ "multi-agent-llm" ], "arxiv_id": "2606.29654", "source": "arxiv", "source_id": "arxiv:2606.29654", "pdf_url": "https://arxiv.org/pdf/2606.29654", "primary_query": "multi-agent-llm" }, { "id": "2606.28958", "title": "When Latent Agents Lie: KV-Cache Integrity in Multi-Agent LLM Collaboration", "url": "https://arxiv.org/abs/2606.28958", "published": "2026-06-27", "updated": "2026-06-27", "authors": [ "Luís Brito", "Carlos Baquero" ], "categories": [ "cs.MA" ], "topics": [ "agent-safety", "memory", "multi-agent", "reasoning" ], "score": 17, "relevance": "high", "matched_queries": [ "llm-agent", "multi-agent-llm" ], "arxiv_id": "2606.28958", "source": "arxiv", "source_id": "arxiv:2606.28958", "pdf_url": "https://arxiv.org/pdf/2606.28958", "primary_query": "llm-agent" }, { "id": "2606.28666", "title": "Why Trust Your Agent? Empirical Security Gains from TRiSM-Guided Agentic Workflows in Healthcare", "url": "https://arxiv.org/abs/2606.28666", "published": "2026-06-27", "updated": "2026-06-27", "authors": [ "Liam Kearns" ], "categories": [ "cs.CR", "cs.AI" ], "topics": [ "agent-evaluation", "agent-safety", "computer-use", "rag", "tool-use", "workflow-agent" ], "score": 17, "relevance": "high", "matched_queries": [ "agentic-ai", "rag-agent" ], "arxiv_id": "2606.28666", "source": "arxiv", "source_id": "arxiv:2606.28666", "pdf_url": "https://arxiv.org/pdf/2606.28666", "primary_query": "agentic-ai" }, { "id": "2606.27990", "title": "AdvancedShelLM: A Stateful Multi-Agent LLM Honeypot for SSH Deception", "url": "https://arxiv.org/abs/2606.27990", "published": "2026-06-26", "updated": "2026-06-26", "authors": [ "Muris Sladić", "Eman Alibalić", "Veronica Valeros", "Carlos Catania", "Sebastian Garcia" ], "categories": [ "cs.CR" ], "topics": [ "agent-evaluation", "memory", "multi-agent", "tool-use" ], "score": 17, "relevance": "high", "matched_queries": [ "llm-agent", "multi-agent-llm" ], "arxiv_id": "2606.27990", "source": "arxiv", "source_id": "arxiv:2606.27990", "pdf_url": "https://arxiv.org/pdf/2606.27990", "primary_query": "llm-agent" }, { "id": "2606.27929", "title": "When Multi-Robot Systems Meet Agentic AI:Towards Embodied Collective Intelligence", "url": "https://arxiv.org/abs/2606.27929", "published": "2026-06-26", "updated": "2026-06-26", "authors": [ "Yuxuan Yan", "Yuanyuan Jia", "Qianqian Yang" ], "categories": [ "cs.RO" ], "topics": [ "agent-evaluation", "embodied-agent", "memory", "multi-agent", "tool-use" ], "score": 17, "relevance": "high", "matched_queries": [ "ai-agent" ], "arxiv_id": "2606.27929", "source": "arxiv", "source_id": "arxiv:2606.27929", "pdf_url": "https://arxiv.org/pdf/2606.27929", "primary_query": "ai-agent" }, { "id": "2606.28434", "title": "SWE-MeM: Learning Adaptive Memory Management for Long-Horizon Coding Agents", "url": "https://arxiv.org/abs/2606.28434", "published": "2026-06-26", "updated": "2026-06-26", "authors": [ "Shuzheng Gao", "Wenhao Zeng", "Zhaojian Yu", "Jianqiao Wangni", "Chaozheng Wang", "Kai Cai", "Shilin He", "Michael R. Lyu" ], "categories": [ "cs.SE", "cs.AI" ], "topics": [ "coding-agent", "memory", "planning", "tool-use", "workflow-agent" ], "score": 17, "relevance": "high", "matched_queries": [ "coding-agent" ], "arxiv_id": "2606.28434", "source": "arxiv", "source_id": "arxiv:2606.28434", "pdf_url": "https://arxiv.org/pdf/2606.28434", "primary_query": "coding-agent" }, { "id": "2606.28570", "title": "Digitizing Coaching Intelligence: An Agentic Framework for Holistic Athlete Profiling using VLM and RAG", "url": "https://arxiv.org/abs/2606.28570", "published": "2026-06-26", "updated": "2026-06-26", "authors": [ "Deep Ghosal", "Ishani Sen", "Wazib Ansar", "Amlan Chakrabarti" ], "categories": [ "cs.CV", "cs.AI", "cs.MA" ], "topics": [ "agent-evaluation", "multi-agent", "rag", "reasoning", "tool-use" ], "score": 17, "relevance": "high", "matched_queries": [ "multi-agent-llm", "rag-agent" ], "arxiv_id": "2606.28570", "source": "arxiv", "source_id": "arxiv:2606.28570", "pdf_url": "https://arxiv.org/pdf/2606.28570", "primary_query": "multi-agent-llm" }, { "id": "2606.28182", "title": "LLawCo: Learning Laws of Cooperation for Modeling Embodied Multi-Agent Behavior", "url": "https://arxiv.org/abs/2606.28182", "published": "2026-06-26", "updated": "2026-06-26", "authors": [ "Qinhong Zhou", "Chuang Gan", "Anoop Cherian" ], "categories": [ "cs.LG", "cs.AI", "cs.CV", "cs.RO" ], "topics": [ "agent-evaluation", "embodied-agent", "multi-agent", "planning", "rag", "reasoning" ], "score": 17, "relevance": "high", "matched_queries": [ "multi-agent-llm" ], "arxiv_id": "2606.28182", "source": "arxiv", "source_id": "arxiv:2606.28182", "pdf_url": "https://arxiv.org/pdf/2606.28182", "primary_query": "multi-agent-llm" }, { "id": "2606.25514", "title": "Unlocking Model Potentials Through Adaptive Multi-Agent Scaffolding for Efficient Issue Resolution", "url": "https://arxiv.org/abs/2606.25514", "published": "2026-06-24", "updated": "2026-06-24", "authors": [ "Yang Chen", "Aliya Ahmad", "Yiheng Zhou", "Reyhaneh Jabbarvand" ], "categories": [ "cs.SE" ], "topics": [ "agent-evaluation", "coding-agent", "multi-agent", "planning", "rag", "tool-use", "workflow-agent" ], "score": 17, "relevance": "high", "matched_queries": [ "coding-agent" ], "arxiv_id": "2606.25514", "source": "arxiv", "source_id": "arxiv:2606.25514", "pdf_url": "https://arxiv.org/pdf/2606.25514", "primary_query": "coding-agent" }, { "id": "2606.25588", "title": "IntentTester: Intent-Driven Multi-agent Framework for Cross-Library Test Migration", "url": "https://arxiv.org/abs/2606.25588", "published": "2026-06-24", "updated": "2026-06-24", "authors": [ "Yi Gao", "Ziyuan Zhang", "Xing Hu", "Xiaohu Yang", "Xin Xia" ], "categories": [ "cs.SE" ], "topics": [ "agent-evaluation", "agent-safety", "coding-agent", "computer-use", "multi-agent", "reasoning", "tool-use" ], "score": 17, "relevance": "high", "matched_queries": [ "multi-agent-llm" ], "arxiv_id": "2606.25588", "source": "arxiv", "source_id": "arxiv:2606.25588", "pdf_url": "https://arxiv.org/pdf/2606.25588", "primary_query": "multi-agent-llm" }, { "id": "2606.25400", "title": "BrainAgent: A Large Language Model-Driven Multi-Agent Framework for Autonomous Brain Signal Understanding", "url": "https://arxiv.org/abs/2606.25400", "published": "2026-06-24", "updated": "2026-06-24", "authors": [ "Yangxuan Zhou", "Sha Zhao", "Jiquan Wang", "Shijian Li", "Gang Pan" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "multi-agent", "planning", "tool-use", "workflow-agent" ], "score": 17, "relevance": "high", "matched_queries": [ "multi-agent-llm" ], "arxiv_id": "2606.25400", "source": "arxiv", "source_id": "arxiv:2606.25400", "pdf_url": "https://arxiv.org/pdf/2606.25400", "primary_query": "multi-agent-llm" }, { "id": "2606.24775", "title": "Are We Ready For An Agent-Native Memory System?", "url": "https://arxiv.org/abs/2606.24775", "published": "2026-06-23", "updated": "2026-06-23", "authors": [ "Wei Zhou", "Xuanhe Zhou", "Shaokun Han", "Hongming Xu", "Guoliang Li", "Zhiyu Li", "Feiyu Xiong", "Fan Wu" ], "categories": [ "cs.CL", "cs.DB", "cs.IR" ], "topics": [ "agent-evaluation", "memory", "planning", "rag", "tool-use" ], "score": 17, "relevance": "high", "matched_queries": [ "agent-memory" ], "arxiv_id": "2606.24775", "source": "arxiv", "source_id": "arxiv:2606.24775", "pdf_url": "https://arxiv.org/pdf/2606.24775", "primary_query": "agent-memory" }, { "id": "2606.24649", "title": "Agentic Collaborative Cognition for Zero-Shot 3D Understanding", "url": "https://arxiv.org/abs/2606.24649", "published": "2026-06-23", "updated": "2026-06-25", "authors": [ "Wenxin Wang", "Bo Zhang", "Feng Chen", "Zixuan Wang", "Wen Li", "Changsheng Li", "Yinjie Lei" ], "categories": [ "cs.CV" ], "topics": [ "agent-evaluation", "computer-use", "multi-agent", "planning", "rag" ], "score": 17, "relevance": "high", "matched_queries": [ "multi-agent-llm" ], "arxiv_id": "2606.24649", "source": "arxiv", "source_id": "arxiv:2606.24649", "pdf_url": "https://arxiv.org/pdf/2606.24649", "primary_query": "multi-agent-llm" }, { "id": "2606.24437", "title": "ReM-MoA: Reasoning Memory Sustains Mixture-of-Agents Scaling", "url": "https://arxiv.org/abs/2606.24437", "published": "2026-06-23", "updated": "2026-06-23", "authors": [ "Heng Ping", "Arijit Bhattacharjee", "Peiyu Zhang", "Shixuan Li", "Wei Yang", "Ali Jannesari", "Nesreen Ahmed", "Paul Bogdan" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "memory", "multi-agent", "reasoning" ], "score": 17, "relevance": "high", "matched_queries": [ "multi-agent-llm" ], "arxiv_id": "2606.24437", "source": "arxiv", "source_id": "arxiv:2606.24437", "pdf_url": "https://arxiv.org/pdf/2606.24437", "primary_query": "multi-agent-llm" }, { "id": "2606.22647", "title": "RAVEN: Agentic RAG for Automated Vulnerability Repair", "url": "https://arxiv.org/abs/2606.22647", "published": "2026-06-21", "updated": "2026-06-21", "authors": [ "Varun Gadey", "Zijie Liu", "Alexandra Dmitrienko" ], "categories": [ "cs.CR", "cs.LG", "cs.SE" ], "topics": [ "agent-evaluation", "coding-agent", "computer-use", "memory", "rag" ], "score": 17, "relevance": "high", "matched_queries": [ "rag-agent" ], "arxiv_id": "2606.22647", "source": "arxiv", "source_id": "arxiv:2606.22647", "pdf_url": "https://arxiv.org/pdf/2606.22647", "primary_query": "rag-agent" }, { "id": "2606.22330", "title": "Hypothesis-Driven Skill Optimization for LLM Agents", "url": "https://arxiv.org/abs/2606.22330", "published": "2026-06-21", "updated": "2026-06-21", "authors": [ "Fangxin Shang", "Yehui Yang" ], "categories": [ "cs.AI", "cs.SE" ], "topics": [ "agent-safety", "coding-agent", "memory", "planning", "reasoning", "tool-use" ], "score": 17, "relevance": "high", "matched_queries": [ "planning-agent" ], "arxiv_id": "2606.22330", "source": "arxiv", "source_id": "arxiv:2606.22330", "pdf_url": "https://arxiv.org/pdf/2606.22330", "primary_query": "planning-agent" }, { "id": "2606.21740", "title": "Training the Orchestrator: A Supervised Approach to End-to-End PDDL Planning with LLM Agents", "url": "https://arxiv.org/abs/2606.21740", "published": "2026-06-19", "updated": "2026-06-19", "authors": [ "Rajesh Mangannavar", "Zachary Coalson", "Pranay Dugar", "Prasad Tadepalli" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "computer-use", "planning", "reasoning", "tool-use" ], "score": 17, "relevance": "high", "matched_queries": [ "planning-agent" ], "arxiv_id": "2606.21740", "source": "arxiv", "source_id": "arxiv:2606.21740", "pdf_url": "https://arxiv.org/pdf/2606.21740", "primary_query": "planning-agent" }, { "id": "2606.18467", "title": "ToolChain-CRC: Conformal Risk Control for Agentic AI Under Retrieval and Tool-Use Drift", "url": "https://arxiv.org/abs/2606.18467", "published": "2026-06-16", "updated": "2026-06-16", "authors": [ "Jeffery Opoku", "David Banahene" ], "categories": [ "stat.ML", "cs.LG" ], "topics": [ "agent-evaluation", "agent-safety", "rag", "tool-use" ], "score": 17, "relevance": "high", "matched_queries": [ "agent-evaluation", "agentic-ai", "ai-agent", "rag-agent", "tool-use" ], "arxiv_id": "2606.18467", "source": "arxiv", "source_id": "arxiv:2606.18467", "pdf_url": "https://arxiv.org/pdf/2606.18467", "primary_query": "agent-evaluation" }, { "id": "2606.18142", "title": "Your AI Travel Agent Would Book You a Bullfight: An Agentic Benchmark for Implicit Animal Welfare in Frontier AI Models", "url": "https://arxiv.org/abs/2606.18142", "published": "2026-06-16", "updated": "2026-07-06", "authors": [ "Jasmine Brazilek", "Joel Christoph", "Maheep Chaudhary", "Oliver Tullio", "Carol Kline", "Miles Tidmarsh", "Arturs Kanepajs" ], "categories": [ "cs.AI", "cs.CL", "cs.CY" ], "topics": [ "agent-evaluation", "agent-safety" ], "score": 17, "relevance": "high", "matched_queries": [ "agent-evaluation" ], "arxiv_id": "2606.18142", "source": "arxiv", "source_id": "arxiv:2606.18142", "pdf_url": "https://arxiv.org/pdf/2606.18142", "primary_query": "agent-evaluation" }, { "id": "2606.16591", "title": "SING: Synthetic Intention Graph for Scalable Active Tool Discovery in LLM Agents", "url": "https://arxiv.org/abs/2606.16591", "published": "2026-06-15", "updated": "2026-06-16", "authors": [ "Qiao Xiao", "Haochen Shi", "Yisen Gao", "Wenbin Hu", "Huihao Jing", "Tianshi Zheng", "Baixuan Xu", "Ziheng Zhang", "Weiqi Wang", "Haoran Li", "Jiaxin Bai", "Yangqiu Song" ], "categories": [ "cs.CL" ], "topics": [ "agent-evaluation", "multi-agent", "planning", "rag", "tool-use" ], "score": 17, "relevance": "high", "matched_queries": [ "tool-use" ], "arxiv_id": "2606.16591", "source": "arxiv", "source_id": "arxiv:2606.16591", "pdf_url": "https://arxiv.org/pdf/2606.16591", "primary_query": "tool-use" }, { "id": "2606.15684", "title": "Multi-agent Framework for Time-Sensitive Complementary Collaboration in Minecraft", "url": "https://arxiv.org/abs/2606.15684", "published": "2026-06-14", "updated": "2026-06-14", "authors": [ "Juheon Yi", "Jinglu Wang", "Xiaoyi Zhang", "Yan Lu" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "agent-safety", "multi-agent", "reasoning", "tool-use" ], "score": 17, "relevance": "high", "matched_queries": [ "agent-evaluation" ], "arxiv_id": "2606.15684", "source": "arxiv", "source_id": "arxiv:2606.15684", "pdf_url": "https://arxiv.org/pdf/2606.15684", "primary_query": "agent-evaluation" }, { "id": "2606.15931", "title": "DeepRoot: A KG-Coordinated Multi-Agent System for Therapeutic Reasoning over Historical Medical Texts", "url": "https://arxiv.org/abs/2606.15931", "published": "2026-06-14", "updated": "2026-06-14", "authors": [ "Zijian Carl Ma", "Sean J. Wang", "Sijbren Kramer", "Li Erran Li" ], "categories": [ "cs.MA", "cs.AI" ], "topics": [ "agent-evaluation", "multi-agent", "rag", "reasoning", "tool-use" ], "score": 17, "relevance": "high", "matched_queries": [ "tool-use" ], "arxiv_id": "2606.15931", "source": "arxiv", "source_id": "arxiv:2606.15931", "pdf_url": "https://arxiv.org/pdf/2606.15931", "primary_query": "tool-use" }, { "id": "2606.12703", "title": "SMSR: Certified Defence Against Runtime Memory Poisoning in Persistent LLM Agent Systems", "url": "https://arxiv.org/abs/2606.12703", "published": "2026-06-10", "updated": "2026-06-10", "authors": [ "Tarun Sharma" ], "categories": [ "cs.CR", "cs.AI", "cs.LG" ], "topics": [ "agent-evaluation", "memory", "rag", "workflow-agent" ], "score": 17, "relevance": "high", "matched_queries": [ "rag-agent" ], "arxiv_id": "2606.12703", "source": "arxiv", "source_id": "arxiv:2606.12703", "pdf_url": "https://arxiv.org/pdf/2606.12703", "primary_query": "rag-agent" }, { "id": "2606.10684", "title": "Divide and Cooperate: Role-Decomposed Multi-Agent LLM Training with Cross-Agent Learning Signals", "url": "https://arxiv.org/abs/2606.10684", "published": "2026-06-09", "updated": "2026-06-09", "authors": [ "Jaewan Park", "Solbee Cho", "Jay-Yoon Lee" ], "categories": [ "cs.LG", "cs.AI" ], "topics": [ "agent-evaluation", "multi-agent", "reasoning", "tool-use" ], "score": 17, "relevance": "high", "matched_queries": [ "language-agent" ], "arxiv_id": "2606.10684", "source": "arxiv", "source_id": "arxiv:2606.10684", "pdf_url": "https://arxiv.org/pdf/2606.10684", "primary_query": "language-agent" }, { "id": "2606.10616", "title": "Learning What to Remember: Observability-Safe Memory Retention via Constrained Optimization for Long-Horizon Language Agents", "url": "https://arxiv.org/abs/2606.10616", "published": "2026-06-09", "updated": "2026-06-29", "authors": [ "Qingcan Kang", "Liu Mingyang", "Shixiong Kai", "Kaichao Liang", "Tao Zhong", "Mingxuan Yuan" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "computer-use", "memory", "planning", "reasoning", "tool-use" ], "score": 17, "relevance": "high", "matched_queries": [ "language-agent" ], "arxiv_id": "2606.10616", "source": "arxiv", "source_id": "arxiv:2606.10616", "pdf_url": "https://arxiv.org/pdf/2606.10616", "primary_query": "language-agent" }, { "id": "2606.10933", "title": "Frontier Coding Agents Use Metaprogramming to Adapt to Unfamiliar Programming Languages", "url": "https://arxiv.org/abs/2606.10933", "published": "2026-06-09", "updated": "2026-06-09", "authors": [ "Aman Sharma", "Sushrut Thorat", "Paras Chopra" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "coding-agent", "computer-use", "tool-use" ], "score": 17, "relevance": "high", "matched_queries": [ "agent-evaluation" ], "arxiv_id": "2606.10933", "source": "arxiv", "source_id": "arxiv:2606.10933", "pdf_url": "https://arxiv.org/pdf/2606.10933", "primary_query": "agent-evaluation" }, { "id": "2606.10577", "title": "AgenticNav: Zero-Shot Vision-and-Language Navigation as a Tool-Calling Harness", "url": "https://arxiv.org/abs/2606.10577", "published": "2026-06-09", "updated": "2026-06-09", "authors": [ "Yijian Li", "Changze Li", "Hantian Shi", "Jiaying Luo", "Jiyuan Cai", "Ming Yang", "Tong Qin" ], "categories": [ "cs.RO" ], "topics": [ "agent-evaluation", "embodied-agent", "memory", "rag", "tool-use" ], "score": 17, "relevance": "high", "matched_queries": [ "agent-memory" ], "arxiv_id": "2606.10577", "source": "arxiv", "source_id": "arxiv:2606.10577", "pdf_url": "https://arxiv.org/pdf/2606.10577", "primary_query": "agent-memory" }, { "id": "2606.10304", "title": "MIRAGE: A Polarity-Flipping Encoding Subspace in LLM Agents", "url": "https://arxiv.org/abs/2606.10304", "published": "2026-06-09", "updated": "2026-06-09", "authors": [ "Pratibha Revankar", "Kargi Chauhan", "Jihye Kim", "Sadiba Nusrat Nur", "Vincent Siu", "Chenguang Wang" ], "categories": [ "cs.CL" ], "topics": [ "agent-evaluation", "coding-agent", "computer-use", "planning", "rag", "tool-use" ], "score": 17, "relevance": "high", "matched_queries": [ "planning-agent" ], "arxiv_id": "2606.10304", "source": "arxiv", "source_id": "arxiv:2606.10304", "pdf_url": "https://arxiv.org/pdf/2606.10304", "primary_query": "planning-agent" }, { "id": "2606.07682", "title": "SWE-Marathon: Can Agents Autonomously Complete Ultra-Long-Horizon Software Work?", "url": "https://arxiv.org/abs/2606.07682", "published": "2026-06-05", "updated": "2026-06-05", "authors": [ "Rishi Desai", "Jesse Hu", "Joan Cabezas", "Neel Harsola", "Pratyush Shukla", "Roey Ben Chaim", "Adnan El Assadi", "Omkaar Mukund Kamath", "Fenil Faldu", "Prannay Hebbar", "Jiankai Sun", "Yiyuan Li", "Pramod Srinivasan", "Ishan Gupta", "Christopher Settles", "Daniel Wang", "Derek Chen", "Pranav Raja", "Albert Liu", "Marek Šuppa", "Nevasini Sasikumar", "Luyang Kong", "Erik Quintanilla", "Xiangyi Li", "Ivan Bercovich", "Steven Dillmann" ], "categories": [ "cs.SE", "cs.AI" ], "topics": [ "agent-evaluation", "coding-agent", "memory", "planning", "rag", "reasoning", "workflow-agent" ], "score": 17, "relevance": "high", "matched_queries": [ "agent-evaluation" ], "arxiv_id": "2606.07682", "source": "arxiv", "source_id": "arxiv:2606.07682", "pdf_url": "https://arxiv.org/pdf/2606.07682", "primary_query": "agent-evaluation" }, { "id": "2606.07867", "title": "The Cold-Start Safety Gap in LLM Agents", "url": "https://arxiv.org/abs/2606.07867", "published": "2026-06-05", "updated": "2026-06-05", "authors": [ "Chung-En Sun", "Linbo Liu", "Tsui-Wei Weng" ], "categories": [ "cs.CL" ], "topics": [ "agent-evaluation", "agent-safety", "tool-use" ], "score": 17, "relevance": "high", "matched_queries": [ "agent-safety" ], "arxiv_id": "2606.07867", "source": "arxiv", "source_id": "arxiv:2606.07867", "pdf_url": "https://arxiv.org/pdf/2606.07867", "primary_query": "agent-safety" }, { "id": "2606.07711", "title": "Rosetta Memory: Adaptive Memory for Cross-LLM Agents", "url": "https://arxiv.org/abs/2606.07711", "published": "2026-06-05", "updated": "2026-06-05", "authors": [ "Hao Yang", "Shiqi Shen", "Haoxuan Li", "Zhipeng Wang", "Zhi Gong", "Xu Chen" ], "categories": [ "cs.LG", "cs.AI" ], "topics": [ "coding-agent", "memory", "planning", "reasoning" ], "score": 17, "relevance": "high", "matched_queries": [ "planning-agent" ], "arxiv_id": "2606.07711", "source": "arxiv", "source_id": "arxiv:2606.07711", "pdf_url": "https://arxiv.org/pdf/2606.07711", "primary_query": "planning-agent" }, { "id": "2606.06054", "title": "Beyond Similarity: Trustworthy Memory Search for Personal AI Agents", "url": "https://arxiv.org/abs/2606.06054", "published": "2026-06-04", "updated": "2026-06-04", "authors": [ "Jiawen Zhang", "Kejia Chen", "Jiachen Ma", "Yangfan Hu", "Lipeng He", "Yechao Zhang", "Jian Liu", "Xiaohu Yang", "Tianwei Zhang", "Ruoxi Jia" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "memory", "tool-use" ], "score": 17, "relevance": "high", "matched_queries": [ "agent-memory" ], "arxiv_id": "2606.06054", "source": "arxiv", "source_id": "arxiv:2606.06054", "pdf_url": "https://arxiv.org/pdf/2606.06054", "primary_query": "agent-memory" }, { "id": "2606.05805", "title": "From Risk Classification to Action Plan Remediation: A Guardrail Feedback Driven Framework for LLM Agents", "url": "https://arxiv.org/abs/2606.05805", "published": "2026-06-04", "updated": "2026-06-04", "authors": [ "Yuhao Sun", "Jiacheng Zhang", "Shaanan Cohney", "Zhexin Zhang", "Feng Liu", "Xingliang Yuan" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "agent-safety", "computer-use", "planning", "rag", "tool-use" ], "score": 17, "relevance": "high", "matched_queries": [ "planning-agent" ], "arxiv_id": "2606.05805", "source": "arxiv", "source_id": "arxiv:2606.05805", "pdf_url": "https://arxiv.org/pdf/2606.05805", "primary_query": "planning-agent" }, { "id": "2606.04599", "title": "Plan First, Judge Later, Run Better: A DMAIC-Inspired Agentic System for Industrial Anomaly Detection", "url": "https://arxiv.org/abs/2606.04599", "published": "2026-06-03", "updated": "2026-06-03", "authors": [ "Yongzi Yu", "Ao Li", "Le Wang", "Ziyue Li", "Fugee Tsung", "Yuxuan Liang", "Man Li" ], "categories": [ "cs.AI", "cs.CE" ], "topics": [ "agent-safety", "multi-agent", "planning", "rag", "workflow-agent" ], "score": 17, "relevance": "high", "matched_queries": [ "planning-agent" ], "arxiv_id": "2606.04599", "source": "arxiv", "source_id": "arxiv:2606.04599", "pdf_url": "https://arxiv.org/pdf/2606.04599", "primary_query": "planning-agent" }, { "id": "2606.04051", "title": "RUBAS: Rubric-Based Reinforcement Learning for Agent Safety", "url": "https://arxiv.org/abs/2606.04051", "published": "2026-06-02", "updated": "2026-06-02", "authors": [ "Xian Qi Loye", "Qinglin Su", "Zhexin Zhang", "Shiyao Cui", "Qi Zhu", "Fei Mi", "Hongning Wang", "Minlie Huang" ], "categories": [ "cs.LG", "cs.AI", "cs.CR" ], "topics": [ "agent-evaluation", "agent-safety", "tool-use" ], "score": 17, "relevance": "high", "matched_queries": [ "agent-safety" ], "arxiv_id": "2606.04051", "source": "arxiv", "source_id": "arxiv:2606.04051", "pdf_url": "https://arxiv.org/pdf/2606.04051", "primary_query": "agent-safety" }, { "id": "2606.02812", "title": "Traj-Evolve: A Self-Evolving Multi-Agent System for Patient Trajectory Modeling in Lung Cancer Early Detection", "url": "https://arxiv.org/abs/2606.02812", "published": "2026-06-01", "updated": "2026-06-01", "authors": [ "Sihang Zeng", "Matthew Thompson", "Ruth Etzioni", "Meliha Yetisgen" ], "categories": [ "cs.AI", "cs.CL" ], "topics": [ "agent-evaluation", "agent-safety", "memory", "multi-agent", "rag", "reasoning" ], "score": 17, "relevance": "high", "matched_queries": [ "agent-memory" ], "arxiv_id": "2606.02812", "source": "arxiv", "source_id": "arxiv:2606.02812", "pdf_url": "https://arxiv.org/pdf/2606.02812", "primary_query": "agent-memory" }, { "id": "2606.01528", "title": "Joint Agent Memory and Exploration Learning via Novelty Signals", "url": "https://arxiv.org/abs/2606.01528", "published": "2026-06-01", "updated": "2026-06-01", "authors": [ "Shizuo Tian", "Xiaohong Weng", "Rui Kong", "Yuxuan Chen", "Guohong Liu", "Yuebing Song", "Jiacheng Liu", "Yuchen Li", "Dawei Yin", "Ting Cao", "Yunxin Liu", "Yuanchun Li" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "computer-use", "memory", "rag", "tool-use" ], "score": 17, "relevance": "high", "matched_queries": [ "agent-memory" ], "arxiv_id": "2606.01528", "source": "arxiv", "source_id": "arxiv:2606.01528", "pdf_url": "https://arxiv.org/pdf/2606.01528", "primary_query": "agent-memory" }, { "id": "2606.01041", "title": "ExpWeaver: LLM Agents Learn from Experience via Latent RAG", "url": "https://arxiv.org/abs/2606.01041", "published": "2026-05-31", "updated": "2026-05-31", "authors": [ "Tao Feng", "Tianyang Luo", "Jingjun Xu", "Zhigang Hua", "Yan Xie", "Shuang Yang", "Ge Liu", "Jiaxuan You" ], "categories": [ "cs.CL" ], "topics": [ "agent-evaluation", "coding-agent", "planning", "rag", "reasoning", "tool-use" ], "score": 17, "relevance": "high", "matched_queries": [ "planning-agent", "rag-agent" ], "arxiv_id": "2606.01041", "source": "arxiv", "source_id": "arxiv:2606.01041", "pdf_url": "https://arxiv.org/pdf/2606.01041", "primary_query": "planning-agent" }, { "id": "2605.30907", "title": "BlueFin: Benchmarking LLM Agents on Financial Spreadsheets", "url": "https://arxiv.org/abs/2605.30907", "published": "2026-05-29", "updated": "2026-05-29", "authors": [ "Srivatsa Kundurthy", "Clara Na", "Colton Moraine", "Anoushka Mohta", "Case Winter", "George Fang", "John Ling", "Emma Strubell", "Zach Kirshner" ], "categories": [ "cs.SE", "cs.AI", "cs.CL", "cs.LG" ], "topics": [ "agent-evaluation", "rag" ], "score": 17, "relevance": "high", "matched_queries": [ "agent-evaluation" ], "arxiv_id": "2605.30907", "source": "arxiv", "source_id": "arxiv:2605.30907", "pdf_url": "https://arxiv.org/pdf/2605.30907", "primary_query": "agent-evaluation" }, { "id": "2605.30858", "title": "ForecastCompass: Guiding Agentic Forecasting with Adaptive Factor Memory", "url": "https://arxiv.org/abs/2605.30858", "published": "2026-05-29", "updated": "2026-05-29", "authors": [ "Yurui Chang", "Yongkang Du", "Yuanpu Cao", "Jinghui Chen", "Lu Lin" ], "categories": [ "cs.LG" ], "topics": [ "agent-evaluation", "computer-use", "memory", "rag", "reasoning", "tool-use" ], "score": 17, "relevance": "high", "matched_queries": [ "agent-memory" ], "arxiv_id": "2605.30858", "source": "arxiv", "source_id": "arxiv:2605.30858", "pdf_url": "https://arxiv.org/pdf/2605.30858", "primary_query": "agent-memory" }, { "id": "2605.30058", "title": "HEART-Bench: Do LLM Agents Exhibit Human-like Psychology?", "url": "https://arxiv.org/abs/2605.30058", "published": "2026-05-28", "updated": "2026-05-28", "authors": [ "Weihan Peng", "Chenxu Zhang", "Qianao Wang", "Yuling Shi", "Heng Lian", "Qihong Mao", "Jiahao Pang", "Chunliang Feng", "Bowen Li", "Xiaodong Gu" ], "categories": [ "cs.CL" ], "topics": [ "agent-evaluation", "computer-use", "memory", "planning", "reasoning", "tool-use" ], "score": 17, "relevance": "high", "matched_queries": [ "planning-agent" ], "arxiv_id": "2605.30058", "source": "arxiv", "source_id": "arxiv:2605.30058", "pdf_url": "https://arxiv.org/pdf/2605.30058", "primary_query": "planning-agent" }, { "id": "2605.27762", "title": "PEAM: Parametric Embodied Agent Memory through Contrastive Internalization of Experience in Minecraft", "url": "https://arxiv.org/abs/2605.27762", "published": "2026-05-26", "updated": "2026-06-01", "authors": [ "Yuchen Guo", "Junli Gong", "Weicheng Wang", "Hongmin Cai", "Yiu-ming Cheung", "Weifeng Su" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "embodied-agent", "memory", "planning", "rag", "reasoning", "tool-use" ], "score": 17, "relevance": "high", "matched_queries": [ "agent-memory" ], "arxiv_id": "2605.27762", "source": "arxiv", "source_id": "arxiv:2605.27762", "pdf_url": "https://arxiv.org/pdf/2605.27762", "primary_query": "agent-memory" }, { "id": "2605.24659", "title": "IterInject: Indirect Prompt Injection Against LLM Agents via Feedback-Guided Iterative Optimization", "url": "https://arxiv.org/abs/2605.24659", "published": "2026-05-23", "updated": "2026-05-23", "authors": [ "Zixuan Chen", "Jiaxiang Chen", "Li Luo", "Ke Xu", "Xiaoxiang Huang", "Tanfeng Sun", "Xinghao Jiang" ], "categories": [ "cs.LG" ], "topics": [ "agent-safety", "coding-agent", "computer-use", "planning", "tool-use" ], "score": 17, "relevance": "high", "matched_queries": [ "planning-agent" ], "arxiv_id": "2605.24659", "source": "arxiv", "source_id": "arxiv:2605.24659", "pdf_url": "https://arxiv.org/pdf/2605.24659", "primary_query": "planning-agent" }, { "id": "2605.23723", "title": "MemAudit: Post-hoc Auditing of Poisoned Agent Memory via Causal Attribution and Structural Anomaly Detection", "url": "https://arxiv.org/abs/2605.23723", "published": "2026-05-22", "updated": "2026-05-22", "authors": [ "Zhewen Tan", "Yilun Yao", "Huiyan Jin", "Wenhan Yu", "Guoan Wang", "Mengyuan Fan", "liang lu", "Feng Liu", "Xiangzheng Zhang", "Duohe Ma", "Tong Yang", "Lin Sun" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "agent-safety", "memory", "planning", "reasoning", "tool-use" ], "score": 17, "relevance": "high", "matched_queries": [ "agent-memory" ], "arxiv_id": "2605.23723", "source": "arxiv", "source_id": "arxiv:2605.23723", "pdf_url": "https://arxiv.org/pdf/2605.23723", "primary_query": "agent-memory" }, { "id": "2605.24219", "title": "Beyond Final Answers: Auditing Trajectory-Level Hallucinations in Multi-Agent Industrial Workflows", "url": "https://arxiv.org/abs/2605.24219", "published": "2026-05-22", "updated": "2026-05-26", "authors": [ "Harshada Badave", "Santosh Borse", "Andrea Gomez", "Harshitha Narahari", "Sara Carter", "Vishwa Bhatt", "Aishani Rachakonda", "Shuxin Lin", "Dhaval Patel" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "multi-agent", "tool-use", "workflow-agent" ], "score": 17, "relevance": "high", "matched_queries": [ "autonomous-agent-llm" ], "arxiv_id": "2605.24219", "source": "arxiv", "source_id": "arxiv:2605.24219", "pdf_url": "https://arxiv.org/pdf/2605.24219", "primary_query": "autonomous-agent-llm" }, { "id": "2605.22154", "title": "IdleSpec: Exploiting Idle Time via Speculative Planning for LLM Agents", "url": "https://arxiv.org/abs/2605.22154", "published": "2026-05-21", "updated": "2026-05-21", "authors": [ "Daewon Choi", "Kyunghyun Park", "Woomin Song", "Saket Dingliwal", "Sai Muralidhar Jayanthi", "Jinwoo Shin", "Aram Galstyan" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "computer-use", "planning", "rag", "reasoning", "tool-use" ], "score": 17, "relevance": "high", "matched_queries": [ "planning-agent" ], "arxiv_id": "2605.22154", "source": "arxiv", "source_id": "arxiv:2605.22154", "pdf_url": "https://arxiv.org/pdf/2605.22154", "primary_query": "planning-agent" }, { "id": "2605.21240", "title": "APEX: Autonomous Policy Exploration for Self-Evolving LLM Agents", "url": "https://arxiv.org/abs/2605.21240", "published": "2026-05-20", "updated": "2026-05-20", "authors": [ "Yibo Li", "Jiashuo Yang", "Zhi Zheng", "Zhiyuan Hu", "Yuan Sui", "Shizun Wang", "Yufei He", "Bryan Hooi" ], "categories": [ "cs.LG", "cs.AI" ], "topics": [ "agent-evaluation", "memory", "planning", "reasoning", "tool-use" ], "score": 17, "relevance": "high", "matched_queries": [ "planning-agent" ], "arxiv_id": "2605.21240", "source": "arxiv", "source_id": "arxiv:2605.21240", "pdf_url": "https://arxiv.org/pdf/2605.21240", "primary_query": "planning-agent" }, { "id": "2605.18930", "title": "OEP: Poisoning Self-Evolving LLM Agents via Locally Correct but Non-Transferable Experiences", "url": "https://arxiv.org/abs/2605.18930", "published": "2026-05-18", "updated": "2026-05-18", "authors": [ "Kaixiang Wang", "Jiong Lou", "Zhaojiacheng Zhou", "Jie Li" ], "categories": [ "cs.CR", "cs.AI", "cs.LG" ], "topics": [ "agent-evaluation", "agent-safety", "memory", "reasoning" ], "score": 17, "relevance": "high", "matched_queries": [ "agent-memory" ], "arxiv_id": "2605.18930", "source": "arxiv", "source_id": "arxiv:2605.18930", "pdf_url": "https://arxiv.org/pdf/2605.18930", "primary_query": "agent-memory" }, { "id": "2605.18284", "title": "CommitDistill: A Lightweight Knowledge-Centric Memory Layer for Software Repositories", "url": "https://arxiv.org/abs/2605.18284", "published": "2026-05-18", "updated": "2026-05-18", "authors": [ "Divya Chukkapalli", "Thejesh Avula", "Aditya Aggarwal", "Harsimran Singh", "Amith Tallanki" ], "categories": [ "cs.SE", "cs.AI" ], "topics": [ "agent-evaluation", "coding-agent", "computer-use", "memory", "rag", "tool-use" ], "score": 17, "relevance": "high", "matched_queries": [ "agent-memory" ], "arxiv_id": "2605.18284", "source": "arxiv", "source_id": "arxiv:2605.18284", "pdf_url": "https://arxiv.org/pdf/2605.18284", "primary_query": "agent-memory" }, { "id": "2605.14421", "title": "MemLineage: Lineage-Guided Enforcement for LLM Agent Memory", "url": "https://arxiv.org/abs/2605.14421", "published": "2026-05-14", "updated": "2026-05-14", "authors": [ "Ciyan Ouyang", "Rui Hou" ], "categories": [ "cs.CR", "cs.AI" ], "topics": [ "agent-evaluation", "computer-use", "memory", "tool-use" ], "score": 17, "relevance": "high", "matched_queries": [ "agent-memory" ], "arxiv_id": "2605.14421", "source": "arxiv", "source_id": "arxiv:2605.14421", "pdf_url": "https://arxiv.org/pdf/2605.14421", "primary_query": "agent-memory" }, { "id": "2605.14892", "title": "Beyond Individual Intelligence: Surveying Collaboration, Failure Attribution, and Self-Evolution in LLM-based Multi-Agent Systems", "url": "https://arxiv.org/abs/2605.14892", "published": "2026-05-14", "updated": "2026-05-15", "authors": [ "Shihao Qi", "Jie Ma", "Rui Xing", "Wei Guo", "Xiao Huang", "Zhitao Gao", "Jianhao Deng", "Jun Liu", "Lingling Zhang", "Bifan Wei", "Boqian Yang", "Pinghui Wang", "Jianwen Sun", "Jing Tao", "Yaqiang Wu", "Hui Liu", "Yu Yao", "Tongliang Liu" ], "categories": [ "cs.AI" ], "topics": [ "agent-safety", "multi-agent", "planning", "rag", "reasoning", "tool-use" ], "score": 17, "relevance": "high", "matched_queries": [ "autonomous-agent-llm" ], "arxiv_id": "2605.14892", "source": "arxiv", "source_id": "arxiv:2605.14892", "pdf_url": "https://arxiv.org/pdf/2605.14892", "primary_query": "autonomous-agent-llm" }, { "id": "2605.14527", "title": "Lang2MLIP: End-to-End Language-to-Machine Learning Interatomic Potential Development with Autonomous Agentic Workflows", "url": "https://arxiv.org/abs/2605.14527", "published": "2026-05-14", "updated": "2026-05-14", "authors": [ "Wenwen Li", "Yuki Orimo", "Nontawat Charoenphakdee" ], "categories": [ "cs.LG", "cond-mat.mtrl-sci", "physics.comp-ph" ], "topics": [ "agent-evaluation", "multi-agent", "tool-use", "workflow-agent", "world-model" ], "score": 17, "relevance": "high", "matched_queries": [ "autonomous-agent-llm" ], "arxiv_id": "2605.14527", "source": "arxiv", "source_id": "arxiv:2605.14527", "pdf_url": "https://arxiv.org/pdf/2605.14527", "primary_query": "autonomous-agent-llm" }, { "id": "2605.13481", "title": "PersonalAI 2.0: Enhancing knowledge graph traversal/retrieval with planning mechanism for Personalized LLM Agents", "url": "https://arxiv.org/abs/2605.13481", "published": "2026-05-13", "updated": "2026-05-13", "authors": [ "Mikhail Menschikov", "Matvey Iskornev", "Alexander Kharitonov", "Alina Bogdanova", "Mikhail Belkin", "Ekaterina Lisitsyna", "Artyom Sosedka", "Victoria Dochkina", "Ruslan Kostoev", "Ilia Perepechkin", "Evgeny Burnaev" ], "categories": [ "cs.CL" ], "topics": [ "agent-evaluation", "computer-use", "planning", "rag", "reasoning" ], "score": 17, "relevance": "high", "matched_queries": [ "planning-agent" ], "arxiv_id": "2605.13481", "source": "arxiv", "source_id": "arxiv:2605.13481", "pdf_url": "https://arxiv.org/pdf/2605.13481", "primary_query": "planning-agent" }, { "id": "2605.12260", "title": "PRISM: Pareto-Efficient Retrieval over Intent-Aware Structured Memory for Long-Horizon Agents", "url": "https://arxiv.org/abs/2605.12260", "published": "2026-05-12", "updated": "2026-05-22", "authors": [ "Jingyi Peng", "Zhongwei Wan", "Weiting Liu", "Qiuzhuang Sun" ], "categories": [ "cs.CL" ], "topics": [ "agent-evaluation", "memory", "planning", "rag", "tool-use" ], "score": 17, "relevance": "high", "matched_queries": [ "language-agent" ], "arxiv_id": "2605.12260", "source": "arxiv", "source_id": "arxiv:2605.12260", "pdf_url": "https://arxiv.org/pdf/2605.12260", "primary_query": "language-agent" }, { "id": "2605.11225", "title": "PIVOT: Bridging Planning and Execution in LLM Agents via Trajectory Refinement", "url": "https://arxiv.org/abs/2605.11225", "published": "2026-05-11", "updated": "2026-05-11", "authors": [ "Tuo Zhang", "Alin-Ionut Popa", "Yan Xu", "Rui Song", "Dimitrios Dimitriadis" ], "categories": [ "cs.AI", "cs.LG", "cs.MA" ], "topics": [ "agent-evaluation", "agent-safety", "coding-agent", "planning", "tool-use" ], "score": 17, "relevance": "high", "matched_queries": [ "autonomous-agent-llm", "planning-agent" ], "arxiv_id": "2605.11225", "source": "arxiv", "source_id": "arxiv:2605.11225", "pdf_url": "https://arxiv.org/pdf/2605.11225", "primary_query": "autonomous-agent-llm" }, { "id": "2605.09692", "title": "Causal state binding predicts action control in language agents", "url": "https://arxiv.org/abs/2605.09692", "published": "2026-05-10", "updated": "2026-06-01", "authors": [ "Xiao Jia" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "coding-agent", "memory", "planning", "tool-use" ], "score": 17, "relevance": "high", "matched_queries": [ "language-agent" ], "arxiv_id": "2605.09692", "source": "arxiv", "source_id": "arxiv:2605.09692", "pdf_url": "https://arxiv.org/pdf/2605.09692", "primary_query": "language-agent" }, { "id": "2605.07251", "title": "Can Agents Price a Reaction? Evaluating LLMs on Chemical Cost Reasoning", "url": "https://arxiv.org/abs/2605.07251", "published": "2026-05-08", "updated": "2026-05-08", "authors": [ "Yuyang Wu", "Yue Huang", "Shuaike Shen", "Xujian Wang", "Shuhao Zhang", "Qiyao Xue", "Weichen Liu", "Runtian Gao", "Jian Ma", "Xiangliang Zhang", "Olexandr Isayev" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "planning", "rag", "reasoning", "tool-use" ], "score": 17, "relevance": "high", "matched_queries": [ "planning-agent" ], "arxiv_id": "2605.07251", "source": "arxiv", "source_id": "arxiv:2605.07251", "pdf_url": "https://arxiv.org/pdf/2605.07251", "primary_query": "planning-agent" }, { "id": "2605.06713", "title": "Agentic AI and the Industrialization of Cyber Offense: Forecast, Consequences, and Defensive Priorities for Enterprises and the Mittelstand", "url": "https://arxiv.org/abs/2605.06713", "published": "2026-05-06", "updated": "2026-05-06", "authors": [ "Christopher Koch" ], "categories": [ "cs.CR", "cs.AI", "cs.HC" ], "topics": [ "agent-safety", "computer-use", "planning", "tool-use", "workflow-agent" ], "score": 17, "relevance": "high", "matched_queries": [ "agent-safety", "planning-agent" ], "arxiv_id": "2605.06713", "source": "arxiv", "source_id": "arxiv:2605.06713", "pdf_url": "https://arxiv.org/pdf/2605.06713", "primary_query": "agent-safety" }, { "id": "2605.02240", "title": "PhysicianBench: Evaluating LLM Agents in Real-World EHR Environments", "url": "https://arxiv.org/abs/2605.02240", "published": "2026-05-04", "updated": "2026-05-04", "authors": [ "Ruoqi Liu", "Imran Q. Mohiuddin", "Austin J. Schoeffler", "Kavita Renduchintala", "Ashwin Nayak", "Prasantha L. Vemu", "Shivam C. Vedak", "Kameron C. Black", "John L. Havlik", "Isaac Ogunmola", "Stephen P. Ma", "Roopa Dhatt", "Jonathan H. Chen" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "planning", "rag", "reasoning", "tool-use", "workflow-agent" ], "score": 17, "relevance": "high", "matched_queries": [ "planning-agent" ], "arxiv_id": "2605.02240", "source": "arxiv", "source_id": "arxiv:2605.02240", "pdf_url": "https://arxiv.org/pdf/2605.02240", "primary_query": "planning-agent" }, { "id": "2604.11557", "title": "UniToolCall: Unifying Tool-Use Representation, Data, and Evaluation for LLM Agents", "url": "https://arxiv.org/abs/2604.11557", "published": "2026-04-13", "updated": "2026-05-25", "authors": [ "Yijuan Liang", "Xinghao Chen", "Yifan Ge", "Ziyi Wu", "Hao Wu", "Changyu Zeng", "Wei Xing", "Xiaoyu Shen" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "reasoning", "tool-use" ], "score": 17, "relevance": "high", "matched_queries": [ "function-calling" ], "arxiv_id": "2604.11557", "source": "arxiv", "source_id": "arxiv:2604.11557", "pdf_url": "https://arxiv.org/pdf/2604.11557", "primary_query": "function-calling" }, { "id": "2604.10577", "title": "The Blind Spot of Agent Safety: How Benign User Instructions Expose Critical Vulnerabilities in Computer-Use Agents", "url": "https://arxiv.org/abs/2604.10577", "published": "2026-04-12", "updated": "2026-04-17", "authors": [ "Xuwei Ding", "Skylar Zhai", "Linxin Song", "Jiate Li", "Taiwei Shi", "Nicholas Meade", "Siva Reddy", "Jian Kang", "Jieyu Zhao" ], "categories": [ "cs.CR", "cs.AI" ], "topics": [ "agent-evaluation", "agent-safety", "multi-agent", "rag", "tool-use" ], "score": 17, "relevance": "high", "matched_queries": [ "agent-safety" ], "arxiv_id": "2604.10577", "source": "arxiv", "source_id": "arxiv:2604.10577", "pdf_url": "https://arxiv.org/pdf/2604.10577", "primary_query": "agent-safety" }, { "id": "2603.24257", "title": "Memory-Augmented Vision-Language Agents for Persistent and Semantically Consistent Object Captioning", "url": "https://arxiv.org/abs/2603.24257", "published": "2026-03-25", "updated": "2026-03-30", "authors": [ "Tommaso Galliena", "Stefano Rosa", "Tommaso Apicella", "Pietro Morerio", "Alessio Del Bue", "Lorenzo Natale" ], "categories": [ "cs.CV" ], "topics": [ "agent-evaluation", "embodied-agent", "memory" ], "score": 17, "relevance": "high", "matched_queries": [ "language-agent" ], "arxiv_id": "2603.24257", "source": "arxiv", "source_id": "arxiv:2603.24257", "pdf_url": "https://arxiv.org/pdf/2603.24257", "primary_query": "language-agent" }, { "id": "2603.10492", "title": "Human-AI Co-reasoning for Clinical Diagnosis with Evidence-Integrated Language Agent", "url": "https://arxiv.org/abs/2603.10492", "published": "2026-03-11", "updated": "2026-03-18", "authors": [ "Zhongzhen Huang", "Yan Ling", "Hong Chen", "Ye Feng", "Li Wu", "Linjie Mu", "Shaoting Zhang", "Xiaofan Zhang", "Kun Qian", "Xiaomu Li" ], "categories": [ "cs.CL" ], "topics": [ "agent-evaluation", "agent-safety", "multi-agent", "rag", "reasoning", "workflow-agent" ], "score": 17, "relevance": "high", "matched_queries": [ "language-agent" ], "arxiv_id": "2603.10492", "source": "arxiv", "source_id": "arxiv:2603.10492", "pdf_url": "https://arxiv.org/pdf/2603.10492", "primary_query": "language-agent" }, { "id": "2603.07496", "title": "From Thinker to Society: Security in Hierarchical Autonomy Evolution of AI Agents", "url": "https://arxiv.org/abs/2603.07496", "published": "2026-03-08", "updated": "2026-03-21", "authors": [ "Xiaolei Zhang", "Lu Zhou", "Xiaogang Xu", "Jiafei Wu", "Tianyu Du", "Heqing Huang", "Hao Peng", "Zhe Liu" ], "categories": [ "cs.CR", "cs.AI" ], "topics": [ "agent-evaluation", "agent-safety", "computer-use", "multi-agent", "reasoning", "tool-use" ], "score": 17, "relevance": "high", "matched_queries": [ "agent-safety" ], "arxiv_id": "2603.07496", "source": "arxiv", "source_id": "arxiv:2603.07496", "pdf_url": "https://arxiv.org/pdf/2603.07496", "primary_query": "agent-safety" }, { "id": "2603.03680", "title": "MAGE: Meta-Reinforcement Learning for Language Agents toward Strategic Exploration and Exploitation", "url": "https://arxiv.org/abs/2603.03680", "published": "2026-03-04", "updated": "2026-03-04", "authors": [ "Lu Yang", "Zelai Xu", "Minyang Xie", "Jiaxuan Gao", "Zhao Shok", "Yu Wang", "Yi Wu" ], "categories": [ "cs.AI" ], "topics": [ "memory", "multi-agent", "reasoning", "tool-use" ], "score": 17, "relevance": "high", "matched_queries": [ "language-agent" ], "arxiv_id": "2603.03680", "source": "arxiv", "source_id": "arxiv:2603.03680", "pdf_url": "https://arxiv.org/pdf/2603.03680", "primary_query": "language-agent" }, { "id": "2603.02711", "title": "A Natural Language Agentic Approach to Study Affective Polarization", "url": "https://arxiv.org/abs/2603.02711", "published": "2026-03-03", "updated": "2026-03-03", "authors": [ "Stephanie Anneris Malvicini", "Ewelina Gajewska", "Arda Derbent", "Katarzyna Budzynska", "Jarosław A. Chudziak", "Maria Vanina Martinez" ], "categories": [ "cs.AI" ], "topics": [ "multi-agent", "rag", "tool-use" ], "score": 17, "relevance": "high", "matched_queries": [ "language-agent" ], "arxiv_id": "2603.02711", "source": "arxiv", "source_id": "arxiv:2603.02711", "pdf_url": "https://arxiv.org/pdf/2603.02711", "primary_query": "language-agent" }, { "id": "2603.03515", "title": "The Controllability Trap: A Governance Framework for Military AI Agents", "url": "https://arxiv.org/abs/2603.03515", "published": "2026-03-03", "updated": "2026-03-03", "authors": [ "Subramanyam Sahoo" ], "categories": [ "cs.CY", "cs.AI" ], "topics": [ "agent-evaluation", "agent-safety", "planning", "tool-use", "world-model" ], "score": 17, "relevance": "high", "matched_queries": [ "agent-safety" ], "arxiv_id": "2603.03515", "source": "arxiv", "source_id": "arxiv:2603.03515", "pdf_url": "https://arxiv.org/pdf/2603.03515", "primary_query": "agent-safety" }, { "id": "2604.03242", "title": "DRAFT: Task Decoupled Latent Reasoning for Agent Safety", "url": "https://arxiv.org/abs/2604.03242", "published": "2026-02-11", "updated": "2026-02-11", "authors": [ "Lin Wang", "Junfeng Fang", "Dan Zhang", "Fei Shen", "Xiang Wang", "Tat-Seng Chua" ], "categories": [ "cs.LG" ], "topics": [ "agent-evaluation", "agent-safety", "rag", "reasoning", "tool-use" ], "score": 17, "relevance": "high", "matched_queries": [ "agent-safety" ], "arxiv_id": "2604.03242", "source": "arxiv", "source_id": "arxiv:2604.03242", "pdf_url": "https://arxiv.org/pdf/2604.03242", "primary_query": "agent-safety" }, { "id": "2603.08721", "title": "KernelCraft: Benchmarking for Agentic Close-to-Metal Kernel Generation on Emerging Hardware", "url": "https://arxiv.org/abs/2603.08721", "published": "2026-02-10", "updated": "2026-05-29", "authors": [ "Jiayi Nie", "Haoran Wu", "Yao Lai", "Zeyu Cao", "Cheng Zhang", "Binglei Lou", "Erwei Wang", "Jianyi Cheng", "Timothy M. Jones", "Robert Mullins", "Rika Antonova", "Yiren Zhao" ], "categories": [ "cs.AR", "cs.LG", "cs.SE" ], "topics": [ "agent-evaluation", "reasoning", "workflow-agent" ], "score": 17, "relevance": "high", "matched_queries": [ "function-calling" ], "arxiv_id": "2603.08721", "source": "arxiv", "source_id": "arxiv:2603.08721", "pdf_url": "https://arxiv.org/pdf/2603.08721", "primary_query": "function-calling" }, { "id": "2602.03224", "title": "TAME: A Trustworthy Test-Time Evolution of Agent Memory with Systematic Benchmarking", "url": "https://arxiv.org/abs/2602.03224", "published": "2026-02-03", "updated": "2026-06-06", "authors": [ "Yu Cheng", "Yongkang Hu", "Jiuan Zhou", "Yushuo Zhang", "Yihang Chen", "Huichi Zhou", "Mingang Chen", "Zhizhong Zhang", "Kun Shao", "Yuan Xie", "Zhaoxia Yin" ], "categories": [ "cs.AI", "cs.LG" ], "topics": [ "agent-evaluation", "agent-safety", "computer-use", "memory", "reasoning" ], "score": 17, "relevance": "high", "matched_queries": [ "agent-safety" ], "arxiv_id": "2602.03224", "source": "arxiv", "source_id": "arxiv:2602.03224", "pdf_url": "https://arxiv.org/pdf/2602.03224", "primary_query": "agent-safety" }, { "id": "2512.11682", "title": "MedAI: Evaluating TxAgent's Therapeutic Agentic Reasoning in the NeurIPS CURE-Bench Competition", "url": "https://arxiv.org/abs/2512.11682", "published": "2025-12-12", "updated": "2026-06-15", "authors": [ "Tim Cofala", "Christian Kalfar", "Jingge Xiao", "Johanna Schrader", "Michelle Tang", "Wolfgang Nejdl" ], "categories": [ "cs.AI", "cs.LG" ], "topics": [ "agent-evaluation", "agent-safety", "computer-use", "planning", "rag", "reasoning", "tool-use" ], "score": 17, "relevance": "high", "matched_queries": [ "function-calling" ], "arxiv_id": "2512.11682", "source": "arxiv", "source_id": "arxiv:2512.11682", "pdf_url": "https://arxiv.org/pdf/2512.11682", "primary_query": "function-calling" }, { "id": "2511.15203", "title": "Taxonomy, Evaluation and Exploitation of IPI-Centric LLM Agent Defense Frameworks", "url": "https://arxiv.org/abs/2511.15203", "published": "2025-11-19", "updated": "2025-11-19", "authors": [ "Zimo Ji", "Xunguang Wang", "Zongjie Li", "Pingchuan Ma", "Yudong Gao", "Daoyuan Wu", "Xincheng Yan", "Tian Tian", "Shuai Wang" ], "categories": [ "cs.CR", "cs.AI" ], "topics": [ "agent-evaluation", "agent-safety", "rag", "tool-use" ], "score": 17, "relevance": "high", "matched_queries": [ "function-calling" ], "arxiv_id": "2511.15203", "source": "arxiv", "source_id": "arxiv:2511.15203", "pdf_url": "https://arxiv.org/pdf/2511.15203", "primary_query": "function-calling" }, { "id": "2510.18586", "title": "TokenCake: A KV-Cache-centric Serving Framework for LLM-based Multi-Agent Applications", "url": "https://arxiv.org/abs/2510.18586", "published": "2025-10-21", "updated": "2026-05-20", "authors": [ "Zhuohang Bian", "Feiyang Wu", "Zhuoran Li", "Teng Ma", "Youwei Zhuo" ], "categories": [ "cs.DC" ], "topics": [ "agent-evaluation", "computer-use", "memory", "multi-agent" ], "score": 17, "relevance": "high", "matched_queries": [ "function-calling" ], "arxiv_id": "2510.18586", "source": "arxiv", "source_id": "arxiv:2510.18586", "pdf_url": "https://arxiv.org/pdf/2510.18586", "primary_query": "function-calling" }, { "id": "2607.06001", "title": "Information Limits and Attractor Dynamics in Economies of Frontier LLM Agents: A Pre-Registered Test", "url": "https://arxiv.org/abs/2607.06001", "published": "2026-07-07", "updated": "2026-07-07", "authors": [ "Cheng Qian" ], "categories": [ "cs.AI", "cs.MA" ], "topics": [ "agent-safety", "multi-agent", "reasoning", "tool-use" ], "score": 16, "relevance": "high", "matched_queries": [ "llm-agent", "multi-agent-llm" ], "arxiv_id": "2607.06001", "source": "arxiv", "source_id": "arxiv:2607.06001", "pdf_url": "https://arxiv.org/pdf/2607.06001", "primary_query": "llm-agent" }, { "id": "2607.06000", "title": "Context-to-Execution Integrity for LLM Agents", "url": "https://arxiv.org/abs/2607.06000", "published": "2026-07-07", "updated": "2026-07-07", "authors": [ "Igor Santos-Grueiro" ], "categories": [ "cs.CR" ], "topics": [ "agent-evaluation", "coding-agent", "tool-use" ], "score": 16, "relevance": "high", "matched_queries": [ "agent-evaluation", "coding-agent", "llm-agent" ], "arxiv_id": "2607.06000", "source": "arxiv", "source_id": "arxiv:2607.06000", "pdf_url": "https://arxiv.org/pdf/2607.06000", "primary_query": "agent-evaluation" }, { "id": "2607.06413", "title": "An Experimental Design Approach to Evaluating Agentic AI's Autonomous Model Discovery", "url": "https://arxiv.org/abs/2607.06413", "published": "2026-07-07", "updated": "2026-07-07", "authors": [ "Hao He", "Xueying Liu", "Chris J. Kuhlman", "Xinwei Deng" ], "categories": [ "stat.ME", "cs.AI" ], "topics": [ "agent-evaluation", "coding-agent", "reasoning" ], "score": 16, "relevance": "high", "matched_queries": [ "agentic-ai", "coding-agent" ], "arxiv_id": "2607.06413", "source": "arxiv", "source_id": "arxiv:2607.06413", "pdf_url": "https://arxiv.org/pdf/2607.06413", "primary_query": "agentic-ai" }, { "id": "2607.06411", "title": "RuBench: A Repository-Level Agentic Coding Benchmark with Natively Authored Russian Task Specifications", "url": "https://arxiv.org/abs/2607.06411", "published": "2026-07-07", "updated": "2026-07-07", "authors": [ "Evgeny Shilov" ], "categories": [ "cs.SE", "cs.AI", "cs.CL" ], "topics": [ "agent-evaluation", "coding-agent", "reasoning" ], "score": 16, "relevance": "high", "matched_queries": [ "agent-evaluation", "coding-agent" ], "arxiv_id": "2607.06411", "source": "arxiv", "source_id": "arxiv:2607.06411", "pdf_url": "https://arxiv.org/pdf/2607.06411", "primary_query": "agent-evaluation" }, { "id": "2607.06101", "title": "Agents That Teach: Towards Designing Incidental Learning Back into AI-Assisted Software Development", "url": "https://arxiv.org/abs/2607.06101", "published": "2026-07-07", "updated": "2026-07-07", "authors": [ "Rohit Mehra", "Samdyuti Suri", "Prithviraj K Tagadinamani", "Kapil Singi", "Vikrant Kaulgud", "Adam P. Burden" ], "categories": [ "cs.SE", "cs.AI", "cs.CY", "cs.HC" ], "topics": [ "agent-safety", "coding-agent", "computer-use", "multi-agent", "rag", "reasoning", "tool-use", "workflow-agent" ], "score": 16, "relevance": "high", "matched_queries": [ "coding-agent" ], "arxiv_id": "2607.06101", "source": "arxiv", "source_id": "arxiv:2607.06101", "pdf_url": "https://arxiv.org/pdf/2607.06101", "primary_query": "coding-agent" }, { "id": "2607.05659", "title": "Agents with Feelings? Personality and Emotion in Multi-Agent Software Teams", "url": "https://arxiv.org/abs/2607.05659", "published": "2026-07-06", "updated": "2026-07-06", "authors": [ "Yunyan Ding", "Thomas Zimmermann", "Iftekhar Ahmed" ], "categories": [ "cs.SE" ], "topics": [ "agent-evaluation", "coding-agent", "multi-agent", "workflow-agent" ], "score": 16, "relevance": "high", "matched_queries": [ "llm-agent", "multi-agent-llm" ], "arxiv_id": "2607.05659", "source": "arxiv", "source_id": "arxiv:2607.05659", "pdf_url": "https://arxiv.org/pdf/2607.05659", "primary_query": "llm-agent" }, { "id": "2607.04713", "title": "RSPO: Reward-Swap Policy Optimization for Multi-Turn LLM Agents", "url": "https://arxiv.org/abs/2607.04713", "published": "2026-07-06", "updated": "2026-07-06", "authors": [ "Qiang Liu", "Taian Guo", "Ruizhi Qiao", "Xing Sun" ], "categories": [ "cs.LG", "cs.AI" ], "topics": [ "agent-evaluation", "agent-safety", "planning", "rag" ], "score": 16, "relevance": "high", "matched_queries": [ "agent-evaluation", "llm-agent" ], "arxiv_id": "2607.04713", "source": "arxiv", "source_id": "arxiv:2607.04713", "pdf_url": "https://arxiv.org/pdf/2607.04713", "primary_query": "agent-evaluation" }, { "id": "2607.04569", "title": "LLMs for Agentic Home Energy Management", "url": "https://arxiv.org/abs/2607.04569", "published": "2026-07-06", "updated": "2026-07-06", "authors": [ "Sokipriala Jonah" ], "categories": [ "eess.SY" ], "topics": [ "agent-evaluation", "agent-safety", "rag", "tool-use" ], "score": 16, "relevance": "high", "matched_queries": [ "function-calling", "llm-agent" ], "arxiv_id": "2607.04569", "source": "arxiv", "source_id": "arxiv:2607.04569", "pdf_url": "https://arxiv.org/pdf/2607.04569", "primary_query": "function-calling" }, { "id": "2607.04617", "title": "MRMS: A Multi-Resolution Memory Substrate for Long-Lived AI Agents", "url": "https://arxiv.org/abs/2607.04617", "published": "2026-07-06", "updated": "2026-07-06", "authors": [ "Jizhizi Li", "Amy Shi-Nash" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "computer-use", "memory", "rag", "tool-use" ], "score": 16, "relevance": "high", "matched_queries": [ "ai-agent" ], "arxiv_id": "2607.04617", "source": "arxiv", "source_id": "arxiv:2607.04617", "pdf_url": "https://arxiv.org/pdf/2607.04617", "primary_query": "ai-agent" }, { "id": "2607.05363", "title": "SovereignPA-Bench: Evaluating User-Owned Personal Agents under Evolving Intent, Platform Mediation, and Consent Constraints", "url": "https://arxiv.org/abs/2607.05363", "published": "2026-07-06", "updated": "2026-07-06", "authors": [ "Dylan Zongmin Liu" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "agent-safety", "computer-use", "embodied-agent", "memory", "tool-use" ], "score": 16, "relevance": "high", "matched_queries": [ "agent-evaluation", "tool-use" ], "arxiv_id": "2607.05363", "source": "arxiv", "source_id": "arxiv:2607.05363", "pdf_url": "https://arxiv.org/pdf/2607.05363", "primary_query": "agent-evaluation" }, { "id": "2607.05458", "title": "Learning to Control LLM Agent Harnesses with Offline Reinforcement Learning", "url": "https://arxiv.org/abs/2607.05458", "published": "2026-07-05", "updated": "2026-07-05", "authors": [ "Haiwen Yi", "Xinyuan Song" ], "categories": [ "cs.LG", "cs.AI" ], "topics": [ "agent-evaluation", "coding-agent", "reasoning", "tool-use", "workflow-agent" ], "score": 16, "relevance": "high", "matched_queries": [ "llm-agent" ], "arxiv_id": "2607.05458", "source": "arxiv", "source_id": "arxiv:2607.05458", "pdf_url": "https://arxiv.org/pdf/2607.05458", "primary_query": "llm-agent" }, { "id": "2607.04219", "title": "Agentic IoT: Architectures, Applications, and Challenges Toward the Internet of Agents", "url": "https://arxiv.org/abs/2607.04219", "published": "2026-07-05", "updated": "2026-07-05", "authors": [ "Rümeysa Hilal Sevinç", "Bahaeddin Türkoğlu", "İbrahim Kök" ], "categories": [ "cs.AI", "cs.MA", "cs.NI" ], "topics": [ "multi-agent", "planning", "reasoning", "tool-use" ], "score": 16, "relevance": "high", "matched_queries": [ "ai-agent", "tool-use" ], "arxiv_id": "2607.04219", "source": "arxiv", "source_id": "arxiv:2607.04219", "pdf_url": "https://arxiv.org/pdf/2607.04219", "primary_query": "ai-agent" }, { "id": "2607.04212", "title": "An Evaluation of Role-Based Multi-Agent Code Generation on Repository-Scale Problems", "url": "https://arxiv.org/abs/2607.04212", "published": "2026-07-05", "updated": "2026-07-05", "authors": [ "Benedetta Donato", "Noah Hagar-Dent", "Aaron Worsnop", "Leonardo Mariani", "Valerio Terragni" ], "categories": [ "cs.SE" ], "topics": [ "agent-evaluation", "coding-agent", "multi-agent" ], "score": 16, "relevance": "high", "matched_queries": [ "multi-agent-llm" ], "arxiv_id": "2607.04212", "source": "arxiv", "source_id": "arxiv:2607.04212", "pdf_url": "https://arxiv.org/pdf/2607.04212", "primary_query": "multi-agent-llm" }, { "id": "2607.04149", "title": "Beyond Scene Priors: Fine-Grained Traffic Scene Reasoning with Benchmarking and Query-Guided Small-Object Focus", "url": "https://arxiv.org/abs/2607.04149", "published": "2026-07-05", "updated": "2026-07-05", "authors": [ "Waikit Xiu", "Qiang Lu", "Zian Wang", "Xinjie Yang", "Zhiwei Chen", "Chen Sun", "Xiying Li" ], "categories": [ "cs.CV" ], "topics": [ "agent-evaluation", "agent-safety", "coding-agent", "computer-use", "multi-agent", "reasoning" ], "score": 16, "relevance": "high", "matched_queries": [ "multi-agent-llm" ], "arxiv_id": "2607.04149", "source": "arxiv", "source_id": "arxiv:2607.04149", "pdf_url": "https://arxiv.org/pdf/2607.04149", "primary_query": "multi-agent-llm" }, { "id": "2607.04034", "title": "The \"I Don't Know\" Filter: Enhancing Agentic Reliability in Function Calling", "url": "https://arxiv.org/abs/2607.04034", "published": "2026-07-04", "updated": "2026-07-04", "authors": [ "Stefan Broecker", "Mason del Rosario", "Boris Selitser", "Thomas Strohmer" ], "categories": [ "cs.SE", "cs.AI" ], "topics": [ "agent-evaluation", "rag", "tool-use" ], "score": 16, "relevance": "high", "matched_queries": [ "agent-evaluation", "function-calling" ], "arxiv_id": "2607.04034", "source": "arxiv", "source_id": "arxiv:2607.04034", "pdf_url": "https://arxiv.org/pdf/2607.04034", "primary_query": "agent-evaluation" }, { "id": "2607.03691", "title": "Don't Blame the Large Language Model: How Scaffolding Evolution Shapes Coding Agent Quality", "url": "https://arxiv.org/abs/2607.03691", "published": "2026-07-04", "updated": "2026-07-04", "authors": [ "Oussama Ben Sghaier", "Hao Li", "Bram Adams", "Ahmed E. Hassan" ], "categories": [ "cs.SE", "cs.AI", "cs.LG" ], "topics": [ "agent-evaluation", "coding-agent", "reasoning", "tool-use" ], "score": 16, "relevance": "high", "matched_queries": [ "coding-agent" ], "arxiv_id": "2607.03691", "source": "arxiv", "source_id": "arxiv:2607.03691", "pdf_url": "https://arxiv.org/pdf/2607.03691", "primary_query": "coding-agent" }, { "id": "2607.02927", "title": "VideoSearcher: Empowering Video Deep Research with Multi-Tool Agentic Reasoning via Reinforcement Learning", "url": "https://arxiv.org/abs/2607.02927", "published": "2026-07-03", "updated": "2026-07-03", "authors": [ "Zhenkun Gao", "Yicheng Bao", "Jinlong Peng", "Xueheng Li", "Theo Huang", "Bangwei Liu", "Kunquan Li", "Zhenye Gan", "Tao Hu", "Chengjun Xie", "Mingqian Yang", "Xuanhua He", "Zhizhong Zhang", "Xin Tan", "Chengjie Wang", "Yuan Xie" ], "categories": [ "cs.CV", "cs.AI" ], "topics": [ "agent-evaluation", "rag", "reasoning", "tool-use" ], "score": 16, "relevance": "high", "matched_queries": [ "tool-use" ], "arxiv_id": "2607.02927", "source": "arxiv", "source_id": "arxiv:2607.02927", "pdf_url": "https://arxiv.org/pdf/2607.02927", "primary_query": "tool-use" }, { "id": "2607.03525", "title": "GameEngineBench: Evaluating Coding Agents on Real C++ Runtime Environments", "url": "https://arxiv.org/abs/2607.03525", "published": "2026-07-03", "updated": "2026-07-03", "authors": [ "Brian La", "Sejoon Chang", "Ben Kim", "Junyoung Bae", "Aamish Ahmad Beg", "Sei Chang", "Gonzalo Gonzalez-Pumariega" ], "categories": [ "cs.SE", "cs.CL" ], "topics": [ "agent-evaluation", "coding-agent", "embodied-agent", "memory", "tool-use", "world-model" ], "score": 16, "relevance": "high", "matched_queries": [ "coding-agent" ], "arxiv_id": "2607.03525", "source": "arxiv", "source_id": "arxiv:2607.03525", "pdf_url": "https://arxiv.org/pdf/2607.03525", "primary_query": "coding-agent" }, { "id": "2607.02689", "title": "S-EMBER: A Large-Scale Benchmark for Streaming Egocentric Memory Retrieval", "url": "https://arxiv.org/abs/2607.02689", "published": "2026-07-02", "updated": "2026-07-02", "authors": [ "Xiaodong Wang", "Xuanyi Zhao", "Pedro Rodriguez", "Devendra Singh Sachan", "Barlas Oguz", "Seungwhan Moon", "Shang-Wen Li", "Gargi Ghosh", "Xin Dong", "Wen-Tau Yih" ], "categories": [ "cs.CV", "cs.AI" ], "topics": [ "agent-evaluation", "memory", "rag", "reasoning", "tool-use" ], "score": 16, "relevance": "high", "matched_queries": [ "ai-agent" ], "arxiv_id": "2607.02689", "source": "arxiv", "source_id": "arxiv:2607.02689", "pdf_url": "https://arxiv.org/pdf/2607.02689", "primary_query": "ai-agent" }, { "id": "2607.01788", "title": "KRCA: An Efficient Root Cause Analysis System in Hyper-Scale Microservice Systems via Agentic AI", "url": "https://arxiv.org/abs/2607.01788", "published": "2026-07-02", "updated": "2026-07-02", "authors": [ "Jiamin Jiang", "Jingfei Feng", "Yu Luo", "Qingliang Zhang", "Yongqian Su", "Wenwei Gu", "Shenglin Zhang", "Tianyu Cui", "Yao Wu", "Jielong Huang", "Nan Qi", "Dan Pei" ], "categories": [ "cs.SE" ], "topics": [ "agent-evaluation", "memory", "multi-agent", "rag", "reasoning", "tool-use" ], "score": 16, "relevance": "high", "matched_queries": [ "agentic-ai" ], "arxiv_id": "2607.01788", "source": "arxiv", "source_id": "arxiv:2607.01788", "pdf_url": "https://arxiv.org/pdf/2607.01788", "primary_query": "agentic-ai" }, { "id": "2607.02716", "title": "Evaluating Large Language Models for Decision-Making in Agent-Based Urban Mobility Simulations", "url": "https://arxiv.org/abs/2607.02716", "published": "2026-07-02", "updated": "2026-07-02", "authors": [ "Bruno Cascaes Alves", "Míriam Blank Born", "Ulisses Gilioli Francescatto Júnior", "Felipe Moura Goulart", "Letícia Brandão Caldas", "Marilton Sanchotene de Aguiar" ], "categories": [ "cs.MA" ], "topics": [ "agent-evaluation", "computer-use", "memory", "multi-agent", "planning", "tool-use", "world-model" ], "score": 16, "relevance": "high", "matched_queries": [ "multi-agent-llm" ], "arxiv_id": "2607.02716", "source": "arxiv", "source_id": "arxiv:2607.02716", "pdf_url": "https://arxiv.org/pdf/2607.02716", "primary_query": "multi-agent-llm" }, { "id": "2607.02186", "title": "UA-ChatDev: Uncertainty-Aware Multi-Agent Collaboration for Reliable Software Development", "url": "https://arxiv.org/abs/2607.02186", "published": "2026-07-02", "updated": "2026-07-02", "authors": [ "Temitayo Olamilekan Ogunsusi", "Lijun Qian", "Xishuang Dong" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "coding-agent", "multi-agent", "rag", "tool-use" ], "score": 16, "relevance": "high", "matched_queries": [ "multi-agent-llm" ], "arxiv_id": "2607.02186", "source": "arxiv", "source_id": "arxiv:2607.02186", "pdf_url": "https://arxiv.org/pdf/2607.02186", "primary_query": "multi-agent-llm" }, { "id": "2607.01084", "title": "Can Agents Generalize to the Open World? Unveiling the Fragility of Static Training in Tool Use", "url": "https://arxiv.org/abs/2607.01084", "published": "2026-07-01", "updated": "2026-07-01", "authors": [ "Song-Lin Lv", "Weiming Wu", "Rui Zhu", "Zi-Jian Cheng", "Lan-Zhe Guo" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "rag", "reasoning", "tool-use" ], "score": 16, "relevance": "high", "matched_queries": [ "llm-agent", "tool-use" ], "arxiv_id": "2607.01084", "source": "arxiv", "source_id": "arxiv:2607.01084", "pdf_url": "https://arxiv.org/pdf/2607.01084", "primary_query": "llm-agent" }, { "id": "2607.02599", "title": "AgentLTL: A Trace-Verification Framework for Measuring, Enforcing, and Training Procedural Compliance in Tool-Using LLM Agents", "url": "https://arxiv.org/abs/2607.02599", "published": "2026-07-01", "updated": "2026-07-01", "authors": [ "Laïla Elkoussy", "Julien Perez" ], "categories": [ "cs.SE", "cs.AI", "cs.LO" ], "topics": [ "agent-evaluation", "agent-safety", "rag", "tool-use" ], "score": 16, "relevance": "high", "matched_queries": [ "llm-agent", "tool-use" ], "arxiv_id": "2607.02599", "source": "arxiv", "source_id": "arxiv:2607.02599", "pdf_url": "https://arxiv.org/pdf/2607.02599", "primary_query": "llm-agent" }, { "id": "2607.00555", "title": "Rise From The Ashes: LLM-based Static Analysis for Deep Learning Framework Bugs", "url": "https://arxiv.org/abs/2607.00555", "published": "2026-07-01", "updated": "2026-07-01", "authors": [ "Shaoyu Yang", "Haifeng Lin", "Chunrong Fang", "Xiang Chen", "Wei Cheng", "Jiawei Liu", "Yiyu Zhang", "Hongyu Liu", "Zhenyu Chen" ], "categories": [ "cs.SE" ], "topics": [ "agent-evaluation", "agent-safety", "coding-agent", "computer-use", "multi-agent", "rag", "tool-use", "workflow-agent" ], "score": 16, "relevance": "high", "matched_queries": [ "agentic-ai", "multi-agent-llm" ], "arxiv_id": "2607.00555", "source": "arxiv", "source_id": "arxiv:2607.00555", "pdf_url": "https://arxiv.org/pdf/2607.00555", "primary_query": "agentic-ai" }, { "id": "2607.00436", "title": "PHREEQC-MCQ-200: A Diagnostic Benchmark for Tool-Augmented Scientific Simulator Agents", "url": "https://arxiv.org/abs/2607.00436", "published": "2026-07-01", "updated": "2026-07-01", "authors": [ "Ke Zhang", "Sahchit Chundur", "Mohammad Javad Qomi", "Maziar Raissi" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "rag", "tool-use", "world-model" ], "score": 16, "relevance": "high", "matched_queries": [ "tool-use" ], "arxiv_id": "2607.00436", "source": "arxiv", "source_id": "arxiv:2607.00436", "pdf_url": "https://arxiv.org/pdf/2607.00436", "primary_query": "tool-use" }, { "id": "2607.00502", "title": "A Task-State Representation for Long-Horizon Mobile GUI Agents", "url": "https://arxiv.org/abs/2607.00502", "published": "2026-07-01", "updated": "2026-07-01", "authors": [ "Yujie Zheng", "Zikang Liu", "Xin Zhao", "Ji-Rong Wen" ], "categories": [ "cs.CL" ], "topics": [ "agent-evaluation", "computer-use", "memory", "planning", "reasoning", "tool-use" ], "score": 16, "relevance": "high", "matched_queries": [ "web-gui-agent" ], "arxiv_id": "2607.00502", "source": "arxiv", "source_id": "arxiv:2607.00502", "pdf_url": "https://arxiv.org/pdf/2607.00502", "primary_query": "web-gui-agent" }, { "id": "2607.00440", "title": "Minos: A Multi-Agent Collaborative Framework for Provenance-Based Backward Tracking", "url": "https://arxiv.org/abs/2607.00440", "published": "2026-07-01", "updated": "2026-07-01", "authors": [ "Jiahui Wang", "Zhenyuan Li", "Zhengkai Wang", "Xiangmin Shen", "Fan Zhang" ], "categories": [ "cs.CR" ], "topics": [ "agent-evaluation", "agent-safety", "computer-use", "multi-agent", "rag", "reasoning" ], "score": 16, "relevance": "high", "matched_queries": [ "multi-agent-llm" ], "arxiv_id": "2607.00440", "source": "arxiv", "source_id": "arxiv:2607.00440", "pdf_url": "https://arxiv.org/pdf/2607.00440", "primary_query": "multi-agent-llm" }, { "id": "2607.00233", "title": "From Signals to Structure: How Memory Architecture Drives Language Emergence in LLM Agents", "url": "https://arxiv.org/abs/2607.00233", "published": "2026-06-30", "updated": "2026-06-30", "authors": [ "Yashar Talebirad", "Eden Redman", "Ali Parsaee", "Osmar R. Zaiane" ], "categories": [ "cs.AI", "cs.CL", "cs.IT", "cs.MA" ], "topics": [ "memory", "rag", "tool-use" ], "score": 16, "relevance": "high", "matched_queries": [ "llm-agent" ], "arxiv_id": "2607.00233", "source": "arxiv", "source_id": "arxiv:2607.00233", "pdf_url": "https://arxiv.org/pdf/2607.00233", "primary_query": "llm-agent" }, { "id": "2606.32034", "title": "QVal: Cheaply Evaluating Dense Supervision Signals for Long-Horizon LLM Agents", "url": "https://arxiv.org/abs/2606.32034", "published": "2026-06-30", "updated": "2026-06-30", "authors": [ "Sergio Hernández-Gutiérrez", "Matteo Merler", "Ilze Amanda Auzina", "Joschka Strüber", "Ameya Prabhu", "Matthias Bethge" ], "categories": [ "cs.LG", "cs.AI", "cs.CL" ], "topics": [ "agent-evaluation", "computer-use", "planning", "tool-use" ], "score": 16, "relevance": "high", "matched_queries": [ "llm-agent" ], "arxiv_id": "2606.32034", "source": "arxiv", "source_id": "arxiv:2606.32034", "pdf_url": "https://arxiv.org/pdf/2606.32034", "primary_query": "llm-agent" }, { "id": "2607.02579", "title": "When Not to Write Memory: Governing False Promotion from Correlated Agent Traces", "url": "https://arxiv.org/abs/2607.02579", "published": "2026-06-30", "updated": "2026-06-30", "authors": [ "Yijiashun Qi", "Xiang Xu", "Yuxuan Li" ], "categories": [ "cs.SE" ], "topics": [ "agent-evaluation", "agent-safety", "coding-agent", "memory", "rag", "tool-use" ], "score": 16, "relevance": "high", "matched_queries": [ "agent-memory", "language-agent" ], "arxiv_id": "2607.02579", "source": "arxiv", "source_id": "arxiv:2607.02579", "pdf_url": "https://arxiv.org/pdf/2607.02579", "primary_query": "agent-memory" }, { "id": "2606.31339", "title": "Verification-Gated Agentic Mission-State Governance for Intelligent Industrial Multi-Robot Systems", "url": "https://arxiv.org/abs/2606.31339", "published": "2026-06-30", "updated": "2026-06-30", "authors": [ "Guoqin Tang", "Qingxuan Jia", "Yichen Tan", "Zeyuan Huang", "Ning Ji", "Gang Chen" ], "categories": [ "cs.RO" ], "topics": [ "agent-evaluation", "agent-safety", "embodied-agent", "planning", "rag", "reasoning", "tool-use" ], "score": 16, "relevance": "high", "matched_queries": [ "agentic-ai" ], "arxiv_id": "2606.31339", "source": "arxiv", "source_id": "arxiv:2606.31339", "pdf_url": "https://arxiv.org/pdf/2606.31339", "primary_query": "agentic-ai" }, { "id": "2606.31980", "title": "DigitalCoach: Communication and Grounding Gaps in Human and Agentic Computer Use Coaching", "url": "https://arxiv.org/abs/2606.31980", "published": "2026-06-30", "updated": "2026-06-30", "authors": [ "Meng Chen", "Anya Ji", "Tsung-Han Wu", "Tobias Maringgele", "David M. Chan", "Alane Suhr", "Amy Pavel" ], "categories": [ "cs.CL", "cs.AI", "cs.HC" ], "topics": [ "agent-evaluation", "computer-use", "planning", "rag" ], "score": 16, "relevance": "high", "matched_queries": [ "web-gui-agent" ], "arxiv_id": "2606.31980", "source": "arxiv", "source_id": "arxiv:2606.31980", "pdf_url": "https://arxiv.org/pdf/2606.31980", "primary_query": "web-gui-agent" }, { "id": "2606.31134", "title": "Beyond the Library: An Agentic Framework for Autoformalizing Research Mathematics", "url": "https://arxiv.org/abs/2606.31134", "published": "2026-06-30", "updated": "2026-07-01", "authors": [ "Arshia Soltani Moakhar", "Iman Gholami", "Max Springer", "Mahdi JafariRaviz", "MohammadTaghi Hajiaghayi" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "coding-agent", "multi-agent", "rag", "reasoning" ], "score": 16, "relevance": "high", "matched_queries": [ "multi-agent-llm" ], "arxiv_id": "2606.31134", "source": "arxiv", "source_id": "arxiv:2606.31134", "pdf_url": "https://arxiv.org/pdf/2606.31134", "primary_query": "multi-agent-llm" }, { "id": "2606.31200", "title": "Agentic RAG-VLM: Affordance-Aware Retrieval-Augmented Generation with Self-Reflective Planning for Robotic Grasping", "url": "https://arxiv.org/abs/2606.31200", "published": "2026-06-30", "updated": "2026-06-30", "authors": [ "Tao Chen", "Lizheng Liu", "Jiaxu Wang", "Ziyue Jiang", "Ruiqi Tian", "JiGuang Huo", "Zhongxue Gan" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "embodied-agent", "planning", "rag", "reasoning" ], "score": 16, "relevance": "high", "matched_queries": [ "rag-agent" ], "arxiv_id": "2606.31200", "source": "arxiv", "source_id": "arxiv:2606.31200", "pdf_url": "https://arxiv.org/pdf/2606.31200", "primary_query": "rag-agent" }, { "id": "2606.30251", "title": "TACO: Tool-Augmented Credit Optimization for Agentic Tool Use", "url": "https://arxiv.org/abs/2606.30251", "published": "2026-06-29", "updated": "2026-06-29", "authors": [ "Mingkuan Feng", "Jinyang Wu", "Hao Gu", "Fangrui Lv", "Ruihan Jin", "Chuyuan Zhang", "Zhengqi Wen", "Jianhua Tao" ], "categories": [ "cs.MA" ], "topics": [ "agent-evaluation", "computer-use", "reasoning", "tool-use" ], "score": 16, "relevance": "high", "matched_queries": [ "tool-use" ], "arxiv_id": "2606.30251", "source": "arxiv", "source_id": "arxiv:2606.30251", "pdf_url": "https://arxiv.org/pdf/2606.30251", "primary_query": "tool-use" }, { "id": "2606.30294", "title": "Rehearsed Multi-Agent Live Product Demonstrations with Real-Time Voice Question Answering", "url": "https://arxiv.org/abs/2606.30294", "published": "2026-06-29", "updated": "2026-06-29", "authors": [ "Rahul Khedar", "Mayank Malhotra", "Avinash Karn", "Mouli V", "Prakhar Mehrotra" ], "categories": [ "cs.AI", "cs.HC", "cs.SE" ], "topics": [ "agent-evaluation", "coding-agent", "computer-use", "multi-agent", "rag", "tool-use", "workflow-agent" ], "score": 16, "relevance": "high", "matched_queries": [ "web-gui-agent" ], "arxiv_id": "2606.30294", "source": "arxiv", "source_id": "arxiv:2606.30294", "pdf_url": "https://arxiv.org/pdf/2606.30294", "primary_query": "web-gui-agent" }, { "id": "2606.29354", "title": "When LLMs Develop Languages: Symbolic Communication for Efficient Multi-Agent Reasoning", "url": "https://arxiv.org/abs/2606.29354", "published": "2026-06-28", "updated": "2026-06-28", "authors": [ "Zhengqi Pei", "Qingming Huang", "Shuhui Wang" ], "categories": [ "cs.AI", "cs.NE" ], "topics": [ "agent-evaluation", "multi-agent", "reasoning" ], "score": 16, "relevance": "high", "matched_queries": [ "llm-agent", "multi-agent-llm" ], "arxiv_id": "2606.29354", "source": "arxiv", "source_id": "arxiv:2606.29354", "pdf_url": "https://arxiv.org/pdf/2606.29354", "primary_query": "llm-agent" }, { "id": "2606.29315", "title": "Hierarchical Experimentalist Agents", "url": "https://arxiv.org/abs/2606.29315", "published": "2026-06-28", "updated": "2026-06-28", "authors": [ "Abhranil Chandra", "Sankaran Vaidyanathan", "Utsav Dhanuka", "Varun Gandhi", "Scott Niekum" ], "categories": [ "cs.AI", "cs.LG" ], "topics": [ "agent-evaluation", "planning", "rag", "reasoning", "tool-use", "world-model" ], "score": 16, "relevance": "high", "matched_queries": [ "llm-agent" ], "arxiv_id": "2606.29315", "source": "arxiv", "source_id": "arxiv:2606.29315", "pdf_url": "https://arxiv.org/pdf/2606.29315", "primary_query": "llm-agent" }, { "id": "2606.29445", "title": "Bridging VideoQA and Video-Guided Agentic Tasks via Generalized Keyframe Extraction", "url": "https://arxiv.org/abs/2606.29445", "published": "2026-06-28", "updated": "2026-06-28", "authors": [ "Sunqi Fan", "Qingle Liu", "Runqi Yin", "Meng-Hao Guo", "Shuojin Yang" ], "categories": [ "cs.CV", "cs.AI" ], "topics": [ "agent-evaluation", "computer-use", "planning", "tool-use" ], "score": 16, "relevance": "high", "matched_queries": [ "web-gui-agent" ], "arxiv_id": "2606.29445", "source": "arxiv", "source_id": "arxiv:2606.29445", "pdf_url": "https://arxiv.org/pdf/2606.29445", "primary_query": "web-gui-agent" }, { "id": "2606.28841", "title": "LAMP: Lean-based Agentic framework with MCP and Proof Repair", "url": "https://arxiv.org/abs/2606.28841", "published": "2026-06-27", "updated": "2026-06-27", "authors": [ "Santhana Srinivasan R", "Maithilee Patawar" ], "categories": [ "cs.LO", "cs.AI", "cs.CL" ], "topics": [ "coding-agent", "multi-agent", "planning", "reasoning", "tool-use" ], "score": 16, "relevance": "high", "matched_queries": [ "multi-agent-llm" ], "arxiv_id": "2606.28841", "source": "arxiv", "source_id": "arxiv:2606.28841", "pdf_url": "https://arxiv.org/pdf/2606.28841", "primary_query": "multi-agent-llm" }, { "id": "2606.28450", "title": "LLM agents security duality: a comprehensive survey of self-security and empowered cybersecurity", "url": "https://arxiv.org/abs/2606.28450", "published": "2026-06-26", "updated": "2026-06-26", "authors": [ "Yiwei Xu", "Yong Zhuang", "Xuanming Liu", "Tian Zhang", "Bowen Xiao", "Xiaoyang Xu", "Delong Jiang", "Juan Wang", "Hongxin Hu" ], "categories": [ "cs.CR", "cs.AI" ], "topics": [ "agent-evaluation", "agent-safety", "tool-use" ], "score": 16, "relevance": "high", "matched_queries": [ "agent-safety", "llm-agent", "tool-use" ], "arxiv_id": "2606.28450", "source": "arxiv", "source_id": "arxiv:2606.28450", "pdf_url": "https://arxiv.org/pdf/2606.28450", "primary_query": "agent-safety" }, { "id": "2606.28480", "title": "TUA-Bench: A Benchmark for General-Purpose Terminal-Use Agents", "url": "https://arxiv.org/abs/2606.28480", "published": "2026-06-26", "updated": "2026-06-26", "authors": [ "Shoufa Chen", "Luyuan Wang", "Xuan Yang", "Zhiheng Liu", "Yuren Cong", "Yuanfeng Ji", "Feiyan Zhou", "Xiaohui Zhang", "Fanny Yang", "Belinda Zeng" ], "categories": [ "cs.SE", "cs.AI" ], "topics": [ "agent-evaluation", "coding-agent", "computer-use", "reasoning", "workflow-agent" ], "score": 16, "relevance": "high", "matched_queries": [ "web-gui-agent" ], "arxiv_id": "2606.28480", "source": "arxiv", "source_id": "arxiv:2606.28480", "pdf_url": "https://arxiv.org/pdf/2606.28480", "primary_query": "web-gui-agent" }, { "id": "2606.27350", "title": "CHIA: An open-source framework for principled, agentic AI-driven hardware/software co-design research", "url": "https://arxiv.org/abs/2606.27350", "published": "2026-06-25", "updated": "2026-06-27", "authors": [ "Angela Cui", "Ferran Hermida-Rivera", "Jack Toubes", "Raghav Gupta", "Jim Fang", "Chengyi Lux Zhang", "Ella Schwarz", "Junha Kim", "Yakun Sophia Shao", "Borivoje Nikolic", "Christopher W. Fletcher", "Sagar Karandikar" ], "categories": [ "cs.AR" ], "topics": [ "agent-safety", "coding-agent", "tool-use", "workflow-agent" ], "score": 16, "relevance": "high", "matched_queries": [ "agentic-ai", "coding-agent" ], "arxiv_id": "2606.27350", "source": "arxiv", "source_id": "arxiv:2606.27350", "pdf_url": "https://arxiv.org/pdf/2606.27350", "primary_query": "agentic-ai" }, { "id": "2606.26721", "title": "Knowledge-Based Pull Requests: A Trusted Workflow for Agent-Mediated Knowledge Collaboration", "url": "https://arxiv.org/abs/2606.26721", "published": "2026-06-25", "updated": "2026-06-25", "authors": [ "Xinyu Zhang", "Weiwei Sun" ], "categories": [ "cs.SE", "cs.HC" ], "topics": [ "agent-evaluation", "agent-safety", "coding-agent", "memory", "multi-agent", "planning", "tool-use", "workflow-agent", "world-model" ], "score": 16, "relevance": "high", "matched_queries": [ "coding-agent" ], "arxiv_id": "2606.26721", "source": "arxiv", "source_id": "arxiv:2606.26721", "pdf_url": "https://arxiv.org/pdf/2606.26721", "primary_query": "coding-agent" }, { "id": "2606.27330", "title": "Empowering GUI Agents via Autonomous Experience Exploration and Hindsight Experience Utilization for Task Planning", "url": "https://arxiv.org/abs/2606.27330", "published": "2026-06-25", "updated": "2026-06-25", "authors": [ "Tianyi Men", "Zhuoran Jin", "Pengfei Cao", "Yubo Chen", "Kang Liu", "Jun Zhao" ], "categories": [ "cs.CL", "cs.AI", "cs.CV", "cs.LG" ], "topics": [ "agent-evaluation", "computer-use", "planning", "rag", "tool-use" ], "score": 16, "relevance": "high", "matched_queries": [ "web-gui-agent" ], "arxiv_id": "2606.27330", "source": "arxiv", "source_id": "arxiv:2606.27330", "pdf_url": "https://arxiv.org/pdf/2606.27330", "primary_query": "web-gui-agent" }, { "id": "2606.27009", "title": "Semantic Early-Stopping for Iterative LLM Agent Loops", "url": "https://arxiv.org/abs/2606.27009", "published": "2026-06-25", "updated": "2026-06-25", "authors": [ "Sahil Shrivastava" ], "categories": [ "cs.AI", "cs.LG", "cs.MA" ], "topics": [ "agent-evaluation", "multi-agent", "rag", "tool-use" ], "score": 16, "relevance": "high", "matched_queries": [ "multi-agent-llm" ], "arxiv_id": "2606.27009", "source": "arxiv", "source_id": "arxiv:2606.27009", "pdf_url": "https://arxiv.org/pdf/2606.27009", "primary_query": "multi-agent-llm" }, { "id": "2606.27483", "title": "Internalizing the Future: A Unified Agentic Training Paradigm for World Model Planning", "url": "https://arxiv.org/abs/2606.27483", "published": "2026-06-25", "updated": "2026-06-25", "authors": [ "Xuan Zhang", "Zhijian Zhou", "Lingfeng Qiao", "Yulei Qin", "Ke Li", "Xing Sun", "Xiaoyu Tan", "Chao Qu", "Yuan Qi" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "planning", "rag", "reasoning", "world-model" ], "score": 16, "relevance": "high", "matched_queries": [ "planning-agent" ], "arxiv_id": "2606.27483", "source": "arxiv", "source_id": "arxiv:2606.27483", "pdf_url": "https://arxiv.org/pdf/2606.27483", "primary_query": "planning-agent" }, { "id": "2606.26216", "title": "CyberChainBench: Can AI Agents Secure Smart Contracts Against Real-World On-Chain Vulnerabilities?", "url": "https://arxiv.org/abs/2606.26216", "published": "2026-06-24", "updated": "2026-06-24", "authors": [ "Jintao Huang", "Fengqing Jiang", "Radha Poovendran", "Zhiqiang Lin" ], "categories": [ "cs.CR", "cs.AI" ], "topics": [ "agent-evaluation", "agent-safety", "tool-use" ], "score": 16, "relevance": "high", "matched_queries": [ "agent-safety", "ai-agent" ], "arxiv_id": "2606.26216", "source": "arxiv", "source_id": "arxiv:2606.26216", "pdf_url": "https://arxiv.org/pdf/2606.26216", "primary_query": "agent-safety" }, { "id": "2606.26203", "title": "Agentic Analysis for Agentic Infrastructure: An LLM-Powered Pipeline for Comparative Governance of DAO and Corporate AI Protocols", "url": "https://arxiv.org/abs/2606.26203", "published": "2026-06-24", "updated": "2026-06-24", "authors": [ "Yutian Wang", "Luyao Zhang" ], "categories": [ "cs.AI", "cs.CY", "cs.MA", "cs.SI" ], "topics": [ "agent-safety", "coding-agent", "rag", "tool-use" ], "score": 16, "relevance": "high", "matched_queries": [ "agentic-ai", "ai-agent" ], "arxiv_id": "2606.26203", "source": "arxiv", "source_id": "arxiv:2606.26203", "pdf_url": "https://arxiv.org/pdf/2606.26203", "primary_query": "agentic-ai" }, { "id": "2606.25760", "title": "Uncertainty Quantification for Computer-Use Agents: A Benchmark across Vision-Language Models and GUI Grounding Datasets", "url": "https://arxiv.org/abs/2606.25760", "published": "2026-06-24", "updated": "2026-06-24", "authors": [ "Divake Kumar", "Sina Tayebati", "Devashri Naik", "Amanda Sofie Rios", "Nilesh Ahuja", "Omesh Tickoo", "Ranganath Krishnan", "Amit Ranjan Trivedi" ], "categories": [ "cs.LG", "cs.AI", "cs.CL", "cs.CV" ], "topics": [ "agent-evaluation", "agent-safety", "computer-use", "rag" ], "score": 16, "relevance": "high", "matched_queries": [ "agent-evaluation", "web-gui-agent" ], "arxiv_id": "2606.25760", "source": "arxiv", "source_id": "arxiv:2606.25760", "pdf_url": "https://arxiv.org/pdf/2606.25760", "primary_query": "agent-evaluation" }, { "id": "2606.25651", "title": "MedGuards: Multi-Agent System for Reliable Medical Error Detection and Correction", "url": "https://arxiv.org/abs/2606.25651", "published": "2026-06-24", "updated": "2026-06-26", "authors": [ "Congbo Ma", "Hu Wang", "Yichun Zhang", "Farah E. Shamout" ], "categories": [ "cs.CL" ], "topics": [ "agent-evaluation", "agent-safety", "computer-use", "multi-agent", "reasoning" ], "score": 16, "relevance": "high", "matched_queries": [ "multi-agent-llm" ], "arxiv_id": "2606.25651", "source": "arxiv", "source_id": "arxiv:2606.25651", "pdf_url": "https://arxiv.org/pdf/2606.25651", "primary_query": "multi-agent-llm" }, { "id": "2606.24595", "title": "MEMPROBE: Probing Long-Term Agent Memory via Hidden User-State Recovery", "url": "https://arxiv.org/abs/2606.24595", "published": "2026-06-23", "updated": "2026-06-23", "authors": [ "Enze Ma", "Yufan Zhou", "Wei-Chieh Huang", "Jie Yang", "Huanhuan Ma", "Zixuan Wang", "Chengze Li", "Chunyu Miao", "Philip S. Yu", "Zhen Wang" ], "categories": [ "cs.CL" ], "topics": [ "agent-evaluation", "memory", "rag", "tool-use" ], "score": 16, "relevance": "high", "matched_queries": [ "agent-memory" ], "arxiv_id": "2606.24595", "source": "arxiv", "source_id": "arxiv:2606.24595", "pdf_url": "https://arxiv.org/pdf/2606.24595", "primary_query": "agent-memory" }, { "id": "2606.24453", "title": "Bayesian control for coding agents", "url": "https://arxiv.org/abs/2606.24453", "published": "2026-06-23", "updated": "2026-06-23", "authors": [ "Theodore Papamarkou", "Vladislav Smirnov", "Viktor Mazanov", "Artem Vazhentsev", "Preslav Nakov", "Timothy Baldwin", "Artem Shelmanov" ], "categories": [ "cs.AI", "cs.CL" ], "topics": [ "agent-evaluation", "coding-agent", "reasoning", "tool-use" ], "score": 16, "relevance": "high", "matched_queries": [ "coding-agent", "tool-use" ], "arxiv_id": "2606.24453", "source": "arxiv", "source_id": "arxiv:2606.24453", "pdf_url": "https://arxiv.org/pdf/2606.24453", "primary_query": "coding-agent" }, { "id": "2606.24525", "title": "VisCritic: Visual State Comparison as Process Reward for GUI Agents", "url": "https://arxiv.org/abs/2606.24525", "published": "2026-06-23", "updated": "2026-06-23", "authors": [ "Jiachen Qian" ], "categories": [ "cs.CV" ], "topics": [ "agent-evaluation", "computer-use", "planning", "reasoning", "tool-use" ], "score": 16, "relevance": "high", "matched_queries": [ "web-gui-agent" ], "arxiv_id": "2606.24525", "source": "arxiv", "source_id": "arxiv:2606.24525", "pdf_url": "https://arxiv.org/pdf/2606.24525", "primary_query": "web-gui-agent" }, { "id": "2606.24689", "title": "Automated Summarization of Software Documents: An LLM-based Multi-Agent Approach", "url": "https://arxiv.org/abs/2606.24689", "published": "2026-06-23", "updated": "2026-06-23", "authors": [ "Duc S. H. Nguyen", "Minh T. Nguyen", "Phuong T. Nguyen", "Juri Di Rocco", "Davide Di Ruscio" ], "categories": [ "cs.SE" ], "topics": [ "agent-evaluation", "coding-agent", "multi-agent", "workflow-agent" ], "score": 16, "relevance": "high", "matched_queries": [ "multi-agent-llm" ], "arxiv_id": "2606.24689", "source": "arxiv", "source_id": "arxiv:2606.24689", "pdf_url": "https://arxiv.org/pdf/2606.24689", "primary_query": "multi-agent-llm" }, { "id": "2606.24623", "title": "Privacy-Preserving RAG via Multi-Agent Semantic Rewriting: Achieving Confidentiality Without Compromising Contextual Fidelity", "url": "https://arxiv.org/abs/2606.24623", "published": "2026-06-23", "updated": "2026-06-23", "authors": [ "Yuanhe Zhao", "Tianyu Zhang", "Huafei Xing", "Derek F. Wong", "Jianbin Li", "Tao Fang" ], "categories": [ "cs.CL", "cs.AI" ], "topics": [ "agent-evaluation", "agent-safety", "multi-agent", "rag", "tool-use" ], "score": 16, "relevance": "high", "matched_queries": [ "multi-agent-llm", "rag-agent" ], "arxiv_id": "2606.24623", "source": "arxiv", "source_id": "arxiv:2606.24623", "pdf_url": "https://arxiv.org/pdf/2606.24623", "primary_query": "multi-agent-llm" }, { "id": "2606.23664", "title": "MAS-PromptBench: When Does Prompt Optimization Improve Multi-Agent LLM Systems?", "url": "https://arxiv.org/abs/2606.23664", "published": "2026-06-22", "updated": "2026-06-22", "authors": [ "Juyang Bai", "Laixi Shi" ], "categories": [ "cs.LG", "cs.MA" ], "topics": [ "agent-evaluation", "multi-agent", "workflow-agent" ], "score": 16, "relevance": "high", "matched_queries": [ "agentic-ai", "multi-agent-llm" ], "arxiv_id": "2606.23664", "source": "arxiv", "source_id": "arxiv:2606.23664", "pdf_url": "https://arxiv.org/pdf/2606.23664", "primary_query": "agentic-ai" }, { "id": "2606.23032", "title": "IPO Finance Agent: Benchmark of LLM Financial Analysts Beyond Finance Agent v2, with Automated Rubric Generation, on the SpaceX (SPCX) IPO", "url": "https://arxiv.org/abs/2606.23032", "published": "2026-06-22", "updated": "2026-06-30", "authors": [ "Mostapha Benhenda" ], "categories": [ "cs.AI", "q-fin.GN" ], "topics": [ "agent-evaluation", "agent-safety", "rag", "tool-use" ], "score": 16, "relevance": "high", "matched_queries": [ "agent-evaluation" ], "arxiv_id": "2606.23032", "source": "arxiv", "source_id": "arxiv:2606.23032", "pdf_url": "https://arxiv.org/pdf/2606.23032", "primary_query": "agent-evaluation" }, { "id": "2606.22741", "title": "GRADE: Graph Representation of LLM Agent Dependency and Execution", "url": "https://arxiv.org/abs/2606.22741", "published": "2026-06-22", "updated": "2026-06-22", "authors": [ "Yue Zhao" ], "categories": [ "cs.LG" ], "topics": [ "coding-agent", "multi-agent", "tool-use" ], "score": 16, "relevance": "high", "matched_queries": [ "multi-agent-llm", "tool-use" ], "arxiv_id": "2606.22741", "source": "arxiv", "source_id": "arxiv:2606.22741", "pdf_url": "https://arxiv.org/pdf/2606.22741", "primary_query": "multi-agent-llm" }, { "id": "2606.23752", "title": "ESAA-Conversational: An Event-Sourced Memory Layer for Continuity, Handoff, and Curation Across Heterogeneous LLM Coding Agents", "url": "https://arxiv.org/abs/2606.23752", "published": "2026-06-22", "updated": "2026-06-22", "authors": [ "Elzo Brito dos Santos Filho" ], "categories": [ "cs.SE" ], "topics": [ "coding-agent", "memory", "tool-use" ], "score": 16, "relevance": "high", "matched_queries": [ "coding-agent" ], "arxiv_id": "2606.23752", "source": "arxiv", "source_id": "arxiv:2606.23752", "pdf_url": "https://arxiv.org/pdf/2606.23752", "primary_query": "coding-agent" }, { "id": "2606.23327", "title": "VideoAgent: All-in-One Framework for Video Understanding and Editing", "url": "https://arxiv.org/abs/2606.23327", "published": "2026-06-22", "updated": "2026-07-03", "authors": [ "Hengji Zhou", "Lingxuan Huang", "Jian Wang", "Bing Zhou", "Si Wu", "Lianghao Xia", "Chao Huang" ], "categories": [ "cs.CV", "cs.AI" ], "topics": [ "agent-evaluation", "multi-agent", "planning", "rag", "tool-use" ], "score": 16, "relevance": "high", "matched_queries": [ "multi-agent-llm" ], "arxiv_id": "2606.23327", "source": "arxiv", "source_id": "arxiv:2606.23327", "pdf_url": "https://arxiv.org/pdf/2606.23327", "primary_query": "multi-agent-llm" }, { "id": "2606.22110", "title": "TraceView: Interactive Visualization of Agentic Program Repair Trajectories", "url": "https://arxiv.org/abs/2606.22110", "published": "2026-06-20", "updated": "2026-06-20", "authors": [ "Amirali Sajadi", "Tu Nguyen", "Kimmie Huynh", "Esteban Parra", "Preetha Chatterjee" ], "categories": [ "cs.SE", "cs.AI", "cs.HC" ], "topics": [ "agent-evaluation", "coding-agent", "reasoning", "tool-use", "workflow-agent" ], "score": 16, "relevance": "high", "matched_queries": [ "tool-use" ], "arxiv_id": "2606.22110", "source": "arxiv", "source_id": "arxiv:2606.22110", "pdf_url": "https://arxiv.org/pdf/2606.22110", "primary_query": "tool-use" }, { "id": "2606.22082", "title": "CodeTeam: An LLM-Powered Multi-Agent Framework for Repository-Level Code Generation", "url": "https://arxiv.org/abs/2606.22082", "published": "2026-06-20", "updated": "2026-06-20", "authors": [ "Yifei Wang", "Ruiyin Li", "Peng Liang", "Qiong Feng", "Zengyang Li", "Mojtaba Shahin", "Arif Ali Khan" ], "categories": [ "cs.SE", "cs.AI" ], "topics": [ "agent-evaluation", "coding-agent", "multi-agent", "planning", "rag" ], "score": 16, "relevance": "high", "matched_queries": [ "multi-agent-llm" ], "arxiv_id": "2606.22082", "source": "arxiv", "source_id": "arxiv:2606.22082", "pdf_url": "https://arxiv.org/pdf/2606.22082", "primary_query": "multi-agent-llm" }, { "id": "2606.21963", "title": "Holmes: Multimodal Agentic Diagnosis for Mixed-Language Mobile Crashes at Industrial Scale", "url": "https://arxiv.org/abs/2606.21963", "published": "2026-06-20", "updated": "2026-06-20", "authors": [ "Jia Li", "Wenyuan Ma", "Ting Peng", "Haibin Zheng", "Yuetang Deng" ], "categories": [ "cs.AI", "cs.SE" ], "topics": [ "agent-evaluation", "computer-use", "multi-agent", "rag", "workflow-agent" ], "score": 16, "relevance": "high", "matched_queries": [ "multi-agent-llm" ], "arxiv_id": "2606.21963", "source": "arxiv", "source_id": "arxiv:2606.21963", "pdf_url": "https://arxiv.org/pdf/2606.21963", "primary_query": "multi-agent-llm" }, { "id": "2606.21445", "title": "AutoRAS: Learning Robust Agentic Systems with Primitive Representations", "url": "https://arxiv.org/abs/2606.21445", "published": "2026-06-19", "updated": "2026-06-19", "authors": [ "Yang Yue", "Xuancheng Zhu", "Yuyang Ma", "Guoshun Nan", "Zihan Dou", "Jingru Shan", "Congyu Guo", "Ji Zhang", "Hua Wang", "Jingfeng Zhang" ], "categories": [ "cs.AI" ], "topics": [ "agent-safety", "multi-agent", "reasoning", "tool-use", "workflow-agent" ], "score": 16, "relevance": "high", "matched_queries": [ "agentic-ai" ], "arxiv_id": "2606.21445", "source": "arxiv", "source_id": "arxiv:2606.21445", "pdf_url": "https://arxiv.org/pdf/2606.21445", "primary_query": "agentic-ai" }, { "id": "2606.20954", "title": "Learning What Not to Forget: Long-Horizon Agent Memory from a Few Kilobytes of Learning", "url": "https://arxiv.org/abs/2606.20954", "published": "2026-06-18", "updated": "2026-06-18", "authors": [ "Nusrat Jahan Lia", "Aritra Mazumder" ], "categories": [ "cs.CL", "cs.AI" ], "topics": [ "agent-evaluation", "memory", "planning", "tool-use" ], "score": 16, "relevance": "high", "matched_queries": [ "agent-memory" ], "arxiv_id": "2606.20954", "source": "arxiv", "source_id": "arxiv:2606.20954", "pdf_url": "https://arxiv.org/pdf/2606.20954", "primary_query": "agent-memory" }, { "id": "2606.19980", "title": "ENPIRE: Agentic Robot Policy Self-Improvement in the Real World", "url": "https://arxiv.org/abs/2606.19980", "published": "2026-06-18", "updated": "2026-06-18", "authors": [ "Wenli Xiao", "Jia Xie", "Tonghe Zhang", "Haotian Lin", "Letian \"Max\" Fu", "Haoru Xue", "Jalen Lu", "Yi Yang", "Cunxi Dai", "Zi Wang", "Jimmy Wu", "Guanzhi Wang", "S. Shankar Sastry", "Ken Goldberg", "Linxi \"Jim\" Fan", "Yuke Zhu", "Guanya Shi" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "coding-agent", "embodied-agent", "reasoning", "tool-use" ], "score": 16, "relevance": "high", "matched_queries": [ "coding-agent", "tool-use" ], "arxiv_id": "2606.19980", "source": "arxiv", "source_id": "arxiv:2606.19980", "pdf_url": "https://arxiv.org/pdf/2606.19980", "primary_query": "coding-agent" }, { "id": "2606.19245", "title": "TxBench-PP: Analyzing AI Agent Performance on Small-Molecule Preclinical Pharmacology", "url": "https://arxiv.org/abs/2606.19245", "published": "2026-06-17", "updated": "2026-06-18", "authors": [ "Hannah Le", "Ramesh Ramasamy", "Alex Urrutia", "Mahsa Yazdani", "Tim Proctor", "Kenny Workman" ], "categories": [ "cs.AI", "cs.LG" ], "topics": [ "agent-evaluation", "agent-safety", "coding-agent", "reasoning", "tool-use", "workflow-agent" ], "score": 16, "relevance": "high", "matched_queries": [ "ai-agent" ], "arxiv_id": "2606.19245", "source": "arxiv", "source_id": "arxiv:2606.19245", "pdf_url": "https://arxiv.org/pdf/2606.19245", "primary_query": "ai-agent" }, { "id": "2606.18619", "title": "Code-Augur: Agentic Vulnerability Detection via Specification Inference", "url": "https://arxiv.org/abs/2606.18619", "published": "2026-06-17", "updated": "2026-06-17", "authors": [ "Zhengxiong Luo", "Mehtab Zafar", "Dylan Wolff", "Abhik Roychoudhury" ], "categories": [ "cs.CR", "cs.AI", "cs.SE" ], "topics": [ "agent-safety", "computer-use", "rag", "reasoning" ], "score": 16, "relevance": "high", "matched_queries": [ "ai-agent" ], "arxiv_id": "2606.18619", "source": "arxiv", "source_id": "arxiv:2606.18619", "pdf_url": "https://arxiv.org/pdf/2606.18619", "primary_query": "ai-agent" }, { "id": "2606.19464", "title": "Deontic Policies for Runtime Governance of Agentic AI Systems", "url": "https://arxiv.org/abs/2606.19464", "published": "2026-06-17", "updated": "2026-06-17", "authors": [ "Anupam Joshi", "Tim Finin", "Karuna Pande Joshi", "Lalana Kagal" ], "categories": [ "cs.AI", "cs.MA" ], "topics": [ "agent-evaluation", "agent-safety", "reasoning", "tool-use", "workflow-agent" ], "score": 16, "relevance": "high", "matched_queries": [ "agentic-ai", "autonomous-agent-llm" ], "arxiv_id": "2606.19464", "source": "arxiv", "source_id": "arxiv:2606.19464", "pdf_url": "https://arxiv.org/pdf/2606.19464", "primary_query": "agentic-ai" }, { "id": "2606.18502", "title": "Towards Scalable Customization and Deployment of Multi-Agent Systems for Enterprise Applications", "url": "https://arxiv.org/abs/2606.18502", "published": "2026-06-16", "updated": "2026-06-16", "authors": [ "Paresh Dashore", "Shreyas Kulkarni", "Uttam Gurram", "Nadia Bathaee", "Kartik Balasubramaniam", "Genta Indra Winata", "Sambit Sahu", "Shi-Xiong Zhang" ], "categories": [ "cs.CL" ], "topics": [ "coding-agent", "multi-agent", "reasoning", "tool-use", "workflow-agent" ], "score": 16, "relevance": "high", "matched_queries": [ "agentic-ai" ], "arxiv_id": "2606.18502", "source": "arxiv", "source_id": "arxiv:2606.18502", "pdf_url": "https://arxiv.org/pdf/2606.18502", "primary_query": "agentic-ai" }, { "id": "2606.18037", "title": "ProvenanceGuard: Source-Aware Factuality Verification for MCP-Based LLM Agents", "url": "https://arxiv.org/abs/2606.18037", "published": "2026-06-16", "updated": "2026-06-16", "authors": [ "Ander Alvarez", "Santhiya Rajan", "Samuel Mugel", "Román Orús" ], "categories": [ "cs.AI", "cs.CL", "cs.MA" ], "topics": [ "agent-evaluation", "agent-safety", "rag", "reasoning", "tool-use" ], "score": 16, "relevance": "high", "matched_queries": [ "tool-use" ], "arxiv_id": "2606.18037", "source": "arxiv", "source_id": "arxiv:2606.18037", "pdf_url": "https://arxiv.org/pdf/2606.18037", "primary_query": "tool-use" }, { "id": "2606.17573", "title": "Cordon: Semantic Transactions for Tool-Using LLM Agents", "url": "https://arxiv.org/abs/2606.17573", "published": "2026-06-16", "updated": "2026-06-16", "authors": [ "Zheng Chen", "Hanqing Liu", "Duling Xu", "Dong Dong", "Jialin Li", "Bangzheng Pu", "Jidong Zhai" ], "categories": [ "cs.OS", "cs.CR" ], "topics": [ "agent-evaluation", "agent-safety", "memory", "tool-use", "workflow-agent" ], "score": 16, "relevance": "high", "matched_queries": [ "tool-use" ], "arxiv_id": "2606.17573", "source": "arxiv", "source_id": "arxiv:2606.17573", "pdf_url": "https://arxiv.org/pdf/2606.17573", "primary_query": "tool-use" }, { "id": "2606.18051", "title": "Compositional Skill Routing for LLM Agents: Decompose, Retrieve, and Compose", "url": "https://arxiv.org/abs/2606.18051", "published": "2026-06-16", "updated": "2026-06-16", "authors": [ "Xueping Gao" ], "categories": [ "cs.CL" ], "topics": [ "agent-evaluation", "planning", "rag", "tool-use" ], "score": 16, "relevance": "high", "matched_queries": [ "planning-agent" ], "arxiv_id": "2606.18051", "source": "arxiv", "source_id": "arxiv:2606.18051", "pdf_url": "https://arxiv.org/pdf/2606.18051", "primary_query": "planning-agent" }, { "id": "2606.16295", "title": "VisualClaw: A Real-Time, Personalized Agent for the Physical World", "url": "https://arxiv.org/abs/2606.16295", "published": "2026-06-15", "updated": "2026-06-15", "authors": [ "Haoqin Tu", "Jianwen Chen", "Zijun Wang", "Siwei Han", "Juncheng Wu", "Hardy Chen", "Haonian Ji", "Kaiwen Xiong", "Jiaqi Liu", "Peng Xia", "Jieru Mei", "Hongliang Fei", "Jason Eshraghian", "Zeyu Zheng", "Yuyin Zhou", "Huaxiu Yao", "Cihang Xie" ], "categories": [ "cs.CV", "cs.CL" ], "topics": [ "agent-evaluation", "coding-agent", "computer-use", "rag", "tool-use" ], "score": 16, "relevance": "high", "matched_queries": [ "agent-evaluation", "tool-use", "web-gui-agent" ], "arxiv_id": "2606.16295", "source": "arxiv", "source_id": "arxiv:2606.16295", "pdf_url": "https://arxiv.org/pdf/2606.16295", "primary_query": "agent-evaluation" }, { "id": "2606.16534", "title": "Generated, Parallel, Scalable? A Study of Agentic AI-Generated Julia Code on Supercomputers", "url": "https://arxiv.org/abs/2606.16534", "published": "2026-06-15", "updated": "2026-06-16", "authors": [ "Linus Bantel", "Anna-Lena Roth", "Jonas Posner", "Dirk Pflüger" ], "categories": [ "cs.DC" ], "topics": [ "agent-evaluation", "memory", "planning", "tool-use" ], "score": 16, "relevance": "high", "matched_queries": [ "tool-use" ], "arxiv_id": "2606.16534", "source": "arxiv", "source_id": "arxiv:2606.16534", "pdf_url": "https://arxiv.org/pdf/2606.16534", "primary_query": "tool-use" }, { "id": "2606.15376", "title": "CoAgent: Concurrency Control for Multi-Agent Systems", "url": "https://arxiv.org/abs/2606.15376", "published": "2026-06-13", "updated": "2026-06-13", "authors": [ "Hongtao Lyu", "Dingyan Zhang", "Mingyu Wu", "Xingda Wei", "Haibo Chen" ], "categories": [ "cs.DC", "cs.AI", "cs.MA" ], "topics": [ "coding-agent", "multi-agent", "planning", "tool-use" ], "score": 16, "relevance": "high", "matched_queries": [ "planning-agent" ], "arxiv_id": "2606.15376", "source": "arxiv", "source_id": "arxiv:2606.15376", "pdf_url": "https://arxiv.org/pdf/2606.15376", "primary_query": "planning-agent" }, { "id": "2606.14517", "title": "From Shield to Target: Denial-of-Service Attacks on LLM-Based Agent Guardrails", "url": "https://arxiv.org/abs/2606.14517", "published": "2026-06-12", "updated": "2026-06-16", "authors": [ "Yuguang Zhou", "Xunguang Wang", "Pingchuan Ma", "Zhantong Xue", "Zhaoyu Wang", "Shuai Wang" ], "categories": [ "cs.CR", "cs.AI" ], "topics": [ "agent-evaluation", "agent-safety", "computer-use", "multi-agent", "reasoning" ], "score": 16, "relevance": "high", "matched_queries": [ "agent-evaluation", "autonomous-agent-llm" ], "arxiv_id": "2606.14517", "source": "arxiv", "source_id": "arxiv:2606.14517", "pdf_url": "https://arxiv.org/pdf/2606.14517", "primary_query": "agent-evaluation" }, { "id": "2606.14470", "title": "GitOfThoughts: Version-Controlled Reasoning and Agent Memory You Can Replay, Diff, and Merge", "url": "https://arxiv.org/abs/2606.14470", "published": "2026-06-12", "updated": "2026-06-22", "authors": [ "Pavan C Shekar", "Abhishek H S", "Aswanth Krishnan" ], "categories": [ "cs.AI", "cs.CL", "cs.LG" ], "topics": [ "agent-evaluation", "coding-agent", "memory", "rag", "reasoning" ], "score": 16, "relevance": "high", "matched_queries": [ "agent-memory" ], "arxiv_id": "2606.14470", "source": "arxiv", "source_id": "arxiv:2606.14470", "pdf_url": "https://arxiv.org/pdf/2606.14470", "primary_query": "agent-memory" }, { "id": "2606.14106", "title": "Naive Visual Memory is Not Enough: A Failure-Mode Study of GUI Agents", "url": "https://arxiv.org/abs/2606.14106", "published": "2026-06-12", "updated": "2026-06-12", "authors": [ "Seoyoung Choi", "Minseok Ko", "Hyunseok Lee", "Kunwoong Kim", "Woomin Song", "Chanseok Jeon", "Jinwoo Shin" ], "categories": [ "cs.MA", "cs.CV" ], "topics": [ "computer-use", "memory", "rag", "reasoning", "tool-use" ], "score": 16, "relevance": "high", "matched_queries": [ "web-gui-agent" ], "arxiv_id": "2606.14106", "source": "arxiv", "source_id": "arxiv:2606.14106", "pdf_url": "https://arxiv.org/pdf/2606.14106", "primary_query": "web-gui-agent" }, { "id": "2606.13608", "title": "AgentBeats: Agentifying Agent Assessment for Openness, Standardization, and Reproducibility", "url": "https://arxiv.org/abs/2606.13608", "published": "2026-06-11", "updated": "2026-06-14", "authors": [ "Xiaoyuan Liu", "Jianhong Tu", "Yuqi Chen", "Siyuan Xie", "Sihan Ren", "Tianneng Shi", "Gal Gantar", "Evan Sandoval", "Donghyun Lee", "Daniel Miao", "Peter J. Gilbert", "Nick Hynes", "Mauro Staver", "Warren He", "David Marn", "Andrew Low", "Xi Zhang", "Elron Bandel", "Michal Shmueli-Scheuer", "Siva Reddy", "Alexandre Drouin", "Alexandre Lacoste", "Ramayya Krishnan", "Elham Tabassi", "Yu Su", "Victor Barres", "Chenguang Wang", "Wenbo Guo", "Dawn Song" ], "categories": [ "cs.AI", "cs.LG" ], "topics": [ "agent-evaluation", "coding-agent", "multi-agent", "rag", "tool-use" ], "score": 16, "relevance": "high", "matched_queries": [ "agent-evaluation" ], "arxiv_id": "2606.13608", "source": "arxiv", "source_id": "arxiv:2606.13608", "pdf_url": "https://arxiv.org/pdf/2606.13608", "primary_query": "agent-evaluation" }, { "id": "2606.13177", "title": "MemRefine: LLM-Guided Compression for Long-Term Agent Memory", "url": "https://arxiv.org/abs/2606.13177", "published": "2026-06-11", "updated": "2026-06-11", "authors": [ "Minjae Kim", "Jinheon Baek", "Soyeong Jeong", "Sung Ju Hwang" ], "categories": [ "cs.CL", "cs.AI", "cs.LG" ], "topics": [ "agent-evaluation", "computer-use", "memory", "rag", "tool-use" ], "score": 16, "relevance": "high", "matched_queries": [ "agent-memory" ], "arxiv_id": "2606.13177", "source": "arxiv", "source_id": "arxiv:2606.13177", "pdf_url": "https://arxiv.org/pdf/2606.13177", "primary_query": "agent-memory" }, { "id": "2606.13148", "title": "TerraBench: Can Agents Reason Over Heterogeneous Earth-System Data?", "url": "https://arxiv.org/abs/2606.13148", "published": "2026-06-11", "updated": "2026-07-01", "authors": [ "Dat Tien Nguyen", "Thao Nguyen", "Fadillah Adamsyah Maani", "Huy M. Le", "Muhammad Umer Sheikh", "Numan Saeed", "Muhammad Haris Khan", "Salman Khan" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "planning", "rag", "reasoning", "tool-use", "workflow-agent", "world-model" ], "score": 16, "relevance": "high", "matched_queries": [ "tool-use" ], "arxiv_id": "2606.13148", "source": "arxiv", "source_id": "arxiv:2606.13148", "pdf_url": "https://arxiv.org/pdf/2606.13148", "primary_query": "tool-use" }, { "id": "2606.28360", "title": "Carolina Guide: A Multi-Agent RAG System with Institutional Guardrails for Academic Policy Assistance", "url": "https://arxiv.org/abs/2606.28360", "published": "2026-06-11", "updated": "2026-06-11", "authors": [ "Ben Torsion", "Jun Zhou" ], "categories": [ "cs.IR", "cs.AI" ], "topics": [ "agent-evaluation", "agent-safety", "computer-use", "multi-agent", "rag" ], "score": 16, "relevance": "high", "matched_queries": [ "rag-agent" ], "arxiv_id": "2606.28360", "source": "arxiv", "source_id": "arxiv:2606.28360", "pdf_url": "https://arxiv.org/pdf/2606.28360", "primary_query": "rag-agent" }, { "id": "2606.12344", "title": "Claw-SWE-Bench: A Benchmark for Evaluating OpenClaw-style Agent Harnesses on Coding Tasks", "url": "https://arxiv.org/abs/2606.12344", "published": "2026-06-10", "updated": "2026-06-10", "authors": [ "Mengyu Zheng", "Kai Han", "Boxun Li", "Haiyang Xu", "Yuchuan Tian", "Wei He", "Hang Zhou", "Jianyuan Guo", "Hailin Hu", "Lin Ma", "Chao Xu", "Guohao Dai", "Lixue Xia", "Yunchao Wei", "Yunhe Wang", "Yu Wang" ], "categories": [ "cs.LG", "cs.CL" ], "topics": [ "agent-evaluation", "coding-agent", "tool-use" ], "score": 16, "relevance": "high", "matched_queries": [ "agent-evaluation" ], "arxiv_id": "2606.12344", "source": "arxiv", "source_id": "arxiv:2606.12344", "pdf_url": "https://arxiv.org/pdf/2606.12344", "primary_query": "agent-evaluation" }, { "id": "2606.11702", "title": "MedCTA: A Benchmark for Clinical Tool Agents", "url": "https://arxiv.org/abs/2606.11702", "published": "2026-06-10", "updated": "2026-06-10", "authors": [ "Tajamul Ashraf", "Hyewon Jeong", "Fida Mohammad Thoker", "Bernard Ghanem" ], "categories": [ "cs.CV", "cs.AI", "cs.CL" ], "topics": [ "agent-evaluation", "planning", "rag", "tool-use" ], "score": 16, "relevance": "high", "matched_queries": [ "tool-use" ], "arxiv_id": "2606.11702", "source": "arxiv", "source_id": "arxiv:2606.11702", "pdf_url": "https://arxiv.org/pdf/2606.11702", "primary_query": "tool-use" }, { "id": "2606.11119", "title": "TRACE: A Unified Rollout Budget Allocation Framework for Efficient Agentic Reinforcement Learning", "url": "https://arxiv.org/abs/2606.11119", "published": "2026-06-09", "updated": "2026-06-09", "authors": [ "Heming Zou", "Qi Wang", "Yun Qu", "Yuhang Jiang", "Lizhou Cai", "Yixiu Mao", "Ru Peng", "Xin Xu", "Weijie Liu", "Kai Yang", "Saiyong Yang", "Xiangyang Ji" ], "categories": [ "cs.LG", "cs.AI", "cs.CL" ], "topics": [ "agent-evaluation", "computer-use", "rag", "reasoning", "tool-use" ], "score": 16, "relevance": "high", "matched_queries": [ "agent-evaluation" ], "arxiv_id": "2606.11119", "source": "arxiv", "source_id": "arxiv:2606.11119", "pdf_url": "https://arxiv.org/pdf/2606.11119", "primary_query": "agent-evaluation" }, { "id": "2606.10921", "title": "Trace Only What You Need: Structure-Aware On-Demand Hypergraph Memory for Long-Document Question Answering", "url": "https://arxiv.org/abs/2606.10921", "published": "2026-06-09", "updated": "2026-06-09", "authors": [ "Xiangjun Zai", "Xingyu Tan", "Chen Chen", "Xiaoyang Wang", "Wenjie Zhang" ], "categories": [ "cs.CL" ], "topics": [ "agent-evaluation", "computer-use", "memory", "multi-agent", "planning", "rag", "reasoning" ], "score": 16, "relevance": "high", "matched_queries": [ "rag-agent" ], "arxiv_id": "2606.10921", "source": "arxiv", "source_id": "arxiv:2606.10921", "pdf_url": "https://arxiv.org/pdf/2606.10921", "primary_query": "rag-agent" }, { "id": "2606.10316", "title": "TabClaw: An Interactive and Self-Evolving Agent for Spreadsheet Manipulation and Table Reasoning", "url": "https://arxiv.org/abs/2606.10316", "published": "2026-06-09", "updated": "2026-06-09", "authors": [ "Mingyue Cheng", "Shuo Yu", "Daoyu Wang", "Qingchuan Li", "Xiaoyu Tao", "Qingyang Mao", "Yitong Zhou", "Qi Liu" ], "categories": [ "cs.CL" ], "topics": [ "agent-evaluation", "memory", "planning", "reasoning", "tool-use", "workflow-agent" ], "score": 16, "relevance": "high", "matched_queries": [ "planning-agent" ], "arxiv_id": "2606.10316", "source": "arxiv", "source_id": "arxiv:2606.10316", "pdf_url": "https://arxiv.org/pdf/2606.10316", "primary_query": "planning-agent" }, { "id": "2606.09037", "title": "A Multi-Agent System for IPMSM Design Optimization via an FEA-AI Hybrid Approach", "url": "https://arxiv.org/abs/2606.09037", "published": "2026-06-08", "updated": "2026-06-08", "authors": [ "Jinseong Han", "Sunwoong Yang", "Namwoo Kang" ], "categories": [ "cs.AI", "cs.MA" ], "topics": [ "agent-evaluation", "multi-agent", "planning", "rag", "reasoning", "workflow-agent" ], "score": 16, "relevance": "high", "matched_queries": [ "rag-agent" ], "arxiv_id": "2606.09037", "source": "arxiv", "source_id": "arxiv:2606.09037", "pdf_url": "https://arxiv.org/pdf/2606.09037", "primary_query": "rag-agent" }, { "id": "2606.09738", "title": "HDSL: A Hierarchical Domain-Specific Language for Structured 3D Indoor Scene Generation and Localized Editing with LLM Agents", "url": "https://arxiv.org/abs/2606.09738", "published": "2026-06-08", "updated": "2026-06-08", "authors": [ "Letian Li", "Chao Shen", "Shuzhao Xie", "Chenghao Gu", "ZhengXiao He", "Yu Meng", "Xin Yang", "Wenyuan Jiang", "Zhi Wang" ], "categories": [ "cs.CV" ], "topics": [ "agent-evaluation", "agent-safety", "coding-agent", "planning", "rag" ], "score": 16, "relevance": "high", "matched_queries": [ "planning-agent" ], "arxiv_id": "2606.09738", "source": "arxiv", "source_id": "arxiv:2606.09738", "pdf_url": "https://arxiv.org/pdf/2606.09738", "primary_query": "planning-agent" }, { "id": "2606.10209", "title": "Less Context, Better Agents: Efficient Context Engineering for Long-Horizon Tool-Using LLM Agents", "url": "https://arxiv.org/abs/2606.10209", "published": "2026-06-08", "updated": "2026-06-08", "authors": [ "Abhilasha Lodha", "Mahsa Pahlavikhah Varnosfaderani", "Abir Chakraborty", "Abhinav Mithal" ], "categories": [ "cs.AI", "cs.LG", "cs.SE" ], "topics": [ "agent-evaluation", "planning", "rag", "tool-use", "workflow-agent" ], "score": 16, "relevance": "high", "matched_queries": [ "autonomous-agent-llm" ], "arxiv_id": "2606.10209", "source": "arxiv", "source_id": "arxiv:2606.10209", "pdf_url": "https://arxiv.org/pdf/2606.10209", "primary_query": "autonomous-agent-llm" }, { "id": "2606.05711", "title": "Beyond tokens: a unified framework for latent communication in LLM-based multi-agent systems", "url": "https://arxiv.org/abs/2606.05711", "published": "2026-06-04", "updated": "2026-06-05", "authors": [ "Yingzhuo Liu" ], "categories": [ "cs.CL" ], "topics": [ "agent-safety", "computer-use", "multi-agent", "planning", "reasoning", "tool-use" ], "score": 16, "relevance": "high", "matched_queries": [ "language-agent" ], "arxiv_id": "2606.05711", "source": "arxiv", "source_id": "arxiv:2606.05711", "pdf_url": "https://arxiv.org/pdf/2606.05711", "primary_query": "language-agent" }, { "id": "2606.06090", "title": "Beyond Semantic Organization: Memory as Execution State Management for Long-Horizon Agents", "url": "https://arxiv.org/abs/2606.06090", "published": "2026-06-04", "updated": "2026-06-04", "authors": [ "Yaoqi Chen", "Haibin Lai", "Yuru Feng", "Chuyu Han", "Qianxi Zhang", "Baotong Lu", "Menghao Li", "Xinjiang Wang", "Zhirui Wang", "Shusen Xu", "Zengzhong Li", "Zewen Jin", "Hao Wu", "Cheng Li", "Qi Chen" ], "categories": [ "cs.AI" ], "topics": [ "computer-use", "memory", "planning", "rag", "tool-use" ], "score": 16, "relevance": "high", "matched_queries": [ "agent-memory", "rag-agent" ], "arxiv_id": "2606.06090", "source": "arxiv", "source_id": "arxiv:2606.06090", "pdf_url": "https://arxiv.org/pdf/2606.06090", "primary_query": "agent-memory" }, { "id": "2606.06388", "title": "Humans' ALMANAC: A Human Collaboration Dataset of Action-Level Mental Model Annotations for Agent Collaboration", "url": "https://arxiv.org/abs/2606.06388", "published": "2026-06-04", "updated": "2026-06-06", "authors": [ "Jiaju Chen", "Yuxuan Lu", "Jiayi Su", "Chaoran Chen", "Songlin Xiao", "Zheng Zhang", "Yun Wang", "Yunyao Li", "Jian Zhao", "Tongshuang Wu", "Toby Jia-Jun Li", "Dakuo Wang", "Bingsheng Yao" ], "categories": [ "cs.AI", "cs.CL" ], "topics": [ "agent-evaluation", "computer-use", "multi-agent", "planning", "reasoning", "tool-use" ], "score": 16, "relevance": "high", "matched_queries": [ "planning-agent" ], "arxiv_id": "2606.06388", "source": "arxiv", "source_id": "arxiv:2606.06388", "pdf_url": "https://arxiv.org/pdf/2606.06388", "primary_query": "planning-agent" }, { "id": "2606.05241", "title": "Search-Time Contamination in Deep Research Agents: Measuring Performance Inflation in Public Benchmark Evaluation", "url": "https://arxiv.org/abs/2606.05241", "published": "2026-06-03", "updated": "2026-06-03", "authors": [ "Yongjie Wang", "Xinyue Zhang", "Kunhong Yao", "Zhiwei Zeng", "Kaisong Song", "Jun Lin", "Zhiqi Shen" ], "categories": [ "cs.CR", "cs.AI" ], "topics": [ "agent-evaluation", "rag", "reasoning" ], "score": 16, "relevance": "high", "matched_queries": [ "agent-evaluation" ], "arxiv_id": "2606.05241", "source": "arxiv", "source_id": "arxiv:2606.05241", "pdf_url": "https://arxiv.org/pdf/2606.05241", "primary_query": "agent-evaluation" }, { "id": "2606.03197", "title": "MemTrain: Self-Supervised Context Memory Training", "url": "https://arxiv.org/abs/2606.03197", "published": "2026-06-02", "updated": "2026-06-02", "authors": [ "Ziheng Li", "Xingrun Xing", "Haoqing Wang", "Zhi-Hong Deng", "Yehui Tang" ], "categories": [ "cs.CL" ], "topics": [ "agent-evaluation", "memory", "planning", "rag", "reasoning", "tool-use" ], "score": 16, "relevance": "high", "matched_queries": [ "agent-memory" ], "arxiv_id": "2606.03197", "source": "arxiv", "source_id": "arxiv:2606.03197", "pdf_url": "https://arxiv.org/pdf/2606.03197", "primary_query": "agent-memory" }, { "id": "2606.02497", "title": "Bridging the Last Mile of Time Series Forecasting with LLM Agents", "url": "https://arxiv.org/abs/2606.02497", "published": "2026-06-01", "updated": "2026-06-01", "authors": [ "Yuhua Liao", "Zetian Wang", "Qiangqiang Nie", "Zhenhua Zhang" ], "categories": [ "cs.AI" ], "topics": [ "agent-safety", "memory", "planning", "reasoning", "tool-use" ], "score": 16, "relevance": "high", "matched_queries": [ "planning-agent" ], "arxiv_id": "2606.02497", "source": "arxiv", "source_id": "arxiv:2606.02497", "pdf_url": "https://arxiv.org/pdf/2606.02497", "primary_query": "planning-agent" }, { "id": "2606.01222", "title": "RAG-driven Multi-Agent LLM Framework with Task Decomposition for Beyond 5G Auto-Configuration", "url": "https://arxiv.org/abs/2606.01222", "published": "2026-05-31", "updated": "2026-05-31", "authors": [ "İrşat Emin Sarıdaş", "Onur Salan", "Ali Görçin", "Ibrahim Hokelek", "Hakan Ali Çırpan" ], "categories": [ "eess.SP" ], "topics": [ "agent-evaluation", "multi-agent", "planning", "rag", "reasoning" ], "score": 16, "relevance": "high", "matched_queries": [ "rag-agent" ], "arxiv_id": "2606.01222", "source": "arxiv", "source_id": "arxiv:2606.01222", "pdf_url": "https://arxiv.org/pdf/2606.01222", "primary_query": "rag-agent" }, { "id": "2606.00619", "title": "MemPro: Agentic Memory Systems as Evolvable Programs", "url": "https://arxiv.org/abs/2606.00619", "published": "2026-05-30", "updated": "2026-05-30", "authors": [ "Qingshan Liu", "Guoqing Wang", "Wen Wu", "Jingqi Huang", "Xinqi Tao", "Dejia Song", "Jie Zhou", "Liang He" ], "categories": [ "cs.CL", "cs.AI" ], "topics": [ "agent-evaluation", "computer-use", "memory", "planning", "rag" ], "score": 16, "relevance": "high", "matched_queries": [ "agent-memory" ], "arxiv_id": "2606.00619", "source": "arxiv", "source_id": "arxiv:2606.00619", "pdf_url": "https://arxiv.org/pdf/2606.00619", "primary_query": "agent-memory" }, { "id": "2606.00922", "title": "A Machine-to-Machine Knowledge-Guided LLM Agent for Generalizable Radiotherapy Treatment Planning", "url": "https://arxiv.org/abs/2606.00922", "published": "2026-05-30", "updated": "2026-05-30", "authors": [ "Md Mainul Abrar", "Xun Jia", "Yujie Chi" ], "categories": [ "physics.med-ph", "cs.RO" ], "topics": [ "agent-evaluation", "agent-safety", "computer-use", "planning", "reasoning" ], "score": 16, "relevance": "high", "matched_queries": [ "planning-agent" ], "arxiv_id": "2606.00922", "source": "arxiv", "source_id": "arxiv:2606.00922", "pdf_url": "https://arxiv.org/pdf/2606.00922", "primary_query": "planning-agent" }, { "id": "2606.07595", "title": "VisualLeakBench: Reproducible Action-Boundary Propagation Failures in Vision-Language Agents", "url": "https://arxiv.org/abs/2606.07595", "published": "2026-05-29", "updated": "2026-05-29", "authors": [ "Youting Wang", "Yuan Tang", "Yitian Qian", "Chen Zhao" ], "categories": [ "cs.CV", "cs.AI", "cs.IR" ], "topics": [ "agent-evaluation", "agent-safety", "memory", "tool-use", "workflow-agent" ], "score": 16, "relevance": "high", "matched_queries": [ "language-agent" ], "arxiv_id": "2606.07595", "source": "arxiv", "source_id": "arxiv:2606.07595", "pdf_url": "https://arxiv.org/pdf/2606.07595", "primary_query": "language-agent" }, { "id": "2606.00341", "title": "ROGUE: Misaligned Agent Behavior Arising from Ordinary Computer Use", "url": "https://arxiv.org/abs/2606.00341", "published": "2026-05-29", "updated": "2026-05-29", "authors": [ "Jeremy Tien", "Abishek Anand", "Yu-Rou Tuan", "Yuchen Shen", "J. Zico Kolter", "Aran Nayebi" ], "categories": [ "cs.LG", "cs.AI" ], "topics": [ "agent-evaluation", "agent-safety", "computer-use", "tool-use", "workflow-agent" ], "score": 16, "relevance": "high", "matched_queries": [ "agent-safety" ], "arxiv_id": "2606.00341", "source": "arxiv", "source_id": "arxiv:2606.00341", "pdf_url": "https://arxiv.org/pdf/2606.00341", "primary_query": "agent-safety" }, { "id": "2605.31377", "title": "DynaTree: Dynamic Agentic Retrieval Tree for Time-Sensitive News Retrieval", "url": "https://arxiv.org/abs/2605.31377", "published": "2026-05-29", "updated": "2026-05-29", "authors": [ "Siyuan Qi", "Xinyuan Wang", "Yingxuan Yang", "Haochuan Guo", "Jianghao Lin", "Weiwen Liu", "Yong Yu", "Weinan Zhang" ], "categories": [ "cs.IR", "cs.AI" ], "topics": [ "agent-evaluation", "planning", "rag", "reasoning", "tool-use" ], "score": 16, "relevance": "high", "matched_queries": [ "rag-agent" ], "arxiv_id": "2605.31377", "source": "arxiv", "source_id": "arxiv:2605.31377", "pdf_url": "https://arxiv.org/pdf/2605.31377", "primary_query": "rag-agent" }, { "id": "2605.30947", "title": "Extending AI for Research to the Humanities: A Multi-Agent Framework for Evidence-Grounded Scholarship", "url": "https://arxiv.org/abs/2605.30947", "published": "2026-05-29", "updated": "2026-06-03", "authors": [ "Yating Pan", "Jiajun Zhang", "Jun Wang", "Qi Su" ], "categories": [ "cs.CL" ], "topics": [ "agent-evaluation", "multi-agent", "rag", "reasoning", "tool-use" ], "score": 16, "relevance": "high", "matched_queries": [ "rag-agent" ], "arxiv_id": "2605.30947", "source": "arxiv", "source_id": "arxiv:2605.30947", "pdf_url": "https://arxiv.org/pdf/2605.30947", "primary_query": "rag-agent" }, { "id": "2605.29960", "title": "Hijacking Agent Memory: Stealthy Trojan Attacks Through Conversational Interaction", "url": "https://arxiv.org/abs/2605.29960", "published": "2026-05-28", "updated": "2026-05-28", "authors": [ "Hongtao Wang", "Se Yang", "Yu Chen", "Puzhuo Liu" ], "categories": [ "cs.CR", "cs.AI" ], "topics": [ "agent-evaluation", "memory", "rag", "tool-use" ], "score": 16, "relevance": "high", "matched_queries": [ "agent-memory" ], "arxiv_id": "2605.29960", "source": "arxiv", "source_id": "arxiv:2605.29960", "pdf_url": "https://arxiv.org/pdf/2605.29960", "primary_query": "agent-memory" }, { "id": "2605.25435", "title": "Security of OpenClaw Agents: Fundamentals, Attacks, and Countermeasures", "url": "https://arxiv.org/abs/2605.25435", "published": "2026-05-25", "updated": "2026-05-25", "authors": [ "Yuntao Wang", "Jianle Ba", "Han Liu", "Yanghe Pan", "Jintao Wei", "Zhou Su", "Tom H. Luan", "Linkang Du" ], "categories": [ "cs.AI" ], "topics": [ "agent-safety", "computer-use", "memory", "multi-agent", "reasoning", "tool-use" ], "score": 16, "relevance": "high", "matched_queries": [ "autonomous-agent-llm" ], "arxiv_id": "2605.25435", "source": "arxiv", "source_id": "arxiv:2605.25435", "pdf_url": "https://arxiv.org/pdf/2605.25435", "primary_query": "autonomous-agent-llm" }, { "id": "2605.24812", "title": "CoRe-Code: Collaborative Reinforcement Learning for Code Generation", "url": "https://arxiv.org/abs/2605.24812", "published": "2026-05-24", "updated": "2026-05-24", "authors": [ "Zhihao Dou", "Qinjian Zhao", "Zhongwei Wan", "Xiaoyu Xia", "Sumon Biswas" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "agent-safety", "coding-agent", "memory", "multi-agent", "planning", "rag" ], "score": 16, "relevance": "high", "matched_queries": [ "planning-agent" ], "arxiv_id": "2605.24812", "source": "arxiv", "source_id": "arxiv:2605.24812", "pdf_url": "https://arxiv.org/pdf/2605.24812", "primary_query": "planning-agent" }, { "id": "2605.23067", "title": "What Training Data Teaches RL Memory Agents: An Empirical Study of Curriculum Effects in Memory-Augmented QA", "url": "https://arxiv.org/abs/2605.23067", "published": "2026-05-21", "updated": "2026-05-21", "authors": [ "Xinjie He", "Zhiyuan Lin", "Su Liu", "Jialun Wu", "Qiyang Xie", "Weikai Zhou", "Shuai Xiao" ], "categories": [ "cs.CL" ], "topics": [ "agent-evaluation", "memory", "reasoning" ], "score": 16, "relevance": "high", "matched_queries": [ "agent-memory" ], "arxiv_id": "2605.23067", "source": "arxiv", "source_id": "arxiv:2605.23067", "pdf_url": "https://arxiv.org/pdf/2605.23067", "primary_query": "agent-memory" }, { "id": "2605.20616", "title": "Auto-Dreamer: Learning Offline Memory Consolidation for Language Agents", "url": "https://arxiv.org/abs/2605.20616", "published": "2026-05-20", "updated": "2026-05-20", "authors": [ "Chongrui Ye", "Yuxiang Liu", "Yu Wang", "Haofei Yu", "Yining Zhao", "Ge Liu", "Julian McAuley", "Jiaxuan You" ], "categories": [ "cs.CL" ], "topics": [ "agent-evaluation", "memory", "rag", "tool-use" ], "score": 16, "relevance": "high", "matched_queries": [ "agent-memory", "language-agent" ], "arxiv_id": "2605.20616", "source": "arxiv", "source_id": "arxiv:2605.20616", "pdf_url": "https://arxiv.org/pdf/2605.20616", "primary_query": "agent-memory" }, { "id": "2605.16233", "title": "FORGE: Self-Evolving Agent Memory With No Weight Updates via Population Broadcast", "url": "https://arxiv.org/abs/2605.16233", "published": "2026-05-15", "updated": "2026-05-15", "authors": [ "Igor Bogdanov", "Chung-Horng Lung", "Thomas Kunz", "Jie Gao", "Adrian Taylor", "Marzia Zaman" ], "categories": [ "cs.AI", "cs.CL", "cs.LG", "cs.MA", "eess.SY" ], "topics": [ "agent-evaluation", "memory", "rag", "reasoning" ], "score": 16, "relevance": "high", "matched_queries": [ "agent-memory" ], "arxiv_id": "2605.16233", "source": "arxiv", "source_id": "arxiv:2605.16233", "pdf_url": "https://arxiv.org/pdf/2605.16233", "primary_query": "agent-memory" }, { "id": "2605.15759", "title": "DimMem: Dimensional Structuring for Efficient Long-Term Agent Memory", "url": "https://arxiv.org/abs/2605.15759", "published": "2026-05-15", "updated": "2026-05-24", "authors": [ "Wentao Qiu", "Haotian Hu", "Fanyi Wang", "Jinwei Kong", "Yu Zhang" ], "categories": [ "cs.CL" ], "topics": [ "agent-evaluation", "memory", "rag", "tool-use" ], "score": 16, "relevance": "high", "matched_queries": [ "agent-memory" ], "arxiv_id": "2605.15759", "source": "arxiv", "source_id": "arxiv:2605.15759", "pdf_url": "https://arxiv.org/pdf/2605.15759", "primary_query": "agent-memory" }, { "id": "2605.15710", "title": "SMMBench: A Benchmark for Source-Distributed Multimodal Agent Memory", "url": "https://arxiv.org/abs/2605.15710", "published": "2026-05-15", "updated": "2026-05-15", "authors": [ "Huacan Chai", "Yukai Wang", "Yingxuan Yang", "Dan Peng", "Yuanyi Song", "Zhihui Fu", "Weiwen Liu", "Jianghao Lin", "Jun Wang", "Weinan Zhang" ], "categories": [ "cs.CL" ], "topics": [ "agent-evaluation", "memory", "rag", "reasoning", "tool-use" ], "score": 16, "relevance": "high", "matched_queries": [ "agent-memory" ], "arxiv_id": "2605.15710", "source": "arxiv", "source_id": "arxiv:2605.15710", "pdf_url": "https://arxiv.org/pdf/2605.15710", "primary_query": "agent-memory" }, { "id": "2605.15128", "title": "MemEye: A Visual-Centric Evaluation Framework for Multimodal Agent Memory", "url": "https://arxiv.org/abs/2605.15128", "published": "2026-05-14", "updated": "2026-05-14", "authors": [ "Minghao Guo", "Qingyue Jiao", "Zeru Shi", "Yihao Quan", "Boxuan Zhang", "Danrui Li", "Liwei Che", "Wujiang Xu", "Shilong Liu", "Zirui Liu", "Mubbasir Kapadia", "Vladimir Pavlovic", "Jiang Liu", "Mengdi Wang", "Yiyu Shi", "Dimitris N. Metaxas", "Ruixiang Tang" ], "categories": [ "cs.CV", "cs.CL", "cs.IR" ], "topics": [ "agent-evaluation", "memory", "reasoning", "tool-use" ], "score": 16, "relevance": "high", "matched_queries": [ "agent-memory" ], "arxiv_id": "2605.15128", "source": "arxiv", "source_id": "arxiv:2605.15128", "pdf_url": "https://arxiv.org/pdf/2605.15128", "primary_query": "agent-memory" }, { "id": "2605.14906", "title": "MemLens: Benchmarking Multimodal Long-Term Memory in Large Vision-Language Models", "url": "https://arxiv.org/abs/2605.14906", "published": "2026-05-14", "updated": "2026-05-14", "authors": [ "Xiyu Ren", "Zhaowei Wang", "Yiming Du", "Zhongwei Xie", "Chi Liu", "Xinlin Yang", "Haoyue Feng", "Wenjun Pan", "Tianshi Zheng", "Baixuan Xu", "Zhengnan Li", "Yangqiu Song", "Ginny Wong", "Simon See" ], "categories": [ "cs.CV" ], "topics": [ "agent-evaluation", "memory", "rag", "reasoning", "tool-use" ], "score": 16, "relevance": "high", "matched_queries": [ "agent-memory" ], "arxiv_id": "2605.14906", "source": "arxiv", "source_id": "arxiv:2605.14906", "pdf_url": "https://arxiv.org/pdf/2605.14906", "primary_query": "agent-memory" }, { "id": "2605.14932", "title": "Toward Securing AI Agents Like Operating Systems", "url": "https://arxiv.org/abs/2605.14932", "published": "2026-05-14", "updated": "2026-05-14", "authors": [ "Lukas Pirch", "Micha Horlboge", "Patrick Großmann", "Syeda Mahnur Asif", "Klim Kireev", "Thorsten Holz", "Konrad Rieck" ], "categories": [ "cs.CR" ], "topics": [ "agent-evaluation", "agent-safety", "computer-use", "tool-use" ], "score": 16, "relevance": "high", "matched_queries": [ "autonomous-agent-llm" ], "arxiv_id": "2605.14932", "source": "arxiv", "source_id": "arxiv:2605.14932", "pdf_url": "https://arxiv.org/pdf/2605.14932", "primary_query": "autonomous-agent-llm" }, { "id": "2605.11946", "title": "Counterfactual Trace Auditing of LLM Agent Skills", "url": "https://arxiv.org/abs/2605.11946", "published": "2026-05-12", "updated": "2026-05-28", "authors": [ "Xiaolin Zhou", "Jinbo Liu", "Li Li", "Ryan A. Rossi", "Xiyang Hu" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "coding-agent", "planning", "rag" ], "score": 16, "relevance": "high", "matched_queries": [ "planning-agent" ], "arxiv_id": "2605.11946", "source": "arxiv", "source_id": "arxiv:2605.11946", "pdf_url": "https://arxiv.org/pdf/2605.11946", "primary_query": "planning-agent" }, { "id": "2605.08964", "title": "Trustworthy AI: Ensuring Reliability and Accountability from Models to Agents", "url": "https://arxiv.org/abs/2605.08964", "published": "2026-05-09", "updated": "2026-05-09", "authors": [ "Carol Xuan Long" ], "categories": [ "cs.LG" ], "topics": [ "agent-evaluation", "agent-safety", "coding-agent", "multi-agent", "rag", "tool-use" ], "score": 16, "relevance": "high", "matched_queries": [ "autonomous-agent-llm" ], "arxiv_id": "2605.08964", "source": "arxiv", "source_id": "arxiv:2605.08964", "pdf_url": "https://arxiv.org/pdf/2605.08964", "primary_query": "autonomous-agent-llm" }, { "id": "2605.07830", "title": "CyBiasBench: Benchmarking Bias in LLM Agents for Cyber-Attack Scenarios", "url": "https://arxiv.org/abs/2605.07830", "published": "2026-05-08", "updated": "2026-05-08", "authors": [ "Taein Lim", "Seongyong Ju", "Munhyeok Kim", "Hyunjun Kim", "Hoki Kim" ], "categories": [ "cs.CR", "cs.AI" ], "topics": [ "agent-evaluation", "agent-safety" ], "score": 16, "relevance": "high", "matched_queries": [ "autonomous-agent-llm" ], "arxiv_id": "2605.07830", "source": "arxiv", "source_id": "arxiv:2605.07830", "pdf_url": "https://arxiv.org/pdf/2605.07830", "primary_query": "autonomous-agent-llm" }, { "id": "2605.04808", "title": "DecodingTrust-Agent Platform (DTap): A Controllable and Interactive Red-Teaming Platform for AI Agents", "url": "https://arxiv.org/abs/2605.04808", "published": "2026-05-06", "updated": "2026-05-06", "authors": [ "Zhaorun Chen", "Xun Liu", "Haibo Tong", "Chengquan Guo", "Yuzhou Nie", "Jiawei Zhang", "Mintong Kang", "Chejian Xu", "Qichang Liu", "Xiaogeng Liu", "Tianneng Shi", "Chaowei Xiao", "Sanmi Koyejo", "Percy Liang", "Wenbo Guo", "Dawn Song", "Bo Li" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "agent-safety", "coding-agent", "planning", "tool-use", "workflow-agent", "world-model" ], "score": 16, "relevance": "high", "matched_queries": [ "agent-safety" ], "arxiv_id": "2605.04808", "source": "arxiv", "source_id": "arxiv:2605.04808", "pdf_url": "https://arxiv.org/pdf/2605.04808", "primary_query": "agent-safety" }, { "id": "2605.01644", "title": "Toward a Principled Framework for Agent Safety Measurement", "url": "https://arxiv.org/abs/2605.01644", "published": "2026-05-02", "updated": "2026-05-02", "authors": [ "Shuyi Lin", "Anshuman Suri", "Alina Oprea", "Cheng Tan" ], "categories": [ "cs.CR" ], "topics": [ "agent-evaluation", "agent-safety", "coding-agent", "tool-use" ], "score": 16, "relevance": "high", "matched_queries": [ "agent-safety" ], "arxiv_id": "2605.01644", "source": "arxiv", "source_id": "arxiv:2605.01644", "pdf_url": "https://arxiv.org/pdf/2605.01644", "primary_query": "agent-safety" }, { "id": "2605.00741", "title": "Self-Adaptive Multi-Agent LLM-Based Security Pattern Selection for IoT Systems", "url": "https://arxiv.org/abs/2605.00741", "published": "2026-05-01", "updated": "2026-05-01", "authors": [ "Saeid Jamshidi", "Foutse Khomh", "Carol Fung", "Kawser Wazed Nafi" ], "categories": [ "cs.CR" ], "topics": [ "agent-evaluation", "agent-safety", "multi-agent", "reasoning", "tool-use" ], "score": 16, "relevance": "high", "matched_queries": [ "agent-safety" ], "arxiv_id": "2605.00741", "source": "arxiv", "source_id": "arxiv:2605.00741", "pdf_url": "https://arxiv.org/pdf/2605.00741", "primary_query": "agent-safety" }, { "id": "2604.24212", "title": "Empowering Autonomous Debugging Agents with Efficient Dynamic Analysis", "url": "https://arxiv.org/abs/2604.24212", "published": "2026-04-27", "updated": "2026-04-27", "authors": [ "Jiahong Xiang", "Xiaoyang Xu", "Xiaopan Chu", "Hongliang Tian", "Yuqun Zhang" ], "categories": [ "cs.SE" ], "topics": [ "agent-evaluation", "coding-agent", "embodied-agent", "memory", "rag", "tool-use" ], "score": 16, "relevance": "high", "matched_queries": [ "autonomous-agent-llm" ], "arxiv_id": "2604.24212", "source": "arxiv", "source_id": "arxiv:2604.24212", "pdf_url": "https://arxiv.org/pdf/2604.24212", "primary_query": "autonomous-agent-llm" }, { "id": "2604.23210", "title": "Discovering Agentic Safety Specifications from 1-Bit Danger Signals", "url": "https://arxiv.org/abs/2604.23210", "published": "2026-04-25", "updated": "2026-04-25", "authors": [ "Víctor Gallego" ], "categories": [ "cs.AI", "cs.CL" ], "topics": [ "agent-evaluation", "agent-safety", "planning", "rag", "reasoning", "tool-use" ], "score": 16, "relevance": "high", "matched_queries": [ "agent-safety" ], "arxiv_id": "2604.23210", "source": "arxiv", "source_id": "arxiv:2604.23210", "pdf_url": "https://arxiv.org/pdf/2604.23210", "primary_query": "agent-safety" }, { "id": "2604.13954", "title": "HINTBench: Horizon-agent Intrinsic Non-attack Trajectory Benchmark", "url": "https://arxiv.org/abs/2604.13954", "published": "2026-04-15", "updated": "2026-04-15", "authors": [ "Jiacheng Wang", "Jinchang Hou", "Fabian Wang", "Ping Jian", "Chenfu Bao", "Zhonghou Lv" ], "categories": [ "cs.LG", "cs.AI" ], "topics": [ "agent-evaluation", "agent-safety", "planning", "rag" ], "score": 16, "relevance": "high", "matched_queries": [ "agent-safety" ], "arxiv_id": "2604.13954", "source": "arxiv", "source_id": "arxiv:2604.13954", "pdf_url": "https://arxiv.org/pdf/2604.13954", "primary_query": "agent-safety" }, { "id": "2604.13536", "title": "Don't Let AI Agents YOLO Your Files: Shifting Information and Control to Filesystems for Agent Safety and Autonomy", "url": "https://arxiv.org/abs/2604.13536", "published": "2026-04-15", "updated": "2026-04-16", "authors": [ "Shawn Wanxiang Zhong", "Junxuan Liao", "Jing Liu", "Mai Zheng", "Andrea C. Arpaci-Dusseau", "Remzi H. Arpaci-Dusseau" ], "categories": [ "cs.OS" ], "topics": [ "agent-evaluation", "agent-safety", "coding-agent", "tool-use" ], "score": 16, "relevance": "high", "matched_queries": [ "agent-safety" ], "arxiv_id": "2604.13536", "source": "arxiv", "source_id": "arxiv:2604.13536", "pdf_url": "https://arxiv.org/pdf/2604.13536", "primary_query": "agent-safety" }, { "id": "2604.13298", "title": "Can Agents Secure Hardware? Evaluating Agentic LLM-Driven Obfuscation for IP Protection", "url": "https://arxiv.org/abs/2604.13298", "published": "2026-04-14", "updated": "2026-04-14", "authors": [ "Sujan Ghimire", "Parsa Mirfasihi", "Muhtasim Alam Chowdhury", "Veeramani Pugazhenthi", "Harish Kumar Dharavath", "Farshad Firouzi", "Rozhin Yasaei", "Pratik Satam", "Soheil Salehi" ], "categories": [ "cs.CR" ], "topics": [ "agent-evaluation", "agent-safety", "planning", "rag" ], "score": 16, "relevance": "high", "matched_queries": [ "agent-safety" ], "arxiv_id": "2604.13298", "source": "arxiv", "source_id": "arxiv:2604.13298", "pdf_url": "https://arxiv.org/pdf/2604.13298", "primary_query": "agent-safety" }, { "id": "2605.28835", "title": "GenesisFunc: Multi-Agent Data Generation for Accurate and Generalizable Function-Calling", "url": "https://arxiv.org/abs/2605.28835", "published": "2026-04-10", "updated": "2026-04-10", "authors": [ "Hao-Xiang Xu", "Chong Deng", "Jiaqing Liu", "Wen Wang", "Qian Chen", "Lujia Bao", "Xiangang Li", "Zhen-Hua Ling" ], "categories": [ "cs.CL", "cs.AI" ], "topics": [ "agent-evaluation", "multi-agent", "rag", "tool-use" ], "score": 16, "relevance": "high", "matched_queries": [ "function-calling" ], "arxiv_id": "2605.28835", "source": "arxiv", "source_id": "arxiv:2605.28835", "pdf_url": "https://arxiv.org/pdf/2605.28835", "primary_query": "function-calling" }, { "id": "2605.00845", "title": "Graph Query Generation with Constraint-guided Large Language Agents", "url": "https://arxiv.org/abs/2605.00845", "published": "2026-04-09", "updated": "2026-04-09", "authors": [ "Mengying Wang", "Nicolaas Jedema", "Rahul Pandey", "RaviKiran Krishnan", "Jens Lehmann", "Yinghui Wu" ], "categories": [ "cs.DB", "cs.AI", "cs.CL" ], "topics": [ "agent-evaluation", "computer-use", "reasoning", "workflow-agent" ], "score": 16, "relevance": "high", "matched_queries": [ "language-agent" ], "arxiv_id": "2605.00845", "source": "arxiv", "source_id": "arxiv:2605.00845", "pdf_url": "https://arxiv.org/pdf/2605.00845", "primary_query": "language-agent" }, { "id": "2604.26959", "title": "CareGuardAI: Context-Aware Multi-Agent Guardrails for Clinical Safety & Hallucination Mitigation in Patient-Facing LLMs", "url": "https://arxiv.org/abs/2604.26959", "published": "2026-04-07", "updated": "2026-04-07", "authors": [ "Elham Nasarian", "Abhilash Neog", "Kwok-Leung Tsui", "Niyousha HosseiniChimeh" ], "categories": [ "cs.CY", "cs.AI", "cs.MA" ], "topics": [ "agent-evaluation", "agent-safety", "multi-agent", "tool-use" ], "score": 16, "relevance": "high", "matched_queries": [ "agent-safety" ], "arxiv_id": "2604.26959", "source": "arxiv", "source_id": "arxiv:2604.26959", "pdf_url": "https://arxiv.org/pdf/2604.26959", "primary_query": "agent-safety" }, { "id": "2604.04131", "title": "Profile-Then-Reason: Bounded Semantic Complexity for Tool-Augmented Language Agents", "url": "https://arxiv.org/abs/2604.04131", "published": "2026-04-05", "updated": "2026-04-05", "authors": [ "Paulo Akira F. Enabe" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "rag", "reasoning", "tool-use", "workflow-agent" ], "score": 16, "relevance": "high", "matched_queries": [ "language-agent" ], "arxiv_id": "2604.04131", "source": "arxiv", "source_id": "arxiv:2604.04131", "pdf_url": "https://arxiv.org/pdf/2604.04131", "primary_query": "language-agent" }, { "id": "2603.28428", "title": "Synergy: A Next-Generation General-Purpose Agent for Open Agentic Web", "url": "https://arxiv.org/abs/2603.28428", "published": "2026-03-30", "updated": "2026-03-30", "authors": [ "Xiaohang Nie", "Zihan Guo", "Kezhuo Yang", "Zhichong Zheng", "Bochen Ge", "Shuai Pan", "Zeyi Chen", "Youling Xiang", "Yu Zhang", "Weiwen Liu", "Yuanjian Zhou", "Weinan Zhang" ], "categories": [ "cs.CY", "cs.MA" ], "topics": [ "coding-agent", "embodied-agent", "memory", "multi-agent", "rag", "tool-use" ], "score": 16, "relevance": "high", "matched_queries": [ "function-calling" ], "arxiv_id": "2603.28428", "source": "arxiv", "source_id": "arxiv:2603.28428", "pdf_url": "https://arxiv.org/pdf/2603.28428", "primary_query": "function-calling" }, { "id": "2603.28166", "title": "Evaluating Privilege Usage of Agents with Real-World Tools", "url": "https://arxiv.org/abs/2603.28166", "published": "2026-03-30", "updated": "2026-04-20", "authors": [ "Quan Zhang", "Lianhang Fu", "Lvsi Lian", "Gwihwan Go", "Yujue Wang", "Chijin Zhou", "Yu Jiang", "Geguang Pu" ], "categories": [ "cs.CR", "cs.AI" ], "topics": [ "agent-evaluation", "agent-safety", "rag", "tool-use", "workflow-agent" ], "score": 16, "relevance": "high", "matched_queries": [ "agent-safety" ], "arxiv_id": "2603.28166", "source": "arxiv", "source_id": "arxiv:2603.28166", "pdf_url": "https://arxiv.org/pdf/2603.28166", "primary_query": "agent-safety" }, { "id": "2603.21564", "title": "Toward a Theory of Hierarchical Memory for Language Agents", "url": "https://arxiv.org/abs/2603.21564", "published": "2026-03-23", "updated": "2026-03-23", "authors": [ "Yashar Talebirad", "Ali Parsaee", "Csongor Y. Szepesvari", "Amirhossein Nadiri", "Osmar Zaiane" ], "categories": [ "cs.IR", "cs.AI", "cs.IT", "cs.SI" ], "topics": [ "agent-evaluation", "memory", "rag", "tool-use" ], "score": 16, "relevance": "high", "matched_queries": [ "language-agent" ], "arxiv_id": "2603.21564", "source": "arxiv", "source_id": "arxiv:2603.21564", "pdf_url": "https://arxiv.org/pdf/2603.21564", "primary_query": "language-agent" }, { "id": "2603.21357", "title": "AgentHER: Hindsight Experience Replay for LLM Agent Trajectory Relabeling", "url": "https://arxiv.org/abs/2603.21357", "published": "2026-03-22", "updated": "2026-05-10", "authors": [ "Liang Ding" ], "categories": [ "cs.AI", "cs.CL" ], "topics": [ "computer-use", "memory", "tool-use", "workflow-agent" ], "score": 16, "relevance": "high", "matched_queries": [ "language-agent" ], "arxiv_id": "2603.21357", "source": "arxiv", "source_id": "arxiv:2603.21357", "pdf_url": "https://arxiv.org/pdf/2603.21357", "primary_query": "language-agent" }, { "id": "2603.05553", "title": "EigenData: A Self-Evolving Multi-Agent Platform for Function-Calling Data Synthesis, Auditing, and Repair", "url": "https://arxiv.org/abs/2603.05553", "published": "2026-03-05", "updated": "2026-03-05", "authors": [ "Jiaao Chen", "Jingyuan Qi", "Mingye Gao", "Wei-Chen Wang", "Hanrui Wang", "Di Jin" ], "categories": [ "cs.SE", "cs.AI", "cs.CL" ], "topics": [ "agent-evaluation", "coding-agent", "multi-agent", "tool-use" ], "score": 16, "relevance": "high", "matched_queries": [ "function-calling" ], "arxiv_id": "2603.05553", "source": "arxiv", "source_id": "arxiv:2603.05553", "pdf_url": "https://arxiv.org/pdf/2603.05553", "primary_query": "function-calling" }, { "id": "2602.07652", "title": "Agent-Fence: Mapping Security Vulnerabilities Across Deep Research Agents", "url": "https://arxiv.org/abs/2602.07652", "published": "2026-02-07", "updated": "2026-02-07", "authors": [ "Sai Puppala", "Ismail Hossain", "Md Jahangir Alam", "Yoonpyo Lee", "Jay Yoo", "Tanzim Ahad", "Syed Bahauddin Alam", "Sajedul Talukder" ], "categories": [ "cs.CR", "cs.AI" ], "topics": [ "agent-evaluation", "agent-safety", "memory", "planning", "rag", "tool-use" ], "score": 16, "relevance": "high", "matched_queries": [ "agent-safety" ], "arxiv_id": "2602.07652", "source": "arxiv", "source_id": "arxiv:2602.07652", "pdf_url": "https://arxiv.org/pdf/2602.07652", "primary_query": "agent-safety" }, { "id": "2602.05115", "title": "SocialVeil: Probing Social Intelligence of Language Agents under Communication Barriers", "url": "https://arxiv.org/abs/2602.05115", "published": "2026-02-04", "updated": "2026-02-04", "authors": [ "Keyang Xuan", "Pengda Wang", "Chongrui Ye", "Haofei Yu", "Tal August", "Jiaxuan You" ], "categories": [ "cs.AI", "cs.CL" ], "topics": [ "agent-evaluation", "rag", "tool-use" ], "score": 16, "relevance": "high", "matched_queries": [ "language-agent" ], "arxiv_id": "2602.05115", "source": "arxiv", "source_id": "arxiv:2602.05115", "pdf_url": "https://arxiv.org/pdf/2602.05115", "primary_query": "language-agent" }, { "id": "2602.03786", "title": "AOrchestra: Automating Sub-Agent Creation for Agentic Orchestration", "url": "https://arxiv.org/abs/2602.03786", "published": "2026-02-03", "updated": "2026-02-07", "authors": [ "Jianhao Ruan", "Zhihao Xu", "Yiran Peng", "Fashen Ren", "Zhaoyang Yu", "Xinbing Liang", "Jinyu Xiang", "Yongru Chen", "Bang Liu", "Chenglin Wu", "Yuyu Luo", "Jiayi Zhang" ], "categories": [ "cs.AI", "cs.CL" ], "topics": [ "agent-evaluation", "coding-agent", "planning", "tool-use", "workflow-agent" ], "score": 16, "relevance": "high", "matched_queries": [ "language-agent" ], "arxiv_id": "2602.03786", "source": "arxiv", "source_id": "arxiv:2602.03786", "pdf_url": "https://arxiv.org/pdf/2602.03786", "primary_query": "language-agent" }, { "id": "2601.00268", "title": "Beyond Perfect APIs: A Comprehensive Evaluation of LLM Agents Under Real-World API Complexity", "url": "https://arxiv.org/abs/2601.00268", "published": "2026-01-01", "updated": "2026-01-01", "authors": [ "Doyoung Kim", "Zhiwei Ren", "Jie Hao", "Zhongkai Sun", "Lichao Wang", "Xiyao Ma", "Zack Ye", "Xu Han", "Jun Yin", "Heng Ji", "Wei Shen", "Xing Fan", "Benjamin Yao", "Chenlei Guo" ], "categories": [ "cs.CL", "cs.AI" ], "topics": [ "agent-evaluation", "tool-use" ], "score": 16, "relevance": "high", "matched_queries": [ "function-calling" ], "arxiv_id": "2601.00268", "source": "arxiv", "source_id": "arxiv:2601.00268", "pdf_url": "https://arxiv.org/pdf/2601.00268", "primary_query": "function-calling" }, { "id": "2511.22138", "title": "TinyLLM: Evaluation and Optimization of Small Language Models for Agentic Tasks on Edge Devices", "url": "https://arxiv.org/abs/2511.22138", "published": "2025-11-27", "updated": "2025-11-27", "authors": [ "Mohd Ariful Haque", "Fahad Rahman", "Kishor Datta Gupta", "Khalil Shujaee", "Roy George" ], "categories": [ "cs.LG" ], "topics": [ "agent-evaluation", "agent-safety", "tool-use" ], "score": 16, "relevance": "high", "matched_queries": [ "function-calling" ], "arxiv_id": "2511.22138", "source": "arxiv", "source_id": "arxiv:2511.22138", "pdf_url": "https://arxiv.org/pdf/2511.22138", "primary_query": "function-calling" }, { "id": "2511.11169", "title": "Refine and Align: Confidence Calibration through Multi-Agent Interaction in VQA", "url": "https://arxiv.org/abs/2511.11169", "published": "2025-11-14", "updated": "2025-11-14", "authors": [ "Ayush Pandey", "Jai Bardhan", "Ishita Jain", "Ramya S Hebbalaguppe", "Rohan Raju Dhanakshirur", "Lovekesh Vig" ], "categories": [ "cs.CV", "cs.AI", "cs.LG" ], "topics": [ "agent-evaluation", "embodied-agent", "multi-agent", "tool-use" ], "score": 16, "relevance": "high", "matched_queries": [ "function-calling" ], "arxiv_id": "2511.11169", "source": "arxiv", "source_id": "arxiv:2511.11169", "pdf_url": "https://arxiv.org/pdf/2511.11169", "primary_query": "function-calling" }, { "id": "2510.21524", "title": "EU-Agent-Bench: Measuring Illegal Behavior of LLM Agents Under EU Law", "url": "https://arxiv.org/abs/2510.21524", "published": "2025-10-24", "updated": "2025-10-24", "authors": [ "Ilija Lichkovski", "Alexander Müller", "Mariam Ibrahim", "Tiwai Mhundwa" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "agent-safety", "rag", "tool-use" ], "score": 16, "relevance": "high", "matched_queries": [ "function-calling" ], "arxiv_id": "2510.21524", "source": "arxiv", "source_id": "arxiv:2510.21524", "pdf_url": "https://arxiv.org/pdf/2510.21524", "primary_query": "function-calling" }, { "id": "2509.08863", "title": "GeoJSON Agents:A Multi-Agent LLM Architecture for Geospatial Analysis-Function Calling vs Code Generation", "url": "https://arxiv.org/abs/2509.08863", "published": "2025-09-10", "updated": "2025-12-03", "authors": [ "Qianqian Luo", "Qingming Lin", "Liuchang Xu", "Sensen Wu", "Ruichen Mao", "Chao Wang", "Hailin Feng", "Bo Huang", "Zhenhong Du" ], "categories": [ "cs.SE" ], "topics": [ "agent-evaluation", "multi-agent", "planning", "tool-use", "workflow-agent" ], "score": 16, "relevance": "high", "matched_queries": [ "function-calling" ], "arxiv_id": "2509.08863", "source": "arxiv", "source_id": "arxiv:2509.08863", "pdf_url": "https://arxiv.org/pdf/2509.08863", "primary_query": "function-calling" }, { "id": "2508.11027", "title": "Hell or High Water: Evaluating Agentic Recovery from External Failures", "url": "https://arxiv.org/abs/2508.11027", "published": "2025-08-14", "updated": "2025-08-14", "authors": [ "Andrew Wang", "Sophia Hager", "Adi Asija", "Daniel Khashabi", "Nicholas Andrews" ], "categories": [ "cs.CL" ], "topics": [ "agent-evaluation", "planning", "tool-use", "workflow-agent" ], "score": 16, "relevance": "high", "matched_queries": [ "function-calling" ], "arxiv_id": "2508.11027", "source": "arxiv", "source_id": "arxiv:2508.11027", "pdf_url": "https://arxiv.org/pdf/2508.11027", "primary_query": "function-calling" }, { "id": "2607.06223", "title": "Information Gain-based Rollout Policy Optimization: An Adaptive Tree-Structured Rollout Approach for Multi-Turn LLM Agents", "url": "https://arxiv.org/abs/2607.06223", "published": "2026-07-07", "updated": "2026-07-07", "authors": [ "Yijun Zhang", "Fan Xu", "Jiaxin Ding", "Yule Xie", "Shiqing Gao", "Xin Ding", "Haoxiang Zhang", "Luoyi Fu", "Xinbing Wang" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "computer-use", "planning", "rag" ], "score": 15, "relevance": "high", "matched_queries": [ "llm-agent" ], "arxiv_id": "2607.06223", "source": "arxiv", "source_id": "arxiv:2607.06223", "pdf_url": "https://arxiv.org/pdf/2607.06223", "primary_query": "llm-agent" }, { "id": "2607.05915", "title": "PCBWorld: A Benchmark Environment for Engine-Grounded PCB Design Automation", "url": "https://arxiv.org/abs/2607.05915", "published": "2026-07-07", "updated": "2026-07-07", "authors": [ "Hyungseok Song", "Junseok Park", "Won-Seok Choi", "Seohui Bae", "Han-Seul Jeong", "Youngjoon Park", "Soonyoung Lee" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "tool-use", "workflow-agent" ], "score": 15, "relevance": "high", "matched_queries": [ "agentic-ai", "llm-agent", "tool-use" ], "arxiv_id": "2607.05915", "source": "arxiv", "source_id": "arxiv:2607.05915", "pdf_url": "https://arxiv.org/pdf/2607.05915", "primary_query": "agentic-ai" }, { "id": "2607.05805", "title": "Onnes: A Physics-Grounded Multi-Agent LLM Simulator for Cryogenic Fault Diagnosis in Quantum Computing Infrastructure", "url": "https://arxiv.org/abs/2607.05805", "published": "2026-07-07", "updated": "2026-07-07", "authors": [ "Praneeth Narisetty", "Uday Kumar Reddy Kattamanchi", "Shiva Nagendra Babu Kore" ], "categories": [ "cs.AI", "cs.LG", "quant-ph" ], "topics": [ "agent-evaluation", "multi-agent" ], "score": 15, "relevance": "high", "matched_queries": [ "llm-agent", "multi-agent-llm" ], "arxiv_id": "2607.05805", "source": "arxiv", "source_id": "arxiv:2607.05805", "pdf_url": "https://arxiv.org/pdf/2607.05805", "primary_query": "llm-agent" }, { "id": "2607.05743", "title": "The Balkanization of Execution-Security Research for AI Coding Agents: Isolation, Access Control, and Time-of-Check-to-Time-of-Use Vulnerabilities", "url": "https://arxiv.org/abs/2607.05743", "published": "2026-07-07", "updated": "2026-07-07", "authors": [ "Mohammadreza Rashidi" ], "categories": [ "cs.CR", "cs.AI" ], "topics": [ "agent-evaluation", "agent-safety", "coding-agent", "tool-use" ], "score": 15, "relevance": "high", "matched_queries": [ "agentic-ai", "coding-agent" ], "arxiv_id": "2607.05743", "source": "arxiv", "source_id": "arxiv:2607.05743", "pdf_url": "https://arxiv.org/pdf/2607.05743", "primary_query": "agentic-ai" }, { "id": "2607.05690", "title": "Memory in the Loop: In-Process Retrieval as ExtendedWorking Memory for Language Agents", "url": "https://arxiv.org/abs/2607.05690", "published": "2026-07-06", "updated": "2026-07-06", "authors": [ "Yusuf Khan", "Carlo Lipizzi" ], "categories": [ "cs.AI", "cs.CL" ], "topics": [ "agent-evaluation", "memory", "rag", "tool-use" ], "score": 15, "relevance": "high", "matched_queries": [ "language-agent" ], "arxiv_id": "2607.05690", "source": "arxiv", "source_id": "arxiv:2607.05690", "pdf_url": "https://arxiv.org/pdf/2607.05690", "primary_query": "language-agent" }, { "id": "2607.05055", "title": "Toward Trustworthy Large Language Model Agents in Healthcare", "url": "https://arxiv.org/abs/2607.05055", "published": "2026-07-06", "updated": "2026-07-06", "authors": [ "Hadi Hasan", "Safaa Salman", "Adam Tai Abou Dargham", "Ammar Mohanna", "Ali Chehab" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "agent-safety", "rag", "tool-use", "workflow-agent" ], "score": 15, "relevance": "high", "matched_queries": [ "function-calling", "rag-agent" ], "arxiv_id": "2607.05055", "source": "arxiv", "source_id": "arxiv:2607.05055", "pdf_url": "https://arxiv.org/pdf/2607.05055", "primary_query": "function-calling" }, { "id": "2607.04089", "title": "PLACEMEM: Toward a Compute-Aware Memory Plane for Lifelong Agents", "url": "https://arxiv.org/abs/2607.04089", "published": "2026-07-05", "updated": "2026-07-05", "authors": [ "Sukanta Ganguly" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "memory", "planning", "rag" ], "score": 15, "relevance": "high", "matched_queries": [ "agent-memory" ], "arxiv_id": "2607.04089", "source": "arxiv", "source_id": "arxiv:2607.04089", "pdf_url": "https://arxiv.org/pdf/2607.04089", "primary_query": "agent-memory" }, { "id": "2607.03702", "title": "Agent Reinforcement Learning via Pivotal-Aware Self-Feedback Retry", "url": "https://arxiv.org/abs/2607.03702", "published": "2026-07-04", "updated": "2026-07-04", "authors": [ "Weiyang Guo", "Zesheng Shi", "Longhui Zhang", "Zeen Zhu", "Min Zhang", "Jing Li" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "planning", "rag", "reasoning", "tool-use" ], "score": 15, "relevance": "high", "matched_queries": [ "llm-agent" ], "arxiv_id": "2607.03702", "source": "arxiv", "source_id": "arxiv:2607.03702", "pdf_url": "https://arxiv.org/pdf/2607.03702", "primary_query": "llm-agent" }, { "id": "2607.03695", "title": "Social Networks of LLM Agents", "url": "https://arxiv.org/abs/2607.03695", "published": "2026-07-04", "updated": "2026-07-04", "authors": [ "Kaixuan Liu", "Guojun Xiong", "Weinan Zhang", "Shengpu Tang" ], "categories": [ "cs.LG" ], "topics": [ "agent-evaluation", "multi-agent" ], "score": 15, "relevance": "high", "matched_queries": [ "llm-agent", "multi-agent-llm" ], "arxiv_id": "2607.03695", "source": "arxiv", "source_id": "arxiv:2607.03695", "pdf_url": "https://arxiv.org/pdf/2607.03695", "primary_query": "llm-agent" }, { "id": "2607.03821", "title": "DualView: Preventing Indirect Prompt Injection in Personal AI Agents", "url": "https://arxiv.org/abs/2607.03821", "published": "2026-07-04", "updated": "2026-07-04", "authors": [ "Juhee Kim", "Woohyuk Choi", "Taehyun Kang", "Youngmin Kim", "Byoungyoung Lee" ], "categories": [ "cs.CR", "cs.AI" ], "topics": [ "agent-evaluation", "agent-safety", "tool-use" ], "score": 15, "relevance": "high", "matched_queries": [ "ai-agent" ], "arxiv_id": "2607.03821", "source": "arxiv", "source_id": "arxiv:2607.03821", "pdf_url": "https://arxiv.org/pdf/2607.03821", "primary_query": "ai-agent" }, { "id": "2607.04009", "title": "PhysMiner: An Agentic AI Framework for Discovering Turbulence Physics", "url": "https://arxiv.org/abs/2607.04009", "published": "2026-07-04", "updated": "2026-07-04", "authors": [ "Jiawei Chen", "Han Gao", "Ping He" ], "categories": [ "physics.flu-dyn" ], "topics": [ "agent-evaluation", "memory", "reasoning", "tool-use" ], "score": 15, "relevance": "high", "matched_queries": [ "agentic-ai" ], "arxiv_id": "2607.04009", "source": "arxiv", "source_id": "arxiv:2607.04009", "pdf_url": "https://arxiv.org/pdf/2607.04009", "primary_query": "agentic-ai" }, { "id": "2607.03968", "title": "Refused in Chat, Written in Code: Workflow-Level Jailbreak Construction in IDE Coding Agents", "url": "https://arxiv.org/abs/2607.03968", "published": "2026-07-04", "updated": "2026-07-04", "authors": [ "Abhishek Kumar", "Carsten Maple" ], "categories": [ "cs.SE", "cs.AI" ], "topics": [ "agent-evaluation", "agent-safety", "coding-agent", "workflow-agent" ], "score": 15, "relevance": "high", "matched_queries": [ "coding-agent" ], "arxiv_id": "2607.03968", "source": "arxiv", "source_id": "arxiv:2607.03968", "pdf_url": "https://arxiv.org/pdf/2607.03968", "primary_query": "coding-agent" }, { "id": "2607.02846", "title": "Object-Centric Environment Modeling for Agentic Tasks", "url": "https://arxiv.org/abs/2607.02846", "published": "2026-07-03", "updated": "2026-07-03", "authors": [ "Yiyang Li", "Tianyi Ma", "Zehong Wang", "Yijun Ma", "Yanfang Ye" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "rag", "tool-use", "world-model" ], "score": 15, "relevance": "high", "matched_queries": [ "llm-agent" ], "arxiv_id": "2607.02846", "source": "arxiv", "source_id": "arxiv:2607.02846", "pdf_url": "https://arxiv.org/pdf/2607.02846", "primary_query": "llm-agent" }, { "id": "2607.03269", "title": "Agentic-SecPBFT: Agentic AI-Driven Proactive Security Framework for Wireless PBFT Consensus in Mobile Ad-Hoc Networks", "url": "https://arxiv.org/abs/2607.03269", "published": "2026-07-03", "updated": "2026-07-03", "authors": [ "Haoxiang Luo", "Yinqiu Liu", "Ruichen Zhang", "Guangyuan Liu", "Gang Sun", "Hongfang Yu", "Zhu Han", "Dong In Kim" ], "categories": [ "cs.NI" ], "topics": [ "agent-safety", "computer-use", "multi-agent", "rag", "tool-use", "world-model" ], "score": 15, "relevance": "high", "matched_queries": [ "agentic-ai" ], "arxiv_id": "2607.03269", "source": "arxiv", "source_id": "arxiv:2607.03269", "pdf_url": "https://arxiv.org/pdf/2607.03269", "primary_query": "agentic-ai" }, { "id": "2607.02911", "title": "CoACT: Action-Preserving Observation Compression for Coding Agents", "url": "https://arxiv.org/abs/2607.02911", "published": "2026-07-03", "updated": "2026-07-03", "authors": [ "Haorui Chen", "Yuancheng Zhu", "Yitong Zhang", "Jia Li" ], "categories": [ "cs.SE", "cs.AI" ], "topics": [ "coding-agent", "rag", "tool-use" ], "score": 15, "relevance": "high", "matched_queries": [ "coding-agent" ], "arxiv_id": "2607.02911", "source": "arxiv", "source_id": "arxiv:2607.02911", "pdf_url": "https://arxiv.org/pdf/2607.02911", "primary_query": "coding-agent" }, { "id": "2607.01767", "title": "Repair the Amplifier, Not the Symptom: Stable World-Model Correction for Agent Rollouts", "url": "https://arxiv.org/abs/2607.01767", "published": "2026-07-02", "updated": "2026-07-05", "authors": [ "Xinyuan Song", "Zekun Cai" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "memory", "planning", "tool-use", "world-model" ], "score": 15, "relevance": "high", "matched_queries": [ "language-agent" ], "arxiv_id": "2607.01767", "source": "arxiv", "source_id": "arxiv:2607.01767", "pdf_url": "https://arxiv.org/pdf/2607.01767", "primary_query": "language-agent" }, { "id": "2607.01709", "title": "COMFYCLAW: Self-Evolving Skill Harnesses for Image Generation Workflows", "url": "https://arxiv.org/abs/2607.01709", "published": "2026-07-02", "updated": "2026-07-02", "authors": [ "Zongxia Li", "Dawei Liu", "Fuxiao Liu", "Yuhang Zhou", "Xiyang Wu", "Jingxi Chen", "Jing Xie", "Xiaomin Wu", "Lichao Sun" ], "categories": [ "cs.AI", "cs.LG" ], "topics": [ "agent-evaluation", "memory", "rag", "reasoning", "tool-use", "workflow-agent" ], "score": 15, "relevance": "high", "matched_queries": [ "agent-memory" ], "arxiv_id": "2607.01709", "source": "arxiv", "source_id": "arxiv:2607.01709", "pdf_url": "https://arxiv.org/pdf/2607.01709", "primary_query": "agent-memory" }, { "id": "2607.02807", "title": "SwarmResearch: Orchestrating Coding Agents for Open-Ended Discovery", "url": "https://arxiv.org/abs/2607.02807", "published": "2026-07-02", "updated": "2026-07-02", "authors": [ "Yuvraj Virk", "Zack Edds", "Chunqiu Steven Xia", "Lingming Zhang" ], "categories": [ "cs.AI" ], "topics": [ "coding-agent", "computer-use", "multi-agent" ], "score": 15, "relevance": "high", "matched_queries": [ "coding-agent", "multi-agent-llm" ], "arxiv_id": "2607.02807", "source": "arxiv", "source_id": "arxiv:2607.02807", "pdf_url": "https://arxiv.org/pdf/2607.02807", "primary_query": "coding-agent" }, { "id": "2607.02448", "title": "AgentsCAD: Automated Design for Manufacturing of FDM Parts via Multi-Agent LLM Reasoning and Geometric Feature Recognition", "url": "https://arxiv.org/abs/2607.02448", "published": "2026-07-02", "updated": "2026-07-07", "authors": [ "Emmanuel George", "Christopher Keefe", "Peter Pak", "Amir Barati Farimani" ], "categories": [ "cs.MA" ], "topics": [ "agent-evaluation", "multi-agent", "reasoning", "workflow-agent" ], "score": 15, "relevance": "high", "matched_queries": [ "multi-agent-llm" ], "arxiv_id": "2607.02448", "source": "arxiv", "source_id": "arxiv:2607.02448", "pdf_url": "https://arxiv.org/pdf/2607.02448", "primary_query": "multi-agent-llm" }, { "id": "2607.01213", "title": "RepoRescue: An Empirical Study of LLM Agents on Whole-Repository Compatibility Rescue", "url": "https://arxiv.org/abs/2607.01213", "published": "2026-07-01", "updated": "2026-07-01", "authors": [ "Zhihao Lin", "Mingyi Zhou", "Zhensu Sun", "Yizhuo Yang", "Renyu Yang", "David Lo", "Li Li" ], "categories": [ "cs.SE" ], "topics": [ "agent-evaluation", "coding-agent", "reasoning", "tool-use" ], "score": 15, "relevance": "high", "matched_queries": [ "llm-agent" ], "arxiv_id": "2607.01213", "source": "arxiv", "source_id": "arxiv:2607.01213", "pdf_url": "https://arxiv.org/pdf/2607.01213", "primary_query": "llm-agent" }, { "id": "2607.01120", "title": "Next-Generation Agentic Reinforcement Learning Systems Enable Self-Evolving Agents", "url": "https://arxiv.org/abs/2607.01120", "published": "2026-07-01", "updated": "2026-07-02", "authors": [ "Ran Yan", "Wei Fu", "Jiale Li", "Shusheng Xu", "Zhiyu Mei", "Jiaxuan Gao", "Jiarui Zhang", "Wentai Zhang", "Hao Dai", "Xujie Shen", "Chuyi He", "Zhen Pu", "Jun Mei", "Zhiyao Lin", "Haitao Wang", "Zhiqiang Ding", "Jiawei Zhang", "Huaijie Wang", "Ruida Xu", "Honghua Dong", "Youhe Jiang", "Yi Wu", "Tongkai Yang", "Binhang Yuan" ], "categories": [ "cs.DC" ], "topics": [ "coding-agent", "planning", "tool-use", "workflow-agent" ], "score": 15, "relevance": "high", "matched_queries": [ "llm-agent" ], "arxiv_id": "2607.01120", "source": "arxiv", "source_id": "arxiv:2607.01120", "pdf_url": "https://arxiv.org/pdf/2607.01120", "primary_query": "llm-agent" }, { "id": "2607.00345", "title": "Registry-Governed Agent Lifecycle:Completing EDDOps with Evaluation-DrivenRegistration, Promotion, and Retirement on AWS AgentCore", "url": "https://arxiv.org/abs/2607.00345", "published": "2026-07-01", "updated": "2026-07-01", "authors": [ "Richard Kang", "Vincent Wang" ], "categories": [ "cs.SE" ], "topics": [ "agent-evaluation", "agent-safety", "workflow-agent" ], "score": 15, "relevance": "high", "matched_queries": [ "llm-agent" ], "arxiv_id": "2607.00345", "source": "arxiv", "source_id": "arxiv:2607.00345", "pdf_url": "https://arxiv.org/pdf/2607.00345", "primary_query": "llm-agent" }, { "id": "2607.00911", "title": "From Registry to Repository: How AI Agent Skills Are Written, Adapted, and Maintained", "url": "https://arxiv.org/abs/2607.00911", "published": "2026-07-01", "updated": "2026-07-06", "authors": [ "Haoyu Gao", "Jai Lal Lulla", "Hong Yi Lin", "Sebastian Baltes", "Christoph Treude", "Mansooreh Zahedi" ], "categories": [ "cs.SE" ], "topics": [ "coding-agent", "tool-use", "workflow-agent" ], "score": 15, "relevance": "high", "matched_queries": [ "ai-agent", "coding-agent" ], "arxiv_id": "2607.00911", "source": "arxiv", "source_id": "arxiv:2607.00911", "pdf_url": "https://arxiv.org/pdf/2607.00911", "primary_query": "ai-agent" }, { "id": "2607.01421", "title": "Risk Architecture for AI-Native Engineering Teams: An Organizational Framework for Agentic System Governance", "url": "https://arxiv.org/abs/2607.01421", "published": "2026-07-01", "updated": "2026-07-01", "authors": [ "Laxmipriya Ganesh Iyer" ], "categories": [ "cs.SE", "cs.AI" ], "topics": [ "agent-evaluation", "agent-safety", "coding-agent", "computer-use", "rag", "tool-use" ], "score": 15, "relevance": "high", "matched_queries": [ "agentic-ai" ], "arxiv_id": "2607.01421", "source": "arxiv", "source_id": "arxiv:2607.01421", "pdf_url": "https://arxiv.org/pdf/2607.01421", "primary_query": "agentic-ai" }, { "id": "2607.01061", "title": "Agentic generation of verifiable rules for deterministic, self-expanding reaction classification", "url": "https://arxiv.org/abs/2607.01061", "published": "2026-07-01", "updated": "2026-07-05", "authors": [ "Daniel Armstrong", "Maarten Dobbelaere", "Valentas Olikauskas", "Helena Avila", "Octavian Susanu", "Jérôme Waser", "Philippe Schwaller" ], "categories": [ "cs.AI", "cs.CL" ], "topics": [ "coding-agent", "multi-agent", "planning", "tool-use" ], "score": 15, "relevance": "high", "matched_queries": [ "multi-agent-llm" ], "arxiv_id": "2607.01061", "source": "arxiv", "source_id": "arxiv:2607.01061", "pdf_url": "https://arxiv.org/pdf/2607.01061", "primary_query": "multi-agent-llm" }, { "id": "2606.31744", "title": "A Conversational Agentic Interface to Physics-Based Household Digital Twins for Residential Energy Decision Support", "url": "https://arxiv.org/abs/2606.31744", "published": "2026-06-30", "updated": "2026-06-30", "authors": [ "Costas Mylonas", "Titos Georgoulakis", "Magda Foti" ], "categories": [ "eess.SY" ], "topics": [ "agent-evaluation", "planning", "rag", "tool-use", "world-model" ], "score": 15, "relevance": "high", "matched_queries": [ "llm-agent" ], "arxiv_id": "2606.31744", "source": "arxiv", "source_id": "arxiv:2606.31744", "pdf_url": "https://arxiv.org/pdf/2606.31744", "primary_query": "llm-agent" }, { "id": "2606.31209", "title": "Long-term Traffic Simulation via Structured Autoregressive Modeling", "url": "https://arxiv.org/abs/2606.31209", "published": "2026-06-30", "updated": "2026-06-30", "authors": [ "Lingyu Xiao", "Zexin Feng", "Xintao Yan" ], "categories": [ "cs.AI", "cs.RO" ], "topics": [ "agent-evaluation", "agent-safety", "memory", "multi-agent", "planning", "rag", "tool-use", "world-model" ], "score": 15, "relevance": "high", "matched_queries": [ "multi-agent-llm" ], "arxiv_id": "2606.31209", "source": "arxiv", "source_id": "arxiv:2606.31209", "pdf_url": "https://arxiv.org/pdf/2606.31209", "primary_query": "multi-agent-llm" }, { "id": "2606.31085", "title": "DDIAgents: Mechanism-Conditioned Context Flow for Drug-Drug Interaction Prediction", "url": "https://arxiv.org/abs/2606.31085", "published": "2026-06-30", "updated": "2026-06-30", "authors": [ "Zhenqian Shen", "Yu Liu", "Xiaoyi Fu", "Quanming Yao" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "agent-safety", "multi-agent", "planning", "reasoning", "tool-use" ], "score": 15, "relevance": "high", "matched_queries": [ "multi-agent-llm" ], "arxiv_id": "2606.31085", "source": "arxiv", "source_id": "arxiv:2606.31085", "pdf_url": "https://arxiv.org/pdf/2606.31085", "primary_query": "multi-agent-llm" }, { "id": "2606.30383", "title": "Whose Side Is Your Agent On? Multi-Party Principal Loyalty in LLM Agents", "url": "https://arxiv.org/abs/2606.30383", "published": "2026-06-29", "updated": "2026-06-29", "authors": [ "Bojie Li", "Noah Shi" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "agent-safety", "tool-use" ], "score": 15, "relevance": "high", "matched_queries": [ "llm-agent" ], "arxiv_id": "2606.30383", "source": "arxiv", "source_id": "arxiv:2606.30383", "pdf_url": "https://arxiv.org/pdf/2606.30383", "primary_query": "llm-agent" }, { "id": "2606.30697", "title": "LUMOS: A Semantic Operating-System Layer for Accessibility-Grounded AI Agents", "url": "https://arxiv.org/abs/2606.30697", "published": "2026-06-29", "updated": "2026-06-29", "authors": [ "Yogeswar Reddy Thota" ], "categories": [ "cs.OS", "cs.AI", "cs.CV" ], "topics": [ "computer-use", "rag", "tool-use", "workflow-agent" ], "score": 15, "relevance": "high", "matched_queries": [ "ai-agent", "web-gui-agent" ], "arxiv_id": "2606.30697", "source": "arxiv", "source_id": "arxiv:2606.30697", "pdf_url": "https://arxiv.org/pdf/2606.30697", "primary_query": "ai-agent" }, { "id": "2606.30877", "title": "A Systematic Approach to Multi-Agent AI from Advanced Regulatory Control Theory: Safe and Auditable LLM Operator Agents for Process Control", "url": "https://arxiv.org/abs/2606.30877", "published": "2026-06-29", "updated": "2026-06-29", "authors": [ "Idelfonso B. R. Nogueira", "Sigurd Skogestad" ], "categories": [ "eess.SY", "cs.LG" ], "topics": [ "agent-evaluation", "agent-safety", "multi-agent", "tool-use" ], "score": 15, "relevance": "high", "matched_queries": [ "agentic-ai", "multi-agent-llm" ], "arxiv_id": "2606.30877", "source": "arxiv", "source_id": "arxiv:2606.30877", "pdf_url": "https://arxiv.org/pdf/2606.30877", "primary_query": "agentic-ai" }, { "id": "2606.29957", "title": "SWE-Together: Evaluating Coding Agents in Interactive User Sessions", "url": "https://arxiv.org/abs/2606.29957", "published": "2026-06-29", "updated": "2026-06-29", "authors": [ "Yifan Wu", "Zhuokai Zhao", "Songlin Li", "Ho Hin Lee", "Jiacheng Zhu", "Shirley Wu", "Tianhe Yu", "Serena Li", "Lizhu Zhang", "Xiangjun Fan", "Shengzhi Li" ], "categories": [ "cs.SE", "cs.AI" ], "topics": [ "agent-evaluation", "coding-agent", "tool-use" ], "score": 15, "relevance": "high", "matched_queries": [ "agent-evaluation", "coding-agent" ], "arxiv_id": "2606.29957", "source": "arxiv", "source_id": "arxiv:2606.29957", "pdf_url": "https://arxiv.org/pdf/2606.29957", "primary_query": "agent-evaluation" }, { "id": "2606.29778", "title": "Mandol: An Agglomerative Agent Memory System for Long-Term Conversations", "url": "https://arxiv.org/abs/2606.29778", "published": "2026-06-29", "updated": "2026-06-29", "authors": [ "Yuhan Zhang", "Zhiyuan Guo", "Ziheng Zeng", "Wei Wang", "Wentao Wu", "Lijie Xu" ], "categories": [ "cs.DB", "cs.AI", "cs.CL", "cs.IR" ], "topics": [ "agent-evaluation", "memory", "rag" ], "score": 15, "relevance": "high", "matched_queries": [ "agent-memory", "rag-agent" ], "arxiv_id": "2606.29778", "source": "arxiv", "source_id": "arxiv:2606.29778", "pdf_url": "https://arxiv.org/pdf/2606.29778", "primary_query": "agent-memory" }, { "id": "2606.30573", "title": "SWE-INTERACT: Reimagining SWE Benchmarks as User-Driven Long-Horizon Coding Sessions", "url": "https://arxiv.org/abs/2606.30573", "published": "2026-06-29", "updated": "2026-06-29", "authors": [ "Mohit Raghavendra", "Anisha Gunjal", "Aakash Sabharwal", "Yunzhong He" ], "categories": [ "cs.LG" ], "topics": [ "agent-evaluation", "coding-agent", "computer-use", "planning", "tool-use", "workflow-agent" ], "score": 15, "relevance": "high", "matched_queries": [ "coding-agent" ], "arxiv_id": "2606.30573", "source": "arxiv", "source_id": "arxiv:2606.30573", "pdf_url": "https://arxiv.org/pdf/2606.30573", "primary_query": "coding-agent" }, { "id": "2606.30560", "title": "TraceLab: Characterizing Coding Agent Workloads for LLM Serving", "url": "https://arxiv.org/abs/2606.30560", "published": "2026-06-29", "updated": "2026-06-30", "authors": [ "Kan Zhu", "Mathew Jacob", "Chenxi Ma", "Yi Pan", "Stephanie Wang", "Arvind Krishnamurthy", "Baris Kasikci" ], "categories": [ "cs.LG", "cs.AI", "cs.PF" ], "topics": [ "agent-evaluation", "coding-agent", "tool-use" ], "score": 15, "relevance": "high", "matched_queries": [ "coding-agent" ], "arxiv_id": "2606.30560", "source": "arxiv", "source_id": "arxiv:2606.30560", "pdf_url": "https://arxiv.org/pdf/2606.30560", "primary_query": "coding-agent" }, { "id": "2606.30887", "title": "Training Therapeutic Judges and Multi-Agent Systems for Human-Aligned Mental Health Support", "url": "https://arxiv.org/abs/2606.30887", "published": "2026-06-29", "updated": "2026-06-29", "authors": [ "Mizanur Rahman", "Abeer Badawi", "Elahe Rahimi", "Laleh Seyyed-Kalantari", "Frank Rudzicz", "Enamul Hoque", "Elham Dolatabadi" ], "categories": [ "cs.CL", "cs.AI", "cs.MA" ], "topics": [ "agent-evaluation", "agent-safety", "multi-agent", "tool-use" ], "score": 15, "relevance": "high", "matched_queries": [ "multi-agent-llm" ], "arxiv_id": "2606.30887", "source": "arxiv", "source_id": "arxiv:2606.30887", "pdf_url": "https://arxiv.org/pdf/2606.30887", "primary_query": "multi-agent-llm" }, { "id": "2607.00038", "title": "Stop Hand-Holding Your Coding Agent: Engineering the Loops that Replace Step-by-Step Prompting", "url": "https://arxiv.org/abs/2607.00038", "published": "2026-06-28", "updated": "2026-06-28", "authors": [ "Sandeco Macedo" ], "categories": [ "cs.SE" ], "topics": [ "agent-safety", "coding-agent", "computer-use", "memory", "rag", "tool-use" ], "score": 15, "relevance": "high", "matched_queries": [ "coding-agent" ], "arxiv_id": "2607.00038", "source": "arxiv", "source_id": "arxiv:2607.00038", "pdf_url": "https://arxiv.org/pdf/2607.00038", "primary_query": "coding-agent" }, { "id": "2606.29014", "title": "Customized Generative AI Agent for Transportation Engineering Practice: A Development and Continued Pre-training Guideline", "url": "https://arxiv.org/abs/2606.29014", "published": "2026-06-27", "updated": "2026-07-03", "authors": [ "Dianwei Chen", "Yuan-Zheng Lei", "Zifan Zhang", "Yuchen Liu", "Xianfeng Yang" ], "categories": [ "cs.AI", "cs.DL" ], "topics": [ "agent-evaluation", "agent-safety", "computer-use", "planning", "reasoning" ], "score": 15, "relevance": "high", "matched_queries": [ "ai-agent" ], "arxiv_id": "2606.29014", "source": "arxiv", "source_id": "arxiv:2606.29014", "pdf_url": "https://arxiv.org/pdf/2606.29014", "primary_query": "ai-agent" }, { "id": "2606.28781", "title": "HyphaeDB: A Living Knowledge Topology for Agent-First Memory", "url": "https://arxiv.org/abs/2606.28781", "published": "2026-06-27", "updated": "2026-06-27", "authors": [ "Krishna Halaharvi" ], "categories": [ "cs.AI", "cs.MA" ], "topics": [ "coding-agent", "memory", "multi-agent", "rag", "tool-use" ], "score": 15, "relevance": "high", "matched_queries": [ "agent-memory", "agentic-ai" ], "arxiv_id": "2606.28781", "source": "arxiv", "source_id": "arxiv:2606.28781", "pdf_url": "https://arxiv.org/pdf/2606.28781", "primary_query": "agent-memory" }, { "id": "2606.28839", "title": "The Contagion Tensor: A Framework for Measuring Output-Distribution Coupling in Multi-Agent LLM Systems -- and Auditing the Claims It Enables", "url": "https://arxiv.org/abs/2606.28839", "published": "2026-06-27", "updated": "2026-06-27", "authors": [ "Zewen Liu" ], "categories": [ "cs.LG" ], "topics": [ "agent-evaluation", "computer-use", "multi-agent", "tool-use", "world-model" ], "score": 15, "relevance": "high", "matched_queries": [ "multi-agent-llm" ], "arxiv_id": "2606.28839", "source": "arxiv", "source_id": "arxiv:2606.28839", "pdf_url": "https://arxiv.org/pdf/2606.28839", "primary_query": "multi-agent-llm" }, { "id": "2606.28270", "title": "Agent-Native Immune System: Architecture, Taxonomy, and Engineering", "url": "https://arxiv.org/abs/2606.28270", "published": "2026-06-26", "updated": "2026-06-26", "authors": [ "Bo Shen", "Lifeng Chang", "Tianyuan Wei", "Yunpeng Li", "Feng Shi", "Yichen Han", "Peijie Gao", "Shiyi Kuang", "Xin Chang", "Dehui Li" ], "categories": [ "cs.AI", "cs.MA" ], "topics": [ "agent-evaluation", "agent-safety", "memory", "multi-agent", "reasoning", "tool-use", "workflow-agent" ], "score": 15, "relevance": "high", "matched_queries": [ "tool-use" ], "arxiv_id": "2606.28270", "source": "arxiv", "source_id": "arxiv:2606.28270", "pdf_url": "https://arxiv.org/pdf/2606.28270", "primary_query": "tool-use" }, { "id": "2606.28436", "title": "Dockerless: Environment-Free Program Verifier for Coding Agents", "url": "https://arxiv.org/abs/2606.28436", "published": "2026-06-26", "updated": "2026-06-26", "authors": [ "Wenhao Zeng", "Yuling Shi", "Xiaodong Gu", "Chao Hu", "Chaofan Wang", "Yuhao Cui", "Hongting Zhou", "Mengnan Qi", "Jianqiao Wangni", "Zhaojian Yu", "Shuzheng Gao", "Kai Cai", "Shilin He" ], "categories": [ "cs.SE", "cs.AI" ], "topics": [ "agent-evaluation", "coding-agent", "reasoning" ], "score": 15, "relevance": "high", "matched_queries": [ "coding-agent" ], "arxiv_id": "2606.28436", "source": "arxiv", "source_id": "arxiv:2606.28436", "pdf_url": "https://arxiv.org/pdf/2606.28436", "primary_query": "coding-agent" }, { "id": "2606.26918", "title": "Diagnosing Task Insensitivity in Language Agents", "url": "https://arxiv.org/abs/2606.26918", "published": "2026-06-25", "updated": "2026-06-25", "authors": [ "Jingyu Liu", "Xiaopeng Wu", "Kehan Chen", "Chuan Yu", "Yong Liu" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "planning", "rag", "tool-use" ], "score": 15, "relevance": "high", "matched_queries": [ "language-agent" ], "arxiv_id": "2606.26918", "source": "arxiv", "source_id": "arxiv:2606.26918", "pdf_url": "https://arxiv.org/pdf/2606.26918", "primary_query": "language-agent" }, { "id": "2606.26524", "title": "VIGIL: Runtime Enforcement of Behavioral Specifications in AI Agent Skills", "url": "https://arxiv.org/abs/2606.26524", "published": "2026-06-25", "updated": "2026-06-25", "authors": [ "Ying Li", "Yanju Chen", "Hongbo Wen", "Bosi Zhang", "Hanzhi Liu", "Peiran Wang", "Yu Feng", "Yuan Tian" ], "categories": [ "cs.CR" ], "topics": [ "agent-evaluation", "tool-use", "workflow-agent" ], "score": 15, "relevance": "high", "matched_queries": [ "ai-agent" ], "arxiv_id": "2606.26524", "source": "arxiv", "source_id": "arxiv:2606.26524", "pdf_url": "https://arxiv.org/pdf/2606.26524", "primary_query": "ai-agent" }, { "id": "2606.27499", "title": "DMV-Bench: Diagnosing Long-Horizon Multimodal Agents' Visual Memory with Incidental Cue Injection", "url": "https://arxiv.org/abs/2606.27499", "published": "2026-06-25", "updated": "2026-06-25", "authors": [ "Yujin Tang", "Chenming Shang", "Ruize Xu", "Nikhil Singh" ], "categories": [ "cs.CV", "cs.AI", "cs.CL" ], "topics": [ "agent-evaluation", "coding-agent", "memory", "planning", "rag", "tool-use" ], "score": 15, "relevance": "high", "matched_queries": [ "agent-memory" ], "arxiv_id": "2606.27499", "source": "arxiv", "source_id": "arxiv:2606.27499", "pdf_url": "https://arxiv.org/pdf/2606.27499", "primary_query": "agent-memory" }, { "id": "2606.27243", "title": "NOVA: A Verification-Aware Agent Harness for Architecture Evolution in Industrial Recommender Systems", "url": "https://arxiv.org/abs/2606.27243", "published": "2026-06-25", "updated": "2026-06-26", "authors": [ "Shaohua Liu", "Liang Fang", "Yilong Sun", "Shudong Huang", "Qingsong Luo", "Shaoxin Liu", "Xiaoyang Chen", "Dongqiang Liu", "Chuangang Ma", "Zhenzhen Chai", "Henghuan Wang", "Shijie Quan", "Changyuan Cui", "Zhangbin Zhu", "Peng Chen", "Wei Xu", "Lei Xiao", "Haijie Gu", "Jie Jiang" ], "categories": [ "cs.IR", "cs.SE" ], "topics": [ "agent-evaluation", "agent-safety", "coding-agent", "computer-use", "memory", "workflow-agent" ], "score": 15, "relevance": "high", "matched_queries": [ "coding-agent" ], "arxiv_id": "2606.27243", "source": "arxiv", "source_id": "arxiv:2606.27243", "pdf_url": "https://arxiv.org/pdf/2606.27243", "primary_query": "coding-agent" }, { "id": "2606.26924", "title": "A Deterministic Control Plane for LLM Coding Agents", "url": "https://arxiv.org/abs/2606.26924", "published": "2026-06-25", "updated": "2026-06-25", "authors": [ "Padmaraj Madatha" ], "categories": [ "cs.SE", "cs.AI", "cs.CR" ], "topics": [ "agent-evaluation", "coding-agent", "planning", "tool-use", "workflow-agent" ], "score": 15, "relevance": "high", "matched_queries": [ "coding-agent" ], "arxiv_id": "2606.26924", "source": "arxiv", "source_id": "arxiv:2606.26924", "pdf_url": "https://arxiv.org/pdf/2606.26924", "primary_query": "coding-agent" }, { "id": "2606.26883", "title": "EconSimulacra: A Digital Twin Platform of Socio-Economic Systems Powered by LLM Agents", "url": "https://arxiv.org/abs/2606.26883", "published": "2026-06-25", "updated": "2026-06-30", "authors": [ "Ryuji Hashimoto", "Masahiro Kaneko", "Kentaro Ueda", "Takehiro Takayanagi", "Kiyoshi Izumi" ], "categories": [ "cs.DL" ], "topics": [ "memory", "multi-agent", "tool-use" ], "score": 15, "relevance": "high", "matched_queries": [ "multi-agent-llm" ], "arxiv_id": "2606.26883", "source": "arxiv", "source_id": "arxiv:2606.26883", "pdf_url": "https://arxiv.org/pdf/2606.26883", "primary_query": "multi-agent-llm" }, { "id": "2606.28409", "title": "Evidence-Driven LLM Agent for C-to-Synthesizable-C Conversion and Verification", "url": "https://arxiv.org/abs/2606.28409", "published": "2026-06-25", "updated": "2026-06-25", "authors": [ "Zhe Zhao", "Hongbing Lang", "Zhihan Xiao", "Luke Ztz Hu", "John Imoleayo Adebisi", "Songping Mai" ], "categories": [ "cs.AR", "cs.AI" ], "topics": [ "rag", "reasoning", "tool-use", "workflow-agent", "world-model" ], "score": 15, "relevance": "high", "matched_queries": [ "rag-agent" ], "arxiv_id": "2606.28409", "source": "arxiv", "source_id": "arxiv:2606.28409", "pdf_url": "https://arxiv.org/pdf/2606.28409", "primary_query": "rag-agent" }, { "id": "2606.25819", "title": "Beyond Function Calling: Benchmarking Tool-Using Agents under Tool-Environment Unreliability", "url": "https://arxiv.org/abs/2606.25819", "published": "2026-06-24", "updated": "2026-06-27", "authors": [ "Yang Tian", "Zhengpeng Shi", "Yu Zhou", "Bo Zhao" ], "categories": [ "cs.CL", "cs.SE" ], "topics": [ "agent-evaluation", "tool-use", "workflow-agent" ], "score": 15, "relevance": "high", "matched_queries": [ "function-calling", "tool-use" ], "arxiv_id": "2606.25819", "source": "arxiv", "source_id": "arxiv:2606.25819", "pdf_url": "https://arxiv.org/pdf/2606.25819", "primary_query": "function-calling" }, { "id": "2606.26453", "title": "Optimizing CUDA like a Human: Micro-Profiling Tools as Expert Surrogates for LLM-Based GPU Kernel Optimization", "url": "https://arxiv.org/abs/2606.26453", "published": "2026-06-24", "updated": "2026-06-24", "authors": [ "Jiading Gai", "Shuai Zhang", "Kaj Bostrom", "Jin Huang", "Vihang Patil", "Haoyang Fang", "Bernie Wang", "Huzefa Rangwala", "George Karypis" ], "categories": [ "cs.LG" ], "topics": [ "agent-evaluation", "coding-agent", "computer-use", "memory", "multi-agent", "tool-use" ], "score": 15, "relevance": "high", "matched_queries": [ "coding-agent", "multi-agent-llm" ], "arxiv_id": "2606.26453", "source": "arxiv", "source_id": "arxiv:2606.26453", "pdf_url": "https://arxiv.org/pdf/2606.26453", "primary_query": "coding-agent" }, { "id": "2606.25361", "title": "Memory Makes the Difference: Evaluating How Different Memory Roles Shape Conversational Agents", "url": "https://arxiv.org/abs/2606.25361", "published": "2026-06-24", "updated": "2026-06-24", "authors": [ "Yuxin Wang", "Paul Thomas", "Zhiwei Yu", "Yuan Gao", "Saeed Hassanpour", "Soroush Vosoughi", "Robert Sim", "Nick Craswell" ], "categories": [ "cs.CL", "cs.AI", "cs.IR" ], "topics": [ "agent-evaluation", "memory", "rag", "tool-use" ], "score": 15, "relevance": "high", "matched_queries": [ "rag-agent" ], "arxiv_id": "2606.25361", "source": "arxiv", "source_id": "arxiv:2606.25361", "pdf_url": "https://arxiv.org/pdf/2606.25361", "primary_query": "rag-agent" }, { "id": "2606.27397", "title": "SidConArena: An Environment Evaluating Agents in Open-Ended,Positive-Sum Bargaining Game", "url": "https://arxiv.org/abs/2606.27397", "published": "2026-06-24", "updated": "2026-06-24", "authors": [ "Yeqi Feng", "Yuxin Chen", "Tianxing He" ], "categories": [ "cs.MA", "cs.AI", "cs.GT" ], "topics": [ "agent-evaluation", "memory", "planning", "reasoning", "tool-use" ], "score": 15, "relevance": "high", "matched_queries": [ "planning-agent" ], "arxiv_id": "2606.27397", "source": "arxiv", "source_id": "arxiv:2606.27397", "pdf_url": "https://arxiv.org/pdf/2606.27397", "primary_query": "planning-agent" }, { "id": "2606.25139", "title": "Buildrix: An Open Platform for Sharing and Benchmarking Agentic AI Skills in Building Engineering", "url": "https://arxiv.org/abs/2606.25139", "published": "2026-06-23", "updated": "2026-06-23", "authors": [ "Zixin Jiang", "Bing Dong" ], "categories": [ "eess.SY" ], "topics": [ "agent-evaluation", "tool-use", "workflow-agent" ], "score": 15, "relevance": "high", "matched_queries": [ "agentic-ai" ], "arxiv_id": "2606.25139", "source": "arxiv", "source_id": "arxiv:2606.25139", "pdf_url": "https://arxiv.org/pdf/2606.25139", "primary_query": "agentic-ai" }, { "id": "2606.24779", "title": "DeepBD: A Grounded Agentic Workflow for Variant Prioritization and Diagnosis of Genetic Birth Defects", "url": "https://arxiv.org/abs/2606.24779", "published": "2026-06-23", "updated": "2026-06-23", "authors": [ "Shiyu Li", "Ziqi Yan", "Zhihao Wu", "Jielong Lu", "Weiran Liao", "Jiajun Yu", "Genjie Li", "Zeyu Chu", "Jiajun Bu", "Haishuai Wang" ], "categories": [ "q-bio.GN", "cs.AI" ], "topics": [ "agent-evaluation", "memory", "tool-use", "workflow-agent" ], "score": 15, "relevance": "high", "matched_queries": [ "agentic-ai" ], "arxiv_id": "2606.24779", "source": "arxiv", "source_id": "arxiv:2606.24779", "pdf_url": "https://arxiv.org/pdf/2606.24779", "primary_query": "agentic-ai" }, { "id": "2606.24193", "title": "SkyChain Intelligence: A Blockchain-Secured Multi-Agent DRL Framework for Low-Altitude Embodied Artificial Intelligence", "url": "https://arxiv.org/abs/2606.24193", "published": "2026-06-23", "updated": "2026-06-23", "authors": [ "Haoxiang Luo", "Tianqi Jiang", "Ruichen Zhang", "Yinqiu Liu", "Gang Sun", "Hongfang Yu", "Abbas Jamalipour", "Dong In Kim" ], "categories": [ "cs.NI", "cs.DC" ], "topics": [ "agent-safety", "embodied-agent", "multi-agent", "tool-use", "world-model" ], "score": 15, "relevance": "high", "matched_queries": [ "agentic-ai" ], "arxiv_id": "2606.24193", "source": "arxiv", "source_id": "arxiv:2606.24193", "pdf_url": "https://arxiv.org/pdf/2606.24193", "primary_query": "agentic-ai" }, { "id": "2606.24597", "title": "Qwen-AgentWorld: Language World Models for General Agents", "url": "https://arxiv.org/abs/2606.24597", "published": "2026-06-23", "updated": "2026-06-23", "authors": [ "Yuxin Zuo", "Zikai Xiao", "Li Sheng", "Fei Huang", "Jianhong Tu", "Yuxuan Liu", "Tianyi Tang", "Xiaomeng Hu", "Yang Su", "Qingfeng Lan", "Yantao Liu", "Qin Zhu", "Yinger Zhang", "Bowen Yu", "Haiquan Zhao", "Haiyang Xu", "Jianxin Yang", "Jiayang Cheng", "Junyang Wang", "Lianghao Deng", "Mingfeng Xue", "Tianyi Bai", "Yang Fan", "Yubo Ma", "Yucheng Li", "Zeyu Cui", "Zhihai Wang", "Zhihui Xie", "Zhuorui Ye", "An Yang", "Dayiheng Liu", "Jingren Zhou", "Ning Ding" ], "categories": [ "cs.CL" ], "topics": [ "agent-evaluation", "planning", "rag", "reasoning", "tool-use", "world-model" ], "score": 15, "relevance": "high", "matched_queries": [ "agent-evaluation" ], "arxiv_id": "2606.24597", "source": "arxiv", "source_id": "arxiv:2606.24597", "pdf_url": "https://arxiv.org/pdf/2606.24597", "primary_query": "agent-evaluation" }, { "id": "2606.25115", "title": "Forget to Improve: On-Device LLM-Agent Continual Learning via Budget-Curated Memory", "url": "https://arxiv.org/abs/2606.25115", "published": "2026-06-23", "updated": "2026-06-23", "authors": [ "Beining Wu", "Zihao Ding", "Jun Huang", "Yanxiao Zhao" ], "categories": [ "cs.LG", "cs.NI" ], "topics": [ "agent-evaluation", "embodied-agent", "memory" ], "score": 15, "relevance": "high", "matched_queries": [ "agent-memory" ], "arxiv_id": "2606.25115", "source": "arxiv", "source_id": "arxiv:2606.25115", "pdf_url": "https://arxiv.org/pdf/2606.25115", "primary_query": "agent-memory" }, { "id": "2606.24322", "title": "Securing LLM-Agent Long-Term Memory Against Poisoning: Non-Malleable, Origin-Bound Authority with Machine-Checked Guarantees", "url": "https://arxiv.org/abs/2606.24322", "published": "2026-06-23", "updated": "2026-06-23", "authors": [ "Yedidel Louck" ], "categories": [ "cs.CR" ], "topics": [ "agent-evaluation", "memory", "tool-use" ], "score": 15, "relevance": "high", "matched_queries": [ "agent-memory" ], "arxiv_id": "2606.24322", "source": "arxiv", "source_id": "arxiv:2606.24322", "pdf_url": "https://arxiv.org/pdf/2606.24322", "primary_query": "agent-memory" }, { "id": "2606.24839", "title": "Grading the Grader: Lessons from Evaluating an Agentic Data Analysis System", "url": "https://arxiv.org/abs/2606.24839", "published": "2026-06-23", "updated": "2026-06-23", "authors": [ "Tian Zheng", "Kai-Tai Hsu" ], "categories": [ "cs.AI", "stat.AP" ], "topics": [ "agent-evaluation", "computer-use", "multi-agent", "tool-use" ], "score": 15, "relevance": "high", "matched_queries": [ "multi-agent-llm" ], "arxiv_id": "2606.24839", "source": "arxiv", "source_id": "arxiv:2606.24839", "pdf_url": "https://arxiv.org/pdf/2606.24839", "primary_query": "multi-agent-llm" }, { "id": "2606.23927", "title": "RIFT-Bench: Dynamic Red-teaming For Agentic AI Systems", "url": "https://arxiv.org/abs/2606.23927", "published": "2026-06-22", "updated": "2026-06-22", "authors": [ "Yarin Yerushalmi Levi", "Roy Betser", "Amit Giloni", "Lidor Erez", "Itay Gershon", "Oren Rachmil", "Sindhu Padakandla", "Roman Vainshtein" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "agent-safety", "rag", "tool-use" ], "score": 15, "relevance": "high", "matched_queries": [ "agentic-ai" ], "arxiv_id": "2606.23927", "source": "arxiv", "source_id": "arxiv:2606.23927", "pdf_url": "https://arxiv.org/pdf/2606.23927", "primary_query": "agentic-ai" }, { "id": "2606.24551", "title": "GUI vs. CLI: Execution Bottlenecks in Screen-Only and Skill-Mediated Computer-Use Agents", "url": "https://arxiv.org/abs/2606.24551", "published": "2026-06-22", "updated": "2026-06-22", "authors": [ "Xiao Zhou", "Siyue Zhang", "Yilun Zhao", "Jinbiao Wei", "Tingyu Song", "Arman Cohan", "Chen Zhao" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "computer-use", "planning", "rag", "reasoning", "tool-use", "workflow-agent" ], "score": 15, "relevance": "high", "matched_queries": [ "web-gui-agent" ], "arxiv_id": "2606.24551", "source": "arxiv", "source_id": "arxiv:2606.24551", "pdf_url": "https://arxiv.org/pdf/2606.24551", "primary_query": "web-gui-agent" }, { "id": "2606.22948", "title": "ENVS: Environment-Native Verified Search for Long-Horizon GUI Agents", "url": "https://arxiv.org/abs/2606.22948", "published": "2026-06-22", "updated": "2026-06-22", "authors": [ "Yincheng Zhou", "Athena Zhuoming Zhong", "Shijie Zhang", "Kevin Zhang", "Teresa Xiaotao Shang", "Shanghang Zhang" ], "categories": [ "cs.AI", "cs.CV" ], "topics": [ "agent-evaluation", "computer-use", "planning", "reasoning", "tool-use" ], "score": 15, "relevance": "high", "matched_queries": [ "web-gui-agent" ], "arxiv_id": "2606.22948", "source": "arxiv", "source_id": "arxiv:2606.22948", "pdf_url": "https://arxiv.org/pdf/2606.22948", "primary_query": "web-gui-agent" }, { "id": "2606.23764", "title": "Emergent Relational Order in LLM Agent Societies: From Collective Affect to Authority Stratification", "url": "https://arxiv.org/abs/2606.23764", "published": "2026-06-22", "updated": "2026-06-22", "authors": [ "Zhiyuan Ji", "Xinyu Chen", "Ziqi Dai", "Shiyun Tang", "Chunyu Wei", "Yueguo Chen" ], "categories": [ "cs.MA", "cs.AI" ], "topics": [ "multi-agent", "planning", "tool-use", "world-model" ], "score": 15, "relevance": "high", "matched_queries": [ "multi-agent-llm" ], "arxiv_id": "2606.23764", "source": "arxiv", "source_id": "arxiv:2606.23764", "pdf_url": "https://arxiv.org/pdf/2606.23764", "primary_query": "multi-agent-llm" }, { "id": "2606.23343", "title": "Group Selection Promotes Prosocial Prompts in Populations of LLM Agents", "url": "https://arxiv.org/abs/2606.23343", "published": "2026-06-22", "updated": "2026-06-22", "authors": [ "Luis Celiktemel", "Edward Eichhorn", "Levin Brinkmann", "Robin Schimmelpfennig", "Aron Vallinder", "Yaomin Jiang", "Edward Hughes", "Iyad Rahwan" ], "categories": [ "cs.CY" ], "topics": [ "coding-agent", "computer-use", "multi-agent", "world-model" ], "score": 15, "relevance": "high", "matched_queries": [ "multi-agent-llm" ], "arxiv_id": "2606.23343", "source": "arxiv", "source_id": "arxiv:2606.23343", "pdf_url": "https://arxiv.org/pdf/2606.23343", "primary_query": "multi-agent-llm" }, { "id": "2606.22388", "title": "PlanBench-XL: Evaluating Long-Horizon Planning of LLM Tool-Use Agents in Large-Scale Tool Ecosystems", "url": "https://arxiv.org/abs/2606.22388", "published": "2026-06-21", "updated": "2026-06-21", "authors": [ "Jiayu Liu", "Qihan Lin", "Cheng Qian", "Rui Wang", "Emre Can Acikgoz", "Xiaocheng Yang", "Jiateng Liu", "Zhenhailong Wang", "Xiusi Chen", "Heng Ji", "Dilek Hakkani-Tür" ], "categories": [ "cs.AI", "cs.CL" ], "topics": [ "agent-evaluation", "planning", "rag", "tool-use" ], "score": 15, "relevance": "high", "matched_queries": [ "planning-agent", "tool-use" ], "arxiv_id": "2606.22388", "source": "arxiv", "source_id": "arxiv:2606.22388", "pdf_url": "https://arxiv.org/pdf/2606.22388", "primary_query": "planning-agent" }, { "id": "2606.22610", "title": "PaperClaw: Harnessing Agents for Autonomous Research and Human-in-the-Loop Refinement", "url": "https://arxiv.org/abs/2606.22610", "published": "2026-06-21", "updated": "2026-06-21", "authors": [ "Weiwei Ye", "Hangchen Liu", "Dongyuan Li", "Renhe Jiang" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "memory", "multi-agent", "tool-use" ], "score": 15, "relevance": "high", "matched_queries": [ "multi-agent-llm" ], "arxiv_id": "2606.22610", "source": "arxiv", "source_id": "arxiv:2606.22610", "pdf_url": "https://arxiv.org/pdf/2606.22610", "primary_query": "multi-agent-llm" }, { "id": "2606.21565", "title": "Composing Verifiable Conceptual Models via Building Blocks: Towards Design-Time Verification of Agentic AI Workflows", "url": "https://arxiv.org/abs/2606.21565", "published": "2026-06-19", "updated": "2026-06-19", "authors": [ "Noe Y. Flandre", "Alexander C. Nwala", "Philippe J. Giabbanelli" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "reasoning", "tool-use", "workflow-agent", "world-model" ], "score": 15, "relevance": "high", "matched_queries": [ "agentic-ai" ], "arxiv_id": "2606.21565", "source": "arxiv", "source_id": "arxiv:2606.21565", "pdf_url": "https://arxiv.org/pdf/2606.21565", "primary_query": "agentic-ai" }, { "id": "2606.21732", "title": "Safe to Check, Unsafe to Use: Relinking at the Compression Boundary of LLM Agents", "url": "https://arxiv.org/abs/2606.21732", "published": "2026-06-19", "updated": "2026-06-19", "authors": [ "Zesen Liu", "Zihan Zhang", "Dongdong She" ], "categories": [ "cs.CR", "cs.AI" ], "topics": [ "agent-evaluation", "agent-safety", "rag", "tool-use" ], "score": 15, "relevance": "high", "matched_queries": [ "agent-evaluation" ], "arxiv_id": "2606.21732", "source": "arxiv", "source_id": "arxiv:2606.21732", "pdf_url": "https://arxiv.org/pdf/2606.21732", "primary_query": "agent-evaluation" }, { "id": "2606.20479", "title": "GroundControl: Anticipating Navigation Failures in Vision-Language Agents via Trajectory-Consistent Uncertainty Estimates", "url": "https://arxiv.org/abs/2606.20479", "published": "2026-06-18", "updated": "2026-06-18", "authors": [ "Nastaran Darabi", "Divake Kumar", "Sina Tayebati", "Devashri Naik", "Amit Ranjan Trivedi" ], "categories": [ "cs.RO" ], "topics": [ "agent-evaluation", "agent-safety", "embodied-agent", "rag", "tool-use" ], "score": 15, "relevance": "high", "matched_queries": [ "language-agent" ], "arxiv_id": "2606.20479", "source": "arxiv", "source_id": "arxiv:2606.20479", "pdf_url": "https://arxiv.org/pdf/2606.20479", "primary_query": "language-agent" }, { "id": "2606.20041", "title": "AI Economist Agent: An Agentic Framework for Model-Grounded Economic Analysis with RAG, Knowledge Graphs, and Large Language Models", "url": "https://arxiv.org/abs/2606.20041", "published": "2026-06-18", "updated": "2026-06-18", "authors": [ "Masahiro Kato" ], "categories": [ "econ.GN", "cs.AI", "cs.LG", "q-fin.GN" ], "topics": [ "agent-evaluation", "planning", "rag" ], "score": 15, "relevance": "high", "matched_queries": [ "ai-agent", "rag-agent" ], "arxiv_id": "2606.20041", "source": "arxiv", "source_id": "arxiv:2606.20041", "pdf_url": "https://arxiv.org/pdf/2606.20041", "primary_query": "ai-agent" }, { "id": "2606.19852", "title": "Prompt, Plan, Extract: Zero-Shot Agentic LLMs Workflows for Lung Pathology Extraction from Clinical Narratives", "url": "https://arxiv.org/abs/2606.19852", "published": "2026-06-18", "updated": "2026-06-25", "authors": [ "Aman Pathak", "Cheng Peng", "Mengxian Lyu", "Ziyi Chen", "Reema Solan", "Sankalp Talankar", "Yasir Khan", "Hiren Mehta", "Aokun Chen", "Yi Guo", "Yonghui Wu" ], "categories": [ "cs.CL", "cs.LG" ], "topics": [ "agent-evaluation", "planning", "tool-use", "workflow-agent" ], "score": 15, "relevance": "high", "matched_queries": [ "agentic-ai" ], "arxiv_id": "2606.19852", "source": "arxiv", "source_id": "arxiv:2606.19852", "pdf_url": "https://arxiv.org/pdf/2606.19852", "primary_query": "agentic-ai" }, { "id": "2606.20047", "title": "PACMS: Submodular Context Selection as a Pluggable Engine for LLM Agents", "url": "https://arxiv.org/abs/2606.20047", "published": "2026-06-18", "updated": "2026-06-18", "authors": [ "Manu Ghulyani", "Arunabh Singh", "Karan Bharadwaj", "Ankit Nath", "Suranjan Goswami" ], "categories": [ "cs.IR" ], "topics": [ "agent-evaluation", "memory", "rag", "tool-use" ], "score": 15, "relevance": "high", "matched_queries": [ "tool-use" ], "arxiv_id": "2606.20047", "source": "arxiv", "source_id": "arxiv:2606.20047", "pdf_url": "https://arxiv.org/pdf/2606.20047", "primary_query": "tool-use" }, { "id": "2606.19926", "title": "MemGUI-Agent: An End-to-End Long-Horizon Mobile GUI Agent with Proactive Context Management", "url": "https://arxiv.org/abs/2606.19926", "published": "2026-06-18", "updated": "2026-06-18", "authors": [ "Guangyi Liu", "Gao Wu", "Congxiao Liu", "Pengxiang Zhao", "Liang Liu", "Mading Li", "Qi Zhang", "Mengyan Wang", "Liang Guo", "Yong Liu" ], "categories": [ "cs.HC" ], "topics": [ "agent-evaluation", "computer-use", "planning", "tool-use" ], "score": 15, "relevance": "high", "matched_queries": [ "web-gui-agent" ], "arxiv_id": "2606.19926", "source": "arxiv", "source_id": "arxiv:2606.19926", "pdf_url": "https://arxiv.org/pdf/2606.19926", "primary_query": "web-gui-agent" }, { "id": "2606.20922", "title": "Think Twice Before You Act: Protecting LLM Agents Against Tool Description Poisoning via Isolated Planning", "url": "https://arxiv.org/abs/2606.20922", "published": "2026-06-18", "updated": "2026-06-18", "authors": [ "Shanghao Shi", "Xiao Wang", "Chaoyu Zhang", "Hao Li", "Wenjing Lou", "Thomas Hou", "Yevgeniy Vorobeychik", "Chongjie Zhang", "Ning Zhang" ], "categories": [ "cs.CR" ], "topics": [ "agent-evaluation", "agent-safety", "planning", "tool-use" ], "score": 15, "relevance": "high", "matched_queries": [ "planning-agent" ], "arxiv_id": "2606.20922", "source": "arxiv", "source_id": "arxiv:2606.20922", "pdf_url": "https://arxiv.org/pdf/2606.20922", "primary_query": "planning-agent" }, { "id": "2606.19787", "title": "ORAgentBench: Can LLM Agents Solve Challenging Operations Research Tasks End to End?", "url": "https://arxiv.org/abs/2606.19787", "published": "2026-06-18", "updated": "2026-06-18", "authors": [ "Jiajun Li", "Mingshu Cai", "Yixuan Li", "Yu Ding", "Ran Hou", "Guanyu Nie", "Xiongwei Han", "Wanyuan Wang" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "rag", "workflow-agent" ], "score": 15, "relevance": "high", "matched_queries": [ "autonomous-agent-llm" ], "arxiv_id": "2606.19787", "source": "arxiv", "source_id": "arxiv:2606.19787", "pdf_url": "https://arxiv.org/pdf/2606.19787", "primary_query": "autonomous-agent-llm" }, { "id": "2606.18068", "title": "Agentic AI-based Framework for Mitigating Premature Diagnostic Handoff and Silent Hallucination in Healthcare Applications", "url": "https://arxiv.org/abs/2606.18068", "published": "2026-06-16", "updated": "2026-06-16", "authors": [ "Divyansh Srivastava", "Shreya Ghosh", "Anshul Verma", "Rajkumar Buyya" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "agent-safety", "multi-agent", "reasoning" ], "score": 15, "relevance": "high", "matched_queries": [ "agentic-ai" ], "arxiv_id": "2606.18068", "source": "arxiv", "source_id": "arxiv:2606.18068", "pdf_url": "https://arxiv.org/pdf/2606.18068", "primary_query": "agentic-ai" }, { "id": "2606.17680", "title": "EnvRL: Learn from Environment Dynamics in Agentic Reinforcement Learning", "url": "https://arxiv.org/abs/2606.17680", "published": "2026-06-16", "updated": "2026-06-16", "authors": [ "Zhitong Wang", "Songze Li", "Hao Peng", "Shuzheng Si", "Yi Wang", "Maosong Sun", "Juanzi Li" ], "categories": [ "cs.LG", "cs.CL" ], "topics": [ "agent-evaluation", "planning", "rag", "tool-use" ], "score": 15, "relevance": "high", "matched_queries": [ "agent-evaluation" ], "arxiv_id": "2606.17680", "source": "arxiv", "source_id": "arxiv:2606.17680", "pdf_url": "https://arxiv.org/pdf/2606.17680", "primary_query": "agent-evaluation" }, { "id": "2606.18406", "title": "CoreMem: Riemannian Retrieval and Fisher-Guided Distillation for Long-Term Memory in Dialogue Agents", "url": "https://arxiv.org/abs/2606.18406", "published": "2026-06-16", "updated": "2026-06-16", "authors": [ "Jiaqi Chen", "Yongqin Zeng", "Shaoshen Chen", "Yijian Zhang", "Hai-Tao Zheng", "Chunxia Ma", "XiuTeng Zhou" ], "categories": [ "cs.CL" ], "topics": [ "agent-evaluation", "computer-use", "memory", "rag", "reasoning", "tool-use" ], "score": 15, "relevance": "high", "matched_queries": [ "agent-memory" ], "arxiv_id": "2606.18406", "source": "arxiv", "source_id": "arxiv:2606.18406", "pdf_url": "https://arxiv.org/pdf/2606.18406", "primary_query": "agent-memory" }, { "id": "2606.17449", "title": "MODE-RAG: Manifold Outlier Diagnosis and Energy-based Retrieval-Augmented Generation Evaluation", "url": "https://arxiv.org/abs/2606.17449", "published": "2026-06-16", "updated": "2026-06-16", "authors": [ "Zehang Wei", "Jiaxin Dai", "Jiamin Yan", "Xiang Xiang" ], "categories": [ "cs.CL", "cs.AI", "cs.CV", "cs.LG", "cs.MM" ], "topics": [ "agent-evaluation", "agent-safety", "computer-use", "multi-agent", "rag", "reasoning" ], "score": 15, "relevance": "high", "matched_queries": [ "rag-agent" ], "arxiv_id": "2606.17449", "source": "arxiv", "source_id": "arxiv:2606.17449", "pdf_url": "https://arxiv.org/pdf/2606.17449", "primary_query": "rag-agent" }, { "id": "2606.16432", "title": "ACCORD: Action-Conditioned Contextual Grounding for Language Agents", "url": "https://arxiv.org/abs/2606.16432", "published": "2026-06-15", "updated": "2026-06-15", "authors": [ "Lai Jiang", "Cheng Qian", "Zhenhailong Wang", "Pan Lu", "Heng Ji", "Hao Peng" ], "categories": [ "cs.CL", "cs.AI" ], "topics": [ "agent-evaluation", "embodied-agent", "rag", "tool-use" ], "score": 15, "relevance": "high", "matched_queries": [ "language-agent" ], "arxiv_id": "2606.16432", "source": "arxiv", "source_id": "arxiv:2606.16432", "pdf_url": "https://arxiv.org/pdf/2606.16432", "primary_query": "language-agent" }, { "id": "2606.15903", "title": "Control-Plane Placement Shapes Forgetting: An Architectural Study of Agent Memory Across Thirteen System Configurations", "url": "https://arxiv.org/abs/2606.15903", "published": "2026-06-14", "updated": "2026-06-16", "authors": [ "Dongxu Yang" ], "categories": [ "cs.CL", "cs.AI" ], "topics": [ "agent-evaluation", "memory", "planning", "rag" ], "score": 15, "relevance": "high", "matched_queries": [ "agent-memory" ], "arxiv_id": "2606.15903", "source": "arxiv", "source_id": "arxiv:2606.15903", "pdf_url": "https://arxiv.org/pdf/2606.15903", "primary_query": "agent-memory" }, { "id": "2606.15609", "title": "FragFuse: Bypassing Access Control of Large Language Model Agents via Memory-Based Query Fragmentation and Fusion", "url": "https://arxiv.org/abs/2606.15609", "published": "2026-06-14", "updated": "2026-06-14", "authors": [ "Zixin Rao", "Wentian Zhu", "Chan Aristella Lu", "Zhaorun Chen", "Wei Niu", "Le Guan", "Bo Li", "Zhen Xiang" ], "categories": [ "cs.CR", "cs.AI" ], "topics": [ "agent-evaluation", "memory", "rag", "tool-use" ], "score": 15, "relevance": "high", "matched_queries": [ "agent-memory" ], "arxiv_id": "2606.15609", "source": "arxiv", "source_id": "arxiv:2606.15609", "pdf_url": "https://arxiv.org/pdf/2606.15609", "primary_query": "agent-memory" }, { "id": "2606.15079", "title": "Ling and Ring 2.6 Technical Report: Efficient and Instant Agentic Intelligence at Trillion-Parameter Scale", "url": "https://arxiv.org/abs/2606.15079", "published": "2026-06-13", "updated": "2026-06-13", "authors": [ "Ang Li", "Ben Liu", "Bin Han", "Bin Hu", "Bin Jing", "Binbin Hu", "Bing Li", "Cai Chen", "Caizhi Tang", "Changxin Tian", "Chao Huang", "Chao Zhang", "Chen Liang", "Chen Qian", "Chengfu Tang", "Chengyao Wen", "Chilin Fu", "Chunwei Wu", "Cong Zhang", "Cunyin Peng", "Daixin Wang", "Dalong Zhang", "Deng Zhao", "Dingnan Jin", "Dingyuan Zhu", "Donghao Zhang", "Fan Yuan", "Fangzheng Zhao", "Fanzhuang Meng", "Feifan Wu", "Feng Xu", "Fengbin Fang", "Gangshan Wang", "Guodong Yang", "Hailin Zhao", "Haitao Wang", "Haitao Zhang", "Hanxiao Zhang", "Hanzi Wang", "Hao Dai", "Hao Liu", "Hao Qian", "Hao Wu", "Haoxiong Liu", "Haoyu Xu", "Heng Zhang", "Hong Liu", "Hongliang Zhang", "Hongrui Liu", "Hongxun Li", "Hongzhi Ruan", "Huaidong Xiong", "Huihuang Zheng", "Huikang Tang", "Jia Guo", "Jia Li", "Jia Liu", "Jiameng Wang", "Jiaming Liu", "Jiannan Shi", "Jianping Wei", "Jiaolong Yang", "Jiapeng Wang", "Jie Gao", "Jie Wang", "Jiewei Wu", "Jin Yang", "Jinjin Li", "Jinjing Huang", "Jinquan Sun", "Jinyao Chen", "Juanhui Tu", "Jun Liu", "Jun Mei", "Jun Xu", "Jun Zhou", "Junjie Ou", "Junnan Sipan", "Junpeng Fang", "Kaihong Zhang", "Kaiqin Hu", "Ke Shi", "Kuan Xu", "Kun Tang", "Kunlong Chen", "Lanyin Mei", "Lei Chen", "Lei Liang", "Lei Xu", "Li Tang", "Liang Jiang", "Liangcheng Fu", "Lihui Zhang", "Linfeng Shi", "Lintao Ma", "Liyuan Liu", "Longfei Li", "Longfei Zheng", "Lu Liu", "Lu Yu", "Man Li", "Meiqi Zhu", "Meng Li", "Mengjie Gao", "Mengshu Sun", "Mingming Yin", "Mingyang Zhang", "Mingyuan Fan", "Nuo Xu", "Pan Tang", "Peijie Jiang", "Peilong Zhao", "Peng Lin", "Pingping Liu", "Qi Zuo", "Qian Zhao", "Qiang Cheng", "Qianggang Cao", "Qiaoben Bao", "Qing Cui", "Qingyuan Yang", "Qitao Shi", "Qiyin Huang", "Qizheng Zhou", "Quan Wan", "Runyuan Zhao", "Shaomian Zheng", "Shaowei Wei", "Shengnan Zhang", "Shuaicheng Li", "Shujie Li", "Shuo Zhang", "Sikang Bian", "Tianchu Yao", "Tiange Xu", "Tianshu Wang", "Ting Guo", "Tinghao Wang", "Tingwei Huang", "Tong Zhao", "Tongkai Yang", "Wang Hong", "Wanli Gu", "Wei Lu", "Weichang Wu", "Weiguang Han", "Weiquan Li", "Wenbo Shen", "Wenjing Fang", "Wenzhi Tang", "Xiang Shu", "Xiao Shi", "Xiaodong Yan", "Xiaolu Zhang", "Xiaopei Wan", "Xiaqing Sun", "Xin Zhao", "Xingyu Lu", "Xinxing Yang", "Xinyao Tang", "Xinyu Kong", "Xinyu Liu", "Xiong Xu", "Xuan Sun", "Xudong Han", "Xudong Wang", "Xujie Shen", "Yalin Zhang", "Yangyang Hou", "Yankun Ren", "Yao Zhao", "Ye Chen", "Yeyang Chen", "Yibo Cao", "Yifan Zuo", "Yijie Chen", "Ying Li", "Yingjie Song", "Yingxue Li", "Yiqi Wang", "Yixuan Sun", "Yizhu Xiao", "Yongfei Xu", "Yu Liu", "Yuchen Fang", "Yue Gao", "Yue Yu", "Yue Zhang", "Yuqi Zhang", "Yuxiao He", "Yuxiao Lu", "Yuxin Tian", "Yuxuan Li", "Yuzhuo Fu", "Zhankai Xu", "Zhaoxin Huan", "Zhenduo Zhang", "Zhengke Gui", "Zhengyu Huang", "Zhenjun Ma", "Zhenxuan Pan", "Zheping Qu", "Zhibo Zhu", "Zhidong Fan", "Zhigang Huangfu", "Zhihao Wang", "Zhiqiang Zhang", "Zhizhen Liu", "Zhuyan Zhou", "Zibin Lin", "Zihang Zeng", "Zihao Wang", "Zilong Wang", "Ziqi Liu", "Zitao Xuan", "Zixuan Cheng", "Zujie Wen", "Zuoli Tang" ], "categories": [ "cs.CL", "cs.AI" ], "topics": [ "agent-safety", "coding-agent", "computer-use", "reasoning", "tool-use", "workflow-agent" ], "score": 15, "relevance": "high", "matched_queries": [ "tool-use" ], "arxiv_id": "2606.15079", "source": "arxiv", "source_id": "arxiv:2606.15079", "pdf_url": "https://arxiv.org/pdf/2606.15079", "primary_query": "tool-use" }, { "id": "2606.14571", "title": "StreamMemBench: Streaming Evaluation of Agent Memory for Future-Oriented Assistance", "url": "https://arxiv.org/abs/2606.14571", "published": "2026-06-12", "updated": "2026-06-12", "authors": [ "Guanming Liu", "Yuqi Ren", "Hansu Gu", "Peng Zhang", "Weihang Wang", "Jiahao Liu", "Ning Gu", "Tun Lu" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "memory", "tool-use" ], "score": 15, "relevance": "high", "matched_queries": [ "agent-memory" ], "arxiv_id": "2606.14571", "source": "arxiv", "source_id": "arxiv:2606.14571", "pdf_url": "https://arxiv.org/pdf/2606.14571", "primary_query": "agent-memory" }, { "id": "2606.13643", "title": "Recursive Agent Harnesses", "url": "https://arxiv.org/abs/2606.13643", "published": "2026-06-11", "updated": "2026-06-11", "authors": [ "Elias Lumer", "Sahil Sen", "Kevin Paul", "Vamse Kumar Subbiah" ], "categories": [ "cs.CL" ], "topics": [ "agent-evaluation", "coding-agent", "planning", "reasoning", "tool-use", "workflow-agent" ], "score": 15, "relevance": "high", "matched_queries": [ "function-calling" ], "arxiv_id": "2606.13643", "source": "arxiv", "source_id": "arxiv:2606.13643", "pdf_url": "https://arxiv.org/pdf/2606.13643", "primary_query": "function-calling" }, { "id": "2606.13385", "title": "Who Pays the Price? Stakeholder-Centric Prompt Injection Benchmarking for Real-world Web Agents", "url": "https://arxiv.org/abs/2606.13385", "published": "2026-06-11", "updated": "2026-06-11", "authors": [ "Zihao Wang", "Yiming Li", "Yutong Wu", "Zheyu Liu", "Kangjie Chen", "Fok Kar Wai", "Pin-Yu Chen", "Vrizlynn L. L. Thing", "Bo Li", "Dacheng Tao", "Tianwei Zhang" ], "categories": [ "cs.CR", "cs.AI", "cs.CY", "cs.HC", "cs.MM" ], "topics": [ "agent-evaluation", "agent-safety", "computer-use", "tool-use" ], "score": 15, "relevance": "high", "matched_queries": [ "web-gui-agent" ], "arxiv_id": "2606.13385", "source": "arxiv", "source_id": "arxiv:2606.13385", "pdf_url": "https://arxiv.org/pdf/2606.13385", "primary_query": "web-gui-agent" }, { "id": "2606.13904", "title": "SANA: What Matters for QA Agents over Massive Data Lakes?", "url": "https://arxiv.org/abs/2606.13904", "published": "2026-06-11", "updated": "2026-06-11", "authors": [ "Austin Senna Wijaya", "Jiaxiang Liu", "Haonan Wang", "Eugene Wu" ], "categories": [ "cs.CL", "cs.AI", "cs.DB" ], "topics": [ "agent-evaluation", "computer-use", "embodied-agent", "planning", "tool-use" ], "score": 15, "relevance": "high", "matched_queries": [ "planning-agent" ], "arxiv_id": "2606.13904", "source": "arxiv", "source_id": "arxiv:2606.13904", "pdf_url": "https://arxiv.org/pdf/2606.13904", "primary_query": "planning-agent" }, { "id": "2606.12341", "title": "OCELOT: Inference-Leakage Budgets for Privacy-Preserving LLM Agents", "url": "https://arxiv.org/abs/2606.12341", "published": "2026-06-10", "updated": "2026-06-10", "authors": [ "Jin Xie", "Songze Li" ], "categories": [ "cs.CR" ], "topics": [ "agent-evaluation", "agent-safety", "reasoning", "tool-use" ], "score": 15, "relevance": "high", "matched_queries": [ "agent-evaluation" ], "arxiv_id": "2606.12341", "source": "arxiv", "source_id": "arxiv:2606.12341", "pdf_url": "https://arxiv.org/pdf/2606.12341", "primary_query": "agent-evaluation" }, { "id": "2606.12195", "title": "InternVideo3: Agentify Foundation Models with Multimodal Contextual Reasoning", "url": "https://arxiv.org/abs/2606.12195", "published": "2026-06-10", "updated": "2026-06-10", "authors": [ "Ziang Yan", "Sheng Xia", "Jiashuo Yu", "Yue Wu", "Tianxiang Jiang", "Songze Li", "Kanghui Tian", "Yicheng Xu", "Yinan He", "Kai Chen", "Limin Wang", "Yu Qiao", "Yi Wang" ], "categories": [ "cs.CV" ], "topics": [ "agent-evaluation", "memory", "planning", "rag", "reasoning", "tool-use" ], "score": 15, "relevance": "high", "matched_queries": [ "tool-use" ], "arxiv_id": "2606.12195", "source": "arxiv", "source_id": "arxiv:2606.12195", "pdf_url": "https://arxiv.org/pdf/2606.12195", "primary_query": "tool-use" }, { "id": "2606.11869", "title": "Agents All the Way Down; A Methodology for Building Custom AI Agents from Substrate to Production", "url": "https://arxiv.org/abs/2606.11869", "published": "2026-06-10", "updated": "2026-06-10", "authors": [ "Marc Alier Forment", "Juanan Pereira", "Francisco José García-Peñalvo", "María José Casañ Guerrero" ], "categories": [ "cs.SE", "cs.AI" ], "topics": [ "agent-safety", "coding-agent", "multi-agent", "tool-use" ], "score": 15, "relevance": "high", "matched_queries": [ "function-calling" ], "arxiv_id": "2606.11869", "source": "arxiv", "source_id": "arxiv:2606.11869", "pdf_url": "https://arxiv.org/pdf/2606.11869", "primary_query": "function-calling" }, { "id": "2606.17076", "title": "CMIP-Forge: An Agentic System that Retrieves, Computes, and Self-Reviews Climate Science", "url": "https://arxiv.org/abs/2606.17076", "published": "2026-06-10", "updated": "2026-06-10", "authors": [ "Dmitrii Pantiukhin", "Boris Shapkin", "Ivan Kuznetsov", "Thomas Jung", "Nikolay Koldunov" ], "categories": [ "physics.ao-ph", "cs.AI" ], "topics": [ "agent-evaluation", "agent-safety", "planning", "rag", "tool-use", "workflow-agent" ], "score": 15, "relevance": "high", "matched_queries": [ "rag-agent" ], "arxiv_id": "2606.17076", "source": "arxiv", "source_id": "arxiv:2606.17076", "pdf_url": "https://arxiv.org/pdf/2606.17076", "primary_query": "rag-agent" }, { "id": "2606.11349", "title": "Knowing When to Ask: Self-Gated Clarification for Hierarchical Language Agents", "url": "https://arxiv.org/abs/2606.11349", "published": "2026-06-09", "updated": "2026-06-12", "authors": [ "Aijing Gao", "Yiming Kang", "Mengdie Flora Wang", "Jae Oh Woo" ], "categories": [ "cs.AI", "cs.HC" ], "topics": [ "agent-evaluation", "embodied-agent", "reasoning", "tool-use" ], "score": 15, "relevance": "high", "matched_queries": [ "language-agent" ], "arxiv_id": "2606.11349", "source": "arxiv", "source_id": "arxiv:2606.11349", "pdf_url": "https://arxiv.org/pdf/2606.11349", "primary_query": "language-agent" }, { "id": "2606.11078", "title": "A History-Aware Visually Grounded Critic for Computer Use Agents", "url": "https://arxiv.org/abs/2606.11078", "published": "2026-06-09", "updated": "2026-06-09", "authors": [ "Jaewoo Lee", "Zaid Khan", "Archiki Prasad", "Justin Chih-Yao Chen", "Supriyo Chakraborty", "Kartik Balasubramaniam", "Sambit Sahu", "Elias Stengel-Eskin", "Hyunji Lee", "Mohit Bansal" ], "categories": [ "cs.AI", "cs.CL", "cs.CV" ], "topics": [ "agent-evaluation", "computer-use", "planning", "rag", "tool-use" ], "score": 15, "relevance": "high", "matched_queries": [ "web-gui-agent" ], "arxiv_id": "2606.11078", "source": "arxiv", "source_id": "arxiv:2606.11078", "pdf_url": "https://arxiv.org/pdf/2606.11078", "primary_query": "web-gui-agent" }, { "id": "2606.10423", "title": "WebChallenger: A Reliable and Efficient Generalist Web Agent", "url": "https://arxiv.org/abs/2606.10423", "published": "2026-06-09", "updated": "2026-06-09", "authors": [ "Jayoo Hwang", "Xiaowen Zhang", "Vedant Padwal" ], "categories": [ "cs.CL" ], "topics": [ "computer-use", "embodied-agent", "memory", "reasoning", "tool-use", "workflow-agent" ], "score": 15, "relevance": "high", "matched_queries": [ "web-gui-agent" ], "arxiv_id": "2606.10423", "source": "arxiv", "source_id": "arxiv:2606.10423", "pdf_url": "https://arxiv.org/pdf/2606.10423", "primary_query": "web-gui-agent" }, { "id": "2606.10381", "title": "Agentic Hybrid RAG for Evidence-Grounded Muon Collider Analysis", "url": "https://arxiv.org/abs/2606.10381", "published": "2026-06-09", "updated": "2026-06-09", "authors": [ "Ruobing Jiang", "Dawei Fu", "Cheng Jiang", "Tianyi Yang", "Zijian Wang", "Youpeng Wu", "Yong Ban", "Yajun Mao", "Qiang Li" ], "categories": [ "hep-ex", "cs.AI", "cs.CL", "cs.IR", "physics.ins-det" ], "topics": [ "agent-evaluation", "rag", "reasoning", "tool-use", "workflow-agent" ], "score": 15, "relevance": "high", "matched_queries": [ "rag-agent" ], "arxiv_id": "2606.10381", "source": "arxiv", "source_id": "arxiv:2606.10381", "pdf_url": "https://arxiv.org/pdf/2606.10381", "primary_query": "rag-agent" }, { "id": "2606.09764", "title": "iOSWorld: A Benchmark for Personally Intelligent Phone Agents", "url": "https://arxiv.org/abs/2606.09764", "published": "2026-06-08", "updated": "2026-06-08", "authors": [ "Lawrence Keunho Jang", "Mareks Woodside", "Geronimo Carom", "Andrew Keunwoo Jang", "Jing Yu Koh", "Ruslan Salakhutdinov" ], "categories": [ "cs.LG", "cs.CL" ], "topics": [ "agent-evaluation", "computer-use", "memory", "tool-use" ], "score": 15, "relevance": "high", "matched_queries": [ "agent-evaluation", "web-gui-agent" ], "arxiv_id": "2606.09764", "source": "arxiv", "source_id": "arxiv:2606.09764", "pdf_url": "https://arxiv.org/pdf/2606.09764", "primary_query": "agent-evaluation" }, { "id": "2606.09399", "title": "RunAgent SuperBrowser: A Theory of Autonomous Web Navigation Grounded in Human Browsing Behaviour", "url": "https://arxiv.org/abs/2606.09399", "published": "2026-06-08", "updated": "2026-06-08", "authors": [ "Radeen Mostafa", "Sawradip Saha" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "computer-use", "embodied-agent", "memory", "planning", "reasoning", "tool-use" ], "score": 15, "relevance": "high", "matched_queries": [ "web-gui-agent" ], "arxiv_id": "2606.09399", "source": "arxiv", "source_id": "arxiv:2606.09399", "pdf_url": "https://arxiv.org/pdf/2606.09399", "primary_query": "web-gui-agent" }, { "id": "2606.09549", "title": "SecureClaw: Clawing Back Control of LLM Agents", "url": "https://arxiv.org/abs/2606.09549", "published": "2026-06-08", "updated": "2026-06-08", "authors": [ "Yuhan Ma", "Stefan Schmid" ], "categories": [ "cs.CR", "cs.AI" ], "topics": [ "agent-evaluation", "agent-safety", "planning", "tool-use" ], "score": 15, "relevance": "high", "matched_queries": [ "agent-safety", "planning-agent" ], "arxiv_id": "2606.09549", "source": "arxiv", "source_id": "arxiv:2606.09549", "pdf_url": "https://arxiv.org/pdf/2606.09549", "primary_query": "agent-safety" }, { "id": "2606.09071", "title": "REFLECT: Intervention-Supported Error Attribution for Silent Failures in LLM Agent Traces", "url": "https://arxiv.org/abs/2606.09071", "published": "2026-06-08", "updated": "2026-06-08", "authors": [ "Xiaofeng Lin", "Yingxu Wang", "Tung Sum Thomas Kwok", "Daniel Guo", "Sahil Arun Nale", "Charles Fleming", "Guang Cheng" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "planning", "reasoning", "tool-use" ], "score": 15, "relevance": "high", "matched_queries": [ "planning-agent" ], "arxiv_id": "2606.09071", "source": "arxiv", "source_id": "arxiv:2606.09071", "pdf_url": "https://arxiv.org/pdf/2606.09071", "primary_query": "planning-agent" }, { "id": "2606.05463", "title": "PSEBench: A Controllable and Verifiable Benchmark for Evaluating LLMs in Patient Safety Event Triage", "url": "https://arxiv.org/abs/2606.05463", "published": "2026-06-03", "updated": "2026-06-09", "authors": [ "Keqi Han", "Ryan Young", "Annabel Strauss", "Lindsey Hughes", "Katharine M. Nesbitt", "Nicole Schueler", "Che Ngufor", "Carl Yang", "Yuan Xue", "Zhijun Yin" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "agent-safety", "reasoning", "tool-use", "workflow-agent" ], "score": 15, "relevance": "high", "matched_queries": [ "agent-evaluation" ], "arxiv_id": "2606.05463", "source": "arxiv", "source_id": "arxiv:2606.05463", "pdf_url": "https://arxiv.org/pdf/2606.05463", "primary_query": "agent-evaluation" }, { "id": "2606.03135", "title": "Uncertainty-Aware Clarification in LLM Agents with Information Gain", "url": "https://arxiv.org/abs/2606.03135", "published": "2026-06-02", "updated": "2026-06-02", "authors": [ "Mengyi Deng", "Zhiwei Li", "Xin Li", "Tingyu Zhu", "Ying Zhao", "Zhijiang Guo", "Wei Wang" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "computer-use", "rag", "tool-use" ], "score": 15, "relevance": "high", "matched_queries": [ "agent-evaluation" ], "arxiv_id": "2606.03135", "source": "arxiv", "source_id": "arxiv:2606.03135", "pdf_url": "https://arxiv.org/pdf/2606.03135", "primary_query": "agent-evaluation" }, { "id": "2606.04120", "title": "SaliMory: Orchestrating Cognitive Memory for Conversational Agents", "url": "https://arxiv.org/abs/2606.04120", "published": "2026-06-02", "updated": "2026-06-02", "authors": [ "Kai Zhang", "Xinyuan Zhang", "Hongda Jiang", "Shiun-Zu Kuo", "Hyokun Yun", "Ejaz Ahmed", "Shereen Oraby", "Ziyun Li", "Sanat Sharma", "Ann Lee", "Ahmed A Aly", "Anuj Kumar", "Raffay Hamid", "Xin Luna Dong" ], "categories": [ "cs.CL", "cs.AI" ], "topics": [ "agent-evaluation", "memory", "rag", "reasoning", "tool-use" ], "score": 15, "relevance": "high", "matched_queries": [ "agent-memory" ], "arxiv_id": "2606.04120", "source": "arxiv", "source_id": "arxiv:2606.04120", "pdf_url": "https://arxiv.org/pdf/2606.04120", "primary_query": "agent-memory" }, { "id": "2606.04296", "title": "The Saturation Trap and the Subjectivity of Intervention Timing: Why Affect-Based Triggers and LLM Judges Fail to Time Interventions on Autonomous Agents", "url": "https://arxiv.org/abs/2606.04296", "published": "2026-06-02", "updated": "2026-06-02", "authors": [ "Manvendra Modgil" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "agent-safety", "coding-agent", "planning", "reasoning", "tool-use" ], "score": 15, "relevance": "high", "matched_queries": [ "autonomous-agent-llm" ], "arxiv_id": "2606.04296", "source": "arxiv", "source_id": "arxiv:2606.04296", "pdf_url": "https://arxiv.org/pdf/2606.04296", "primary_query": "autonomous-agent-llm" }, { "id": "2606.03108", "title": "EvoTrainer: Co-Evolving LLM Policies and Training Harnesses for Autonomous Agentic Reinforcement Learning", "url": "https://arxiv.org/abs/2606.03108", "published": "2026-06-02", "updated": "2026-06-12", "authors": [ "Guhong Chen", "Yingcheng Shi", "Yongbin Li", "Binhua Li", "Xander Xu", "Hu Wei", "Shiwen Ni", "Min Yang", "Jieping Ye" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "coding-agent", "planning", "reasoning" ], "score": 15, "relevance": "high", "matched_queries": [ "autonomous-agent-llm" ], "arxiv_id": "2606.03108", "source": "arxiv", "source_id": "arxiv:2606.03108", "pdf_url": "https://arxiv.org/pdf/2606.03108", "primary_query": "autonomous-agent-llm" }, { "id": "2606.02388", "title": "Policy and World Modeling Co-Training for Language Agents", "url": "https://arxiv.org/abs/2606.02388", "published": "2026-06-01", "updated": "2026-06-01", "authors": [ "Ning Lu", "Baijiong Lin", "Shengcai Liu", "Jiahao Wu", "Haoze Lv", "Yanbin Wei", "Lingting Zhu", "Shengju Qian", "Xin Wang", "Ying-Cong Chen", "Qi Wang", "Ke Tang" ], "categories": [ "cs.LG", "cs.AI" ], "topics": [ "agent-evaluation", "tool-use", "world-model" ], "score": 15, "relevance": "high", "matched_queries": [ "language-agent" ], "arxiv_id": "2606.02388", "source": "arxiv", "source_id": "arxiv:2606.02388", "pdf_url": "https://arxiv.org/pdf/2606.02388", "primary_query": "language-agent" }, { "id": "2606.01815", "title": "CRAB-Bench: Evaluating LLM Agents under Complex Task Dependencies and Human-aligned User Simulation", "url": "https://arxiv.org/abs/2606.01815", "published": "2026-06-01", "updated": "2026-06-01", "authors": [ "Danqing Wang", "Akshay Sivaraman", "Lei Li" ], "categories": [ "cs.CL" ], "topics": [ "agent-evaluation", "tool-use", "world-model" ], "score": 15, "relevance": "high", "matched_queries": [ "agent-evaluation" ], "arxiv_id": "2606.01815", "source": "arxiv", "source_id": "arxiv:2606.01815", "pdf_url": "https://arxiv.org/pdf/2606.01815", "primary_query": "agent-evaluation" }, { "id": "2606.02380", "title": "SPADE-Bench: Evaluating Spontaneous Strategic Deception in Agents via Plan-Action Divergence", "url": "https://arxiv.org/abs/2606.02380", "published": "2026-06-01", "updated": "2026-06-28", "authors": [ "Yuyan Bu", "Haowei Li", "Qirui Zheng", "Bowen Dong", "Kaiyue Yang", "Jiaming Ji", "Yingshui Tan", "Wenxin Li", "Yaodong Yang", "Juntao Dai" ], "categories": [ "cs.CL", "cs.AI" ], "topics": [ "agent-evaluation", "agent-safety", "computer-use", "planning", "tool-use" ], "score": 15, "relevance": "high", "matched_queries": [ "agent-safety" ], "arxiv_id": "2606.02380", "source": "arxiv", "source_id": "arxiv:2606.02380", "pdf_url": "https://arxiv.org/pdf/2606.02380", "primary_query": "agent-safety" }, { "id": "2606.00914", "title": "Adversarial Feeds Steer LLM Agent Decisions Against Their Defaults", "url": "https://arxiv.org/abs/2606.00914", "published": "2026-05-30", "updated": "2026-05-30", "authors": [ "Rana Muhammad Usman" ], "categories": [ "cs.AI", "cs.CL", "cs.CR" ], "topics": [ "agent-evaluation", "agent-safety", "rag", "tool-use" ], "score": 15, "relevance": "high", "matched_queries": [ "agent-evaluation" ], "arxiv_id": "2606.00914", "source": "arxiv", "source_id": "arxiv:2606.00914", "pdf_url": "https://arxiv.org/pdf/2606.00914", "primary_query": "agent-evaluation" }, { "id": "2606.00915", "title": "Autonomous agentic design for photonics", "url": "https://arxiv.org/abs/2606.00915", "published": "2026-05-30", "updated": "2026-05-30", "authors": [ "Prashanta Kharel", "Amin Khavasi", "Xinzhong Chen", "Tyler W. Hughes" ], "categories": [ "physics.optics" ], "topics": [ "agent-evaluation", "computer-use", "tool-use", "world-model" ], "score": 15, "relevance": "high", "matched_queries": [ "autonomous-agent-llm" ], "arxiv_id": "2606.00915", "source": "arxiv", "source_id": "arxiv:2606.00915", "pdf_url": "https://arxiv.org/pdf/2606.00915", "primary_query": "autonomous-agent-llm" }, { "id": "2605.31278", "title": "Industrializing Prediction-Powered Inference: The GLIDE Library for Reliable GenAI and Agentic Systems Evaluation", "url": "https://arxiv.org/abs/2605.31278", "published": "2026-05-29", "updated": "2026-06-04", "authors": [ "Grégoire Martinon", "Ibrahim Merad", "Mohammed Raki" ], "categories": [ "cs.AI", "cs.LG", "stat.ME" ], "topics": [ "agent-evaluation", "tool-use" ], "score": 15, "relevance": "high", "matched_queries": [ "agent-evaluation" ], "arxiv_id": "2605.31278", "source": "arxiv", "source_id": "arxiv:2605.31278", "pdf_url": "https://arxiv.org/pdf/2605.31278", "primary_query": "agent-evaluation" }, { "id": "2605.29676", "title": "Notation Matters: A Benchmark Study of Token-Optimized Formats in Agentic AI Systems", "url": "https://arxiv.org/abs/2605.29676", "published": "2026-05-28", "updated": "2026-06-17", "authors": [ "Lorenz Kutschka", "Bernhard Geiger" ], "categories": [ "cs.AI", "cs.CL" ], "topics": [ "agent-evaluation", "tool-use" ], "score": 15, "relevance": "high", "matched_queries": [ "agent-evaluation" ], "arxiv_id": "2605.29676", "source": "arxiv", "source_id": "arxiv:2605.29676", "pdf_url": "https://arxiv.org/pdf/2605.29676", "primary_query": "agent-evaluation" }, { "id": "2605.28046", "title": "MemCog: From Memory-as-Tool to Memory-as-Cognition in Conversational Agents", "url": "https://arxiv.org/abs/2605.28046", "published": "2026-05-27", "updated": "2026-05-27", "authors": [ "Zihan Li", "Xingyu Fan", "Feifei Li", "Wenhui Que" ], "categories": [ "cs.AI", "cs.CL" ], "topics": [ "agent-evaluation", "embodied-agent", "memory", "rag", "reasoning", "tool-use" ], "score": 15, "relevance": "high", "matched_queries": [ "agent-memory" ], "arxiv_id": "2605.28046", "source": "arxiv", "source_id": "arxiv:2605.28046", "pdf_url": "https://arxiv.org/pdf/2605.28046", "primary_query": "agent-memory" }, { "id": "2605.28607", "title": "Adaptive Multimodal Agents-Based Framework for Automatic Workflow Execution", "url": "https://arxiv.org/abs/2605.28607", "published": "2026-05-27", "updated": "2026-05-27", "authors": [ "Susanna Cifani", "Mario Luca Bernardi", "Marta Cimitile" ], "categories": [ "cs.AI", "cs.CL" ], "topics": [ "agent-evaluation", "computer-use", "embodied-agent", "multi-agent", "planning", "rag", "workflow-agent" ], "score": 15, "relevance": "high", "matched_queries": [ "rag-agent" ], "arxiv_id": "2605.28607", "source": "arxiv", "source_id": "arxiv:2605.28607", "pdf_url": "https://arxiv.org/pdf/2605.28607", "primary_query": "rag-agent" }, { "id": "2605.28120", "title": "LegalGraphRAG: Multi-Agent Graph Retrieval-Augmented Generation for Reliable Legal Reasoning", "url": "https://arxiv.org/abs/2605.28120", "published": "2026-05-27", "updated": "2026-05-27", "authors": [ "Zerui Chen", "Qinggang Zhang", "Zhishang Xiang", "Zhimin Wei", "Linfeng Gao", "Xiao Huang", "Zhihong Zhang", "Jinsong Su" ], "categories": [ "cs.CL", "cs.AI", "cs.MA" ], "topics": [ "agent-evaluation", "multi-agent", "rag", "reasoning", "tool-use" ], "score": 15, "relevance": "high", "matched_queries": [ "rag-agent" ], "arxiv_id": "2605.28120", "source": "arxiv", "source_id": "arxiv:2605.28120", "pdf_url": "https://arxiv.org/pdf/2605.28120", "primary_query": "rag-agent" }, { "id": "2605.28787", "title": "Do Agents Need Semantic Metadata? A Comparative Study in Agentic Data Retrieval", "url": "https://arxiv.org/abs/2605.28787", "published": "2026-05-27", "updated": "2026-05-27", "authors": [ "Shiyu Chen", "Tarfah Alrashed", "Alon Halevy", "Natasha Noy" ], "categories": [ "cs.IR", "cs.AI" ], "topics": [ "agent-evaluation", "rag", "tool-use", "workflow-agent" ], "score": 15, "relevance": "high", "matched_queries": [ "autonomous-agent-llm" ], "arxiv_id": "2605.28787", "source": "arxiv", "source_id": "arxiv:2605.28787", "pdf_url": "https://arxiv.org/pdf/2605.28787", "primary_query": "autonomous-agent-llm" }, { "id": "2605.27366", "title": "MUSE-Autoskill: Self-Evolving Agents via Skill Creation, Memory, Management, and Evaluation", "url": "https://arxiv.org/abs/2605.27366", "published": "2026-05-26", "updated": "2026-07-03", "authors": [ "Huawei Lin", "Peng Li", "Jie Song", "Fuxin Jiang", "Tieying Zhang" ], "categories": [ "cs.AI", "cs.CL", "cs.LG", "cs.MA" ], "topics": [ "agent-evaluation", "memory" ], "score": 15, "relevance": "high", "matched_queries": [ "agent-memory" ], "arxiv_id": "2605.27366", "source": "arxiv", "source_id": "arxiv:2605.27366", "pdf_url": "https://arxiv.org/pdf/2605.27366", "primary_query": "agent-memory" }, { "id": "2605.26926", "title": "From Norms to Indicators (N2I-RAG): An Agentic Retrieval-Augmented Generation Framework for Legal Indicator Computation", "url": "https://arxiv.org/abs/2605.26926", "published": "2026-05-26", "updated": "2026-05-26", "authors": [ "Youssef Al Mouatamid", "Marie Bonnin", "Jihad Zahir" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "agent-safety", "planning", "rag" ], "score": 15, "relevance": "high", "matched_queries": [ "rag-agent" ], "arxiv_id": "2605.26926", "source": "arxiv", "source_id": "arxiv:2605.26926", "pdf_url": "https://arxiv.org/pdf/2605.26926", "primary_query": "rag-agent" }, { "id": "2605.27333", "title": "FinHarness: An Inline Lifecycle Safety Harness for Finance LLM Agents", "url": "https://arxiv.org/abs/2605.27333", "published": "2026-05-26", "updated": "2026-05-26", "authors": [ "Haoxuan Jia", "Yang Liu", "Bin Chong", "Yingguang Yang", "Yancheng Chen", "Jiayu Liang", "Qian Li", "Hanning Lu", "Kefu Xu", "Hao Zheng", "Chongyang Zhang", "Hao Peng", "Philip S. Yu" ], "categories": [ "cs.CL" ], "topics": [ "agent-evaluation", "agent-safety", "planning", "tool-use", "workflow-agent" ], "score": 15, "relevance": "high", "matched_queries": [ "planning-agent" ], "arxiv_id": "2605.27333", "source": "arxiv", "source_id": "arxiv:2605.27333", "pdf_url": "https://arxiv.org/pdf/2605.27333", "primary_query": "planning-agent" }, { "id": "2605.26720", "title": "Towards Feedback-to-Plan Decisions for Self-Evolving LLM Agents in CUDA Kernel Generation", "url": "https://arxiv.org/abs/2605.26720", "published": "2026-05-26", "updated": "2026-05-26", "authors": [ "Yee Hin Chong", "Jiaming Wu", "Youhui Zhang", "Peng Qu" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "planning", "reasoning", "tool-use" ], "score": 15, "relevance": "high", "matched_queries": [ "planning-agent" ], "arxiv_id": "2605.26720", "source": "arxiv", "source_id": "arxiv:2605.26720", "pdf_url": "https://arxiv.org/pdf/2605.26720", "primary_query": "planning-agent" }, { "id": "2605.26252", "title": "Is Agent Memory a Database? Rethinking Data Foundations for Long-Term AI Agent Memory", "url": "https://arxiv.org/abs/2605.26252", "published": "2026-05-25", "updated": "2026-05-25", "authors": [ "Abdelghny Orogat", "Essam Mansour" ], "categories": [ "cs.AI", "cs.DB" ], "topics": [ "agent-evaluation", "memory", "rag", "tool-use" ], "score": 15, "relevance": "high", "matched_queries": [ "agent-memory" ], "arxiv_id": "2605.26252", "source": "arxiv", "source_id": "arxiv:2605.26252", "pdf_url": "https://arxiv.org/pdf/2605.26252", "primary_query": "agent-memory" }, { "id": "2605.26305", "title": "Experiments in Agentic AI for Science", "url": "https://arxiv.org/abs/2605.26305", "published": "2026-05-25", "updated": "2026-05-29", "authors": [ "Judy Fox", "Geoffrey Fox" ], "categories": [ "cs.AI", "eess.SY", "hep-ph" ], "topics": [ "rag", "reasoning", "tool-use", "workflow-agent" ], "score": 15, "relevance": "high", "matched_queries": [ "autonomous-agent-llm", "rag-agent" ], "arxiv_id": "2605.26305", "source": "arxiv", "source_id": "arxiv:2605.26305", "pdf_url": "https://arxiv.org/pdf/2605.26305", "primary_query": "autonomous-agent-llm" }, { "id": "2605.23636", "title": "RF Instrument Agent (RFIA): Empowering RF Instruments with Natural Language Understanding, Scheduling and Execution of Complex Tasks", "url": "https://arxiv.org/abs/2605.23636", "published": "2026-05-22", "updated": "2026-05-22", "authors": [ "Chunhui Li", "Wei Fan" ], "categories": [ "eess.SY" ], "topics": [ "agent-evaluation", "agent-safety", "planning", "rag", "tool-use", "workflow-agent" ], "score": 15, "relevance": "high", "matched_queries": [ "language-agent" ], "arxiv_id": "2605.23636", "source": "arxiv", "source_id": "arxiv:2605.23636", "pdf_url": "https://arxiv.org/pdf/2605.23636", "primary_query": "language-agent" }, { "id": "2605.22321", "title": "Benchmarking Autonomous Agents against Temporal, Spatial, and Semantic Evasions", "url": "https://arxiv.org/abs/2605.22321", "published": "2026-05-21", "updated": "2026-05-21", "authors": [ "Jianan Ma", "Xiaohu Du", "Ruixiao Lin", "Yaoxiang Bian", "Jialuo Chen", "Jingyi Wang", "Xiaofang Yang", "Shiwen Cui", "Changhua Meng", "Xinhao Deng", "Zhen Wang" ], "categories": [ "cs.CR", "cs.AI", "cs.SE" ], "topics": [ "agent-evaluation", "agent-safety", "memory", "rag", "tool-use" ], "score": 15, "relevance": "high", "matched_queries": [ "autonomous-agent-llm" ], "arxiv_id": "2605.22321", "source": "arxiv", "source_id": "arxiv:2605.22321", "pdf_url": "https://arxiv.org/pdf/2605.22321", "primary_query": "autonomous-agent-llm" }, { "id": "2605.21740", "title": "SMDD-Bench: Can LLMs Solve Real-World Small Molecule Drug Design Tasks?", "url": "https://arxiv.org/abs/2605.21740", "published": "2026-05-20", "updated": "2026-05-24", "authors": [ "Kevin Han", "Renfei Zhang", "Kathy Wei", "Hamed Mahdavi", "Niloofar Mireshghallah", "Amir Barati Farimani" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "planning", "rag", "reasoning", "tool-use" ], "score": 15, "relevance": "high", "matched_queries": [ "planning-agent" ], "arxiv_id": "2605.21740", "source": "arxiv", "source_id": "arxiv:2605.21740", "pdf_url": "https://arxiv.org/pdf/2605.21740", "primary_query": "planning-agent" }, { "id": "2605.20874", "title": "Governance by Construction for Generalist Agents", "url": "https://arxiv.org/abs/2605.20874", "published": "2026-05-20", "updated": "2026-05-20", "authors": [ "Segev Shlomov", "Iftach Shoham", "Alon Oved", "Ido Levy", "Sami Marreed", "Harold Ship", "Offer Akrabi", "Sergey Zeltyn", "Avi Yaeli", "Nir Mashkif" ], "categories": [ "cs.AI", "cs.SE" ], "topics": [ "agent-safety", "computer-use", "planning", "reasoning", "tool-use", "workflow-agent" ], "score": 15, "relevance": "high", "matched_queries": [ "planning-agent" ], "arxiv_id": "2605.20874", "source": "arxiv", "source_id": "arxiv:2605.20874", "pdf_url": "https://arxiv.org/pdf/2605.20874", "primary_query": "planning-agent" }, { "id": "2605.20306", "title": "WildRoadBench: A Wild Aerial Road-Damage Grounding Benchmark for Vision-Language Models and Autonomous Agents", "url": "https://arxiv.org/abs/2605.20306", "published": "2026-05-19", "updated": "2026-06-02", "authors": [ "Bingnan Liu", "Chenhang Cui", "Rui Huang", "Jiani Luo", "Zhirong Shen", "Tinghao Wang", "Xiande Huang", "Lingbei Meng", "Fei Shen", "An Zhang" ], "categories": [ "cs.CV", "cs.LG" ], "topics": [ "agent-evaluation", "coding-agent", "rag", "reasoning", "tool-use" ], "score": 15, "relevance": "high", "matched_queries": [ "autonomous-agent-llm" ], "arxiv_id": "2605.20306", "source": "arxiv", "source_id": "arxiv:2605.20306", "pdf_url": "https://arxiv.org/pdf/2605.20306", "primary_query": "autonomous-agent-llm" }, { "id": "2605.18672", "title": "Position: A Three-Layer Probabilistic Assume-Guarantee Architecture Is Structurally Required for Safe LLM Agent Deployment", "url": "https://arxiv.org/abs/2605.18672", "published": "2026-05-18", "updated": "2026-05-18", "authors": [ "S. Bensalem", "Y. Dong", "M. Franzle", "X. Huang", "J. Kroger", "D. Nickovic", "A. Nouri", "R. Roy", "C. Wu" ], "categories": [ "cs.AI" ], "topics": [ "agent-safety", "multi-agent", "tool-use" ], "score": 15, "relevance": "high", "matched_queries": [ "agent-safety" ], "arxiv_id": "2605.18672", "source": "arxiv", "source_id": "arxiv:2605.18672", "pdf_url": "https://arxiv.org/pdf/2605.18672", "primary_query": "agent-safety" }, { "id": "2605.17348", "title": "Taming \"Zombie'' Agents: A Markov State-Aware Framework for Resilient Multi-Agent Evolution", "url": "https://arxiv.org/abs/2605.17348", "published": "2026-05-17", "updated": "2026-05-17", "authors": [ "Taolin Zhang", "Pukun Zhao", "Qizhou Chen", "Jiuheng Wan", "Chen Chen", "Xiaofeng He", "Chengyu Wang", "Richang Hong" ], "categories": [ "cs.CL" ], "topics": [ "agent-safety", "memory", "multi-agent", "reasoning" ], "score": 15, "relevance": "high", "matched_queries": [ "agent-memory" ], "arxiv_id": "2605.17348", "source": "arxiv", "source_id": "arxiv:2605.17348", "pdf_url": "https://arxiv.org/pdf/2605.17348", "primary_query": "agent-memory" }, { "id": "2605.17453", "title": "Trust No Tool: Evaluating and Defending LLM Agents under Untrusted Tool Feedback", "url": "https://arxiv.org/abs/2605.17453", "published": "2026-05-17", "updated": "2026-05-17", "authors": [ "Lecheng Yan", "Ruizhe Li", "Xicheng Han", "Wenxi Li", "Binwu Wang", "Longyue Wang", "Chenyang Lyu", "Guanhua Chen" ], "categories": [ "cs.CR", "cs.CL" ], "topics": [ "agent-evaluation", "agent-safety", "tool-use" ], "score": 15, "relevance": "high", "matched_queries": [ "agent-safety" ], "arxiv_id": "2605.17453", "source": "arxiv", "source_id": "arxiv:2605.17453", "pdf_url": "https://arxiv.org/pdf/2605.17453", "primary_query": "agent-safety" }, { "id": "2605.23986", "title": "MemForest: An Efficient Agent Memory System with Hierarchical Temporal Indexing", "url": "https://arxiv.org/abs/2605.23986", "published": "2026-05-16", "updated": "2026-05-16", "authors": [ "Han Chen", "Zining Zhang", "Wenqi Pei", "Bingsheng He", "Ming Wu", "Jason Zeng", "Michael Heinrich", "Wei Wu", "Hongbao Zhang" ], "categories": [ "cs.DB", "cs.AI", "cs.MA" ], "topics": [ "agent-evaluation", "memory", "tool-use" ], "score": 15, "relevance": "high", "matched_queries": [ "agent-memory" ], "arxiv_id": "2605.23986", "source": "arxiv", "source_id": "arxiv:2605.23986", "pdf_url": "https://arxiv.org/pdf/2605.23986", "primary_query": "agent-memory" }, { "id": "2605.28850", "title": "Representation Signatures and Risk-Feedback Alignment in LLM Trading Agents", "url": "https://arxiv.org/abs/2605.28850", "published": "2026-05-16", "updated": "2026-05-30", "authors": [ "Weicheng Xue" ], "categories": [ "cs.LG", "q-fin.CP" ], "topics": [ "agent-safety", "coding-agent", "memory", "planning", "reasoning", "tool-use", "world-model" ], "score": 15, "relevance": "high", "matched_queries": [ "planning-agent" ], "arxiv_id": "2605.28850", "source": "arxiv", "source_id": "arxiv:2605.28850", "pdf_url": "https://arxiv.org/pdf/2605.28850", "primary_query": "planning-agent" }, { "id": "2605.14460", "title": "Exploiting LLM Agent Supply Chains via Payload-less Skills", "url": "https://arxiv.org/abs/2605.14460", "published": "2026-05-14", "updated": "2026-05-14", "authors": [ "Xinyu Liu", "Yukai Zhao", "Xing Hu", "Xin Xia" ], "categories": [ "cs.CR", "cs.SE" ], "topics": [ "agent-evaluation", "agent-safety", "coding-agent", "tool-use" ], "score": 15, "relevance": "high", "matched_queries": [ "autonomous-agent-llm" ], "arxiv_id": "2605.14460", "source": "arxiv", "source_id": "arxiv:2605.14460", "pdf_url": "https://arxiv.org/pdf/2605.14460", "primary_query": "autonomous-agent-llm" }, { "id": "2605.14126", "title": "Reinforcement Learning for Tool-Calling Agents in Fast Healthcare Interoperability Resources (FHIR)", "url": "https://arxiv.org/abs/2605.14126", "published": "2026-05-13", "updated": "2026-05-13", "authors": [ "Marius S. Knorr", "Robert Müller", "Jan P. Bremer", "Nils Schweingruber" ], "categories": [ "cs.LG", "cs.AI" ], "topics": [ "agent-evaluation", "planning", "rag", "reasoning", "tool-use" ], "score": 15, "relevance": "high", "matched_queries": [ "planning-agent" ], "arxiv_id": "2605.14126", "source": "arxiv", "source_id": "arxiv:2605.14126", "pdf_url": "https://arxiv.org/pdf/2605.14126", "primary_query": "planning-agent" }, { "id": "2605.11882", "title": "On-Policy Self-Evolution via Failure Trajectories for Agentic Safety Alignment", "url": "https://arxiv.org/abs/2605.11882", "published": "2026-05-12", "updated": "2026-05-12", "authors": [ "Bo Yin", "Qi Li", "Xinchao Wang" ], "categories": [ "cs.AI" ], "topics": [ "agent-safety", "reasoning", "tool-use" ], "score": 15, "relevance": "high", "matched_queries": [ "agent-safety" ], "arxiv_id": "2605.11882", "source": "arxiv", "source_id": "arxiv:2605.11882", "pdf_url": "https://arxiv.org/pdf/2605.11882", "primary_query": "agent-safety" }, { "id": "2605.11534", "title": "PRISM: : Planning and Reasoning with Intent in Simulated Embodied Environments", "url": "https://arxiv.org/abs/2605.11534", "published": "2026-05-12", "updated": "2026-05-12", "authors": [ "Yunn Kang Lim", "Pengzhan Sun", "Ziyi Bai", "Xun Xu", "Angela Yao", "Xulei Yang", "Shijie Li" ], "categories": [ "cs.RO" ], "topics": [ "agent-evaluation", "embodied-agent", "memory", "planning", "rag", "reasoning", "tool-use" ], "score": 15, "relevance": "high", "matched_queries": [ "planning-agent" ], "arxiv_id": "2605.11534", "source": "arxiv", "source_id": "arxiv:2605.11534", "pdf_url": "https://arxiv.org/pdf/2605.11534", "primary_query": "planning-agent" }, { "id": "2605.11388", "title": "Deep Reasoning in General Purpose Agents via Structured Meta-Cognition", "url": "https://arxiv.org/abs/2605.11388", "published": "2026-05-12", "updated": "2026-05-12", "authors": [ "Dean Light", "Michael Theologitis", "Kshitish Ghate", "Shuyue Stella Li", "Benjamin Newman", "Chirag Shah", "Aylin Caliskan", "Pang Wei Koh", "Dan Suciu", "Yulia Tsvetkov" ], "categories": [ "cs.CL", "cs.AI" ], "topics": [ "agent-evaluation", "computer-use", "planning", "rag", "reasoning" ], "score": 15, "relevance": "high", "matched_queries": [ "planning-agent" ], "arxiv_id": "2605.11388", "source": "arxiv", "source_id": "arxiv:2605.11388", "pdf_url": "https://arxiv.org/pdf/2605.11388", "primary_query": "planning-agent" }, { "id": "2605.11039", "title": "The Granularity Mismatch in Agent Security: Argument-Level Provenance Solves Enforcement and Isolates the LLM Reasoning Bottleneck", "url": "https://arxiv.org/abs/2605.11039", "published": "2026-05-11", "updated": "2026-05-11", "authors": [ "Linfeng Fan", "Ziwei Li", "Yuan Tian", "Yichen Wang", "Rongsheng Li", "Xiong Wang" ], "categories": [ "cs.CR", "cs.AI" ], "topics": [ "agent-evaluation", "agent-safety", "planning", "rag", "reasoning", "tool-use", "workflow-agent" ], "score": 15, "relevance": "high", "matched_queries": [ "agent-safety" ], "arxiv_id": "2605.11039", "source": "arxiv", "source_id": "arxiv:2605.11039", "pdf_url": "https://arxiv.org/pdf/2605.11039", "primary_query": "agent-safety" }, { "id": "2605.10763", "title": "MATRA: Modeling the Attack Surface of Agentic AI Systems -- OpenClaw Case Study", "url": "https://arxiv.org/abs/2605.10763", "published": "2026-05-11", "updated": "2026-05-11", "authors": [ "Tim Van hamme", "Thomas Vissers", "Javier Carnerero-Cano", "Mario Fritz", "Emil C. Lupu", "Lieven Desmet", "Dinil Mon Divakaran" ], "categories": [ "cs.AI", "cs.CR" ], "topics": [ "agent-evaluation", "agent-safety", "rag", "tool-use" ], "score": 15, "relevance": "high", "matched_queries": [ "autonomous-agent-llm" ], "arxiv_id": "2605.10763", "source": "arxiv", "source_id": "arxiv:2605.10763", "pdf_url": "https://arxiv.org/pdf/2605.10763", "primary_query": "autonomous-agent-llm" }, { "id": "2605.10365", "title": "Agent-ValueBench: A Comprehensive Benchmark for Evaluating Agent Values", "url": "https://arxiv.org/abs/2605.10365", "published": "2026-05-11", "updated": "2026-05-11", "authors": [ "Haonan Dong", "Qiguan Feng", "Kehan Jiang", "Haoran Ye", "Xin Zhang", "Guojie Song" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "agent-safety", "tool-use" ], "score": 15, "relevance": "high", "matched_queries": [ "autonomous-agent-llm" ], "arxiv_id": "2605.10365", "source": "arxiv", "source_id": "arxiv:2605.10365", "pdf_url": "https://arxiv.org/pdf/2605.10365", "primary_query": "autonomous-agent-llm" }, { "id": "2605.09168", "title": "CIVeX: Causal Intervention Verification for Language Agents", "url": "https://arxiv.org/abs/2605.09168", "published": "2026-05-09", "updated": "2026-05-09", "authors": [ "Fabio Rovai" ], "categories": [ "cs.AI", "cs.LG" ], "topics": [ "agent-safety", "reasoning", "tool-use", "workflow-agent" ], "score": 15, "relevance": "high", "matched_queries": [ "language-agent" ], "arxiv_id": "2605.09168", "source": "arxiv", "source_id": "arxiv:2605.09168", "pdf_url": "https://arxiv.org/pdf/2605.09168", "primary_query": "language-agent" }, { "id": "2605.08763", "title": "When LLMs Team Up: A Coordinated Attack Framework for Automated Cyber Intrusions", "url": "https://arxiv.org/abs/2605.08763", "published": "2026-05-09", "updated": "2026-05-09", "authors": [ "Minfeng Qi", "Tianqing Zhu", "Zijie Xu", "Congcong Zhu", "Qin Wang", "Wanlei Zhou" ], "categories": [ "cs.CR" ], "topics": [ "agent-evaluation", "agent-safety", "multi-agent", "planning", "rag", "tool-use", "workflow-agent" ], "score": 15, "relevance": "high", "matched_queries": [ "planning-agent" ], "arxiv_id": "2605.08763", "source": "arxiv", "source_id": "arxiv:2605.08763", "pdf_url": "https://arxiv.org/pdf/2605.08763", "primary_query": "planning-agent" }, { "id": "2605.06078", "title": "Milestone-Guided Policy Learning for Long-Horizon Language Agents", "url": "https://arxiv.org/abs/2605.06078", "published": "2026-05-07", "updated": "2026-05-07", "authors": [ "Zixuan Wang", "Yuchen Yan", "Hongxing Li", "Teng Pan", "Dingming Li", "Ruiqing Zhang", "Weiming Lu", "Jun Xiao", "Yueting Zhuang", "Yongliang Shen" ], "categories": [ "cs.CL", "cs.AI" ], "topics": [ "agent-evaluation", "computer-use", "planning", "rag", "tool-use" ], "score": 15, "relevance": "high", "matched_queries": [ "language-agent" ], "arxiv_id": "2605.06078", "source": "arxiv", "source_id": "arxiv:2605.06078", "pdf_url": "https://arxiv.org/pdf/2605.06078", "primary_query": "language-agent" }, { "id": "2605.03505", "title": "LATS-RCA: Language Agent Tree Search for Root Cause Analysis in Microservices", "url": "https://arxiv.org/abs/2605.03505", "published": "2026-05-05", "updated": "2026-06-12", "authors": [ "Alexander Naakka", "Yuqing Wang", "Mika V Mäntylä" ], "categories": [ "cs.SE" ], "topics": [ "agent-evaluation", "computer-use", "rag", "reasoning" ], "score": 15, "relevance": "high", "matched_queries": [ "language-agent" ], "arxiv_id": "2605.03505", "source": "arxiv", "source_id": "arxiv:2605.03505", "pdf_url": "https://arxiv.org/pdf/2605.03505", "primary_query": "language-agent" }, { "id": "2605.04107", "title": "TSCG: Deterministic Tool-Schema Compilation for Agentic LLM Deployments", "url": "https://arxiv.org/abs/2605.04107", "published": "2026-05-04", "updated": "2026-05-04", "authors": [ "Furkan Sakizli" ], "categories": [ "cs.SE", "cs.AI", "cs.CL" ], "topics": [ "agent-evaluation", "computer-use", "tool-use" ], "score": 15, "relevance": "high", "matched_queries": [ "function-calling" ], "arxiv_id": "2605.04107", "source": "arxiv", "source_id": "arxiv:2605.04107", "pdf_url": "https://arxiv.org/pdf/2605.04107", "primary_query": "function-calling" }, { "id": "2605.05242", "title": "Beyond Semantic Similarity: Rethinking Retrieval for Agentic Search via Direct Corpus Interaction", "url": "https://arxiv.org/abs/2605.05242", "published": "2026-05-03", "updated": "2026-05-03", "authors": [ "Zhuofeng Li", "Haoxiang Zhang", "Cong Wei", "Pan Lu", "Ping Nie", "Yi Lu", "Yuyang Bai", "Shangbin Feng", "Hangxiao Zhu", "Ming Zhong", "Yuyu Zhang", "Jianwen Xie", "Yejin Choi", "James Zou", "Jiawei Han", "Wenhu Chen", "Jimmy Lin", "Dongfu Jiang", "Yu Zhang" ], "categories": [ "cs.IR", "cs.AI" ], "topics": [ "agent-evaluation", "planning", "rag", "reasoning", "tool-use" ], "score": 15, "relevance": "high", "matched_queries": [ "language-agent" ], "arxiv_id": "2605.05242", "source": "arxiv", "source_id": "arxiv:2605.05242", "pdf_url": "https://arxiv.org/pdf/2605.05242", "primary_query": "language-agent" }, { "id": "2605.00081", "title": "Alignment Contracts for Agentic Security Systems", "url": "https://arxiv.org/abs/2605.00081", "published": "2026-04-30", "updated": "2026-04-30", "authors": [ "Isaac David", "Marco Guarnieri", "Arthur Gervais" ], "categories": [ "cs.CR", "cs.LO" ], "topics": [ "agent-evaluation", "agent-safety", "planning", "tool-use", "workflow-agent" ], "score": 15, "relevance": "high", "matched_queries": [ "agent-safety" ], "arxiv_id": "2605.00081", "source": "arxiv", "source_id": "arxiv:2605.00081", "pdf_url": "https://arxiv.org/pdf/2605.00081", "primary_query": "agent-safety" }, { "id": "2604.27699", "title": "Bridging Values and Behavior: A Hierarchical Framework for Proactive Embodied Agents", "url": "https://arxiv.org/abs/2604.27699", "published": "2026-04-30", "updated": "2026-04-30", "authors": [ "Chunhui Zhang", "Yuxuan Wang", "Aoyang Qin", "Yi-Long Lu", "Kunlun Wu", "Yizhou Wang", "Wei Wang" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "agent-safety", "embodied-agent", "memory", "planning", "reasoning", "tool-use" ], "score": 15, "relevance": "high", "matched_queries": [ "autonomous-agent-llm" ], "arxiv_id": "2604.27699", "source": "arxiv", "source_id": "arxiv:2604.27699", "pdf_url": "https://arxiv.org/pdf/2604.27699", "primary_query": "autonomous-agent-llm" }, { "id": "2604.26274", "title": "Enforcing Benign Trajectories: A Behavioral Firewall for Structured-Workflow AI Agents", "url": "https://arxiv.org/abs/2604.26274", "published": "2026-04-29", "updated": "2026-04-29", "authors": [ "Hung Dang" ], "categories": [ "cs.CR", "cs.AI" ], "topics": [ "agent-evaluation", "agent-safety", "rag", "tool-use", "workflow-agent" ], "score": 15, "relevance": "high", "matched_queries": [ "agent-safety" ], "arxiv_id": "2604.26274", "source": "arxiv", "source_id": "arxiv:2604.26274", "pdf_url": "https://arxiv.org/pdf/2604.26274", "primary_query": "agent-safety" }, { "id": "2604.24826", "title": "A Comparative Evaluation of AI Agent Security Guardrails", "url": "https://arxiv.org/abs/2604.24826", "published": "2026-04-27", "updated": "2026-04-27", "authors": [ "Qi Li", "Jiu Li", "Pingtao Wei", "Jianjun Xu", "Xueyi Wei", "Jiwei Shi", "Xuan Zhang", "Yanhui Yang", "Xiaodong Hui", "Peng Xu", "Lingquan Zhou" ], "categories": [ "cs.CR", "cs.AI" ], "topics": [ "agent-evaluation", "agent-safety", "tool-use" ], "score": 15, "relevance": "high", "matched_queries": [ "agent-safety" ], "arxiv_id": "2604.24826", "source": "arxiv", "source_id": "arxiv:2604.24826", "pdf_url": "https://arxiv.org/pdf/2604.24826", "primary_query": "agent-safety" }, { "id": "2606.13686", "title": "Benchmarking Web Agent Safety under E-commerce Deceptive Interfaces", "url": "https://arxiv.org/abs/2606.13686", "published": "2026-04-26", "updated": "2026-04-26", "authors": [ "Zijing Shi", "Meng Fang", "Ling Chen" ], "categories": [ "cs.CL", "cs.CY" ], "topics": [ "agent-evaluation", "agent-safety", "computer-use" ], "score": 15, "relevance": "high", "matched_queries": [ "agent-safety" ], "arxiv_id": "2606.13686", "source": "arxiv", "source_id": "arxiv:2606.13686", "pdf_url": "https://arxiv.org/pdf/2606.13686", "primary_query": "agent-safety" }, { "id": "2604.19821", "title": "JTPRO: A Joint Tool-Prompt Reflective Optimization Framework for Language Agents", "url": "https://arxiv.org/abs/2604.19821", "published": "2026-04-20", "updated": "2026-04-20", "authors": [ "Sandip Ghoshal", "Anshul Mittal", "Jyotika Singh", "Miguel Ballesteros", "Weiyi Sun", "Fang Tu", "Shailender Singh", "Yassine Benajiba", "Fahad Shah", "Sujeeth Bharadwaj", "Sujith Ravi", "Dan Roth" ], "categories": [ "cs.AI", "cs.SE" ], "topics": [ "agent-evaluation", "computer-use", "reasoning", "tool-use" ], "score": 15, "relevance": "high", "matched_queries": [ "language-agent" ], "arxiv_id": "2604.19821", "source": "arxiv", "source_id": "arxiv:2604.19821", "pdf_url": "https://arxiv.org/pdf/2604.19821", "primary_query": "language-agent" }, { "id": "2604.18718", "title": "Towards Optimal Agentic Architectures for Offensive Security Tasks", "url": "https://arxiv.org/abs/2604.18718", "published": "2026-04-20", "updated": "2026-04-20", "authors": [ "Isaac David", "Arthur Gervais" ], "categories": [ "cs.CR", "cs.AI" ], "topics": [ "agent-evaluation", "agent-safety", "rag", "tool-use" ], "score": 15, "relevance": "high", "matched_queries": [ "agent-safety" ], "arxiv_id": "2604.18718", "source": "arxiv", "source_id": "arxiv:2604.18718", "pdf_url": "https://arxiv.org/pdf/2604.18718", "primary_query": "agent-safety" }, { "id": "2604.12986", "title": "Parallax: Why AI Agents That Think Must Never Act", "url": "https://arxiv.org/abs/2604.12986", "published": "2026-04-14", "updated": "2026-04-14", "authors": [ "Joel Fokou" ], "categories": [ "cs.CR", "cs.AI" ], "topics": [ "agent-evaluation", "agent-safety", "reasoning", "tool-use", "workflow-agent" ], "score": 15, "relevance": "high", "matched_queries": [ "agent-safety" ], "arxiv_id": "2604.12986", "source": "arxiv", "source_id": "arxiv:2604.12986", "pdf_url": "https://arxiv.org/pdf/2604.12986", "primary_query": "agent-safety" }, { "id": "2604.06972", "title": "Differentiable Environment-Trajectory Co-Optimization for Safe Multi-Agent Navigation", "url": "https://arxiv.org/abs/2604.06972", "published": "2026-04-08", "updated": "2026-04-08", "authors": [ "Zhan Gao", "Gabriele Fadini", "Stelian Coros", "Amanda Prorok" ], "categories": [ "cs.RO", "cs.MA" ], "topics": [ "agent-evaluation", "agent-safety", "computer-use", "embodied-agent", "multi-agent", "rag", "tool-use" ], "score": 15, "relevance": "high", "matched_queries": [ "agent-safety" ], "arxiv_id": "2604.06972", "source": "arxiv", "source_id": "arxiv:2604.06972", "pdf_url": "https://arxiv.org/pdf/2604.06972", "primary_query": "agent-safety" }, { "id": "2604.04426", "title": "ShieldNet: Network-Level Guardrails against Emerging Supply-Chain Injections in Agentic Systems", "url": "https://arxiv.org/abs/2604.04426", "published": "2026-04-06", "updated": "2026-04-06", "authors": [ "Zhuowen Yuan", "Zhaorun Chen", "Zhen Xiang", "Nathaniel D. Bastian", "Seyyed Hadi Hashemi", "Chaowei Xiao", "Wenbo Guo", "Bo Li" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "agent-safety", "tool-use" ], "score": 15, "relevance": "high", "matched_queries": [ "agent-safety" ], "arxiv_id": "2604.04426", "source": "arxiv", "source_id": "arxiv:2604.04426", "pdf_url": "https://arxiv.org/pdf/2604.04426", "primary_query": "agent-safety" }, { "id": "2604.03098", "title": "Co-Evolution of Policy and Internal Reward for Language Agents", "url": "https://arxiv.org/abs/2604.03098", "published": "2026-04-03", "updated": "2026-04-03", "authors": [ "Xinyu Wang", "Hanwei Wu", "Jingwei Song", "Shuyuan Zhang", "Jiayi Zhang", "Fanqi Kong", "Tung Sum Thomas Kwok", "Xiao-Wen Chang", "Yuyu Luo", "Chenglin Wu", "Bang Liu" ], "categories": [ "cs.LG", "cs.AI", "cs.CL" ], "topics": [ "agent-evaluation", "computer-use", "planning", "tool-use" ], "score": 15, "relevance": "high", "matched_queries": [ "language-agent" ], "arxiv_id": "2604.03098", "source": "arxiv", "source_id": "arxiv:2604.03098", "pdf_url": "https://arxiv.org/pdf/2604.03098", "primary_query": "language-agent" }, { "id": "2603.15309", "title": "CCTU: A Benchmark for Tool Use under Complex Constraints", "url": "https://arxiv.org/abs/2603.15309", "published": "2026-03-16", "updated": "2026-03-16", "authors": [ "Junjie Ye", "Guoqiang Zhang", "Wenjie Fu", "Tao Gui", "Qi Zhang", "Xuanjing Huang" ], "categories": [ "cs.CL", "cs.AI" ], "topics": [ "agent-evaluation", "rag", "tool-use" ], "score": 15, "relevance": "high", "matched_queries": [ "function-calling" ], "arxiv_id": "2603.15309", "source": "arxiv", "source_id": "arxiv:2603.15309", "pdf_url": "https://arxiv.org/pdf/2603.15309", "primary_query": "function-calling" }, { "id": "2603.11890", "title": "QUARE: Quality-Aware Requirements Analysis through Multi-Agent Dialectical Negotiation", "url": "https://arxiv.org/abs/2603.11890", "published": "2026-03-12", "updated": "2026-06-05", "authors": [ "Haowei Cheng", "Milhan Kim", "Foutse Khomh", "Teeradaj Racharak", "Nobukazu Yoshioka", "Naoyasu Ubayashi", "Hironori Washizaki" ], "categories": [ "cs.SE" ], "topics": [ "agent-evaluation", "agent-safety", "multi-agent", "rag" ], "score": 15, "relevance": "high", "matched_queries": [ "agent-safety" ], "arxiv_id": "2603.11890", "source": "arxiv", "source_id": "arxiv:2603.11890", "pdf_url": "https://arxiv.org/pdf/2603.11890", "primary_query": "agent-safety" }, { "id": "2603.07557", "title": "AgentRaft: Automated Detection of Data Over-Exposure in LLM Agents", "url": "https://arxiv.org/abs/2603.07557", "published": "2026-03-08", "updated": "2026-03-08", "authors": [ "Yixi Lin", "Jiangrong Wu", "Yuhong Nan", "Xueqiang Wang", "Xinyuan Zhang", "Zibin Zheng" ], "categories": [ "cs.SE" ], "topics": [ "agent-evaluation", "agent-safety", "rag", "reasoning", "tool-use" ], "score": 15, "relevance": "high", "matched_queries": [ "function-calling" ], "arxiv_id": "2603.07557", "source": "arxiv", "source_id": "arxiv:2603.07557", "pdf_url": "https://arxiv.org/pdf/2603.07557", "primary_query": "function-calling" }, { "id": "2603.05578", "title": "Tool-Genesis: A Task-Driven Tool Creation Benchmark for Self-Evolving Language Agent", "url": "https://arxiv.org/abs/2603.05578", "published": "2026-03-05", "updated": "2026-03-05", "authors": [ "Bowei Xia", "Mengkang Hu", "Shijian Wang", "Jiarui Jin", "Wenxiang Jiao", "Yuan Lu", "Kexin Li", "Ping Luo" ], "categories": [ "cs.SE", "cs.AI" ], "topics": [ "agent-evaluation", "computer-use", "tool-use" ], "score": 15, "relevance": "high", "matched_queries": [ "language-agent" ], "arxiv_id": "2603.05578", "source": "arxiv", "source_id": "arxiv:2603.05578", "pdf_url": "https://arxiv.org/pdf/2603.05578", "primary_query": "language-agent" }, { "id": "2603.01712", "title": "FT-Dojo: Towards Autonomous LLM Fine-Tuning with Language Agents", "url": "https://arxiv.org/abs/2603.01712", "published": "2026-03-02", "updated": "2026-05-20", "authors": [ "Qizheng Li", "Yifei Zhang", "Xiao Yang", "Xu Yang", "Zhuo Wang", "Weiqing Liu", "Jiang Bian" ], "categories": [ "cs.AI", "cs.LG" ], "topics": [ "agent-evaluation", "coding-agent", "planning" ], "score": 15, "relevance": "high", "matched_queries": [ "language-agent" ], "arxiv_id": "2603.01712", "source": "arxiv", "source_id": "arxiv:2603.01712", "pdf_url": "https://arxiv.org/pdf/2603.01712", "primary_query": "language-agent" }, { "id": "2602.13379", "title": "Unsafer in Many Turns: Benchmarking and Defending Multi-Turn Safety Risks in Tool-Using Agents", "url": "https://arxiv.org/abs/2602.13379", "published": "2026-02-13", "updated": "2026-06-10", "authors": [ "Xu Li", "Simon Yu", "Minzhou Pan", "Yiyou Sun", "Bo Li", "Dawn Song", "Xue Lin", "Weiyan Shi" ], "categories": [ "cs.CR", "cs.AI", "cs.CL", "cs.LG", "cs.SE" ], "topics": [ "agent-evaluation", "agent-safety", "rag", "tool-use" ], "score": 15, "relevance": "high", "matched_queries": [ "agent-safety" ], "arxiv_id": "2602.13379", "source": "arxiv", "source_id": "arxiv:2602.13379", "pdf_url": "https://arxiv.org/pdf/2602.13379", "primary_query": "agent-safety" }, { "id": "2602.11749", "title": "AIR: Improving Agent Safety through Incident Response", "url": "https://arxiv.org/abs/2602.11749", "published": "2026-02-12", "updated": "2026-06-20", "authors": [ "Zibo Xiao", "Jun Sun", "Junjie Chen" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "agent-safety", "computer-use", "tool-use" ], "score": 15, "relevance": "high", "matched_queries": [ "agent-safety" ], "arxiv_id": "2602.11749", "source": "arxiv", "source_id": "arxiv:2602.11749", "pdf_url": "https://arxiv.org/pdf/2602.11749", "primary_query": "agent-safety" }, { "id": "2602.18456", "title": "Beyond single-channel agentic benchmarking", "url": "https://arxiv.org/abs/2602.18456", "published": "2026-02-05", "updated": "2026-02-05", "authors": [ "Nelu D. Radpour" ], "categories": [ "cs.CY", "cs.AI", "cs.HC" ], "topics": [ "agent-evaluation", "agent-safety" ], "score": 15, "relevance": "high", "matched_queries": [ "agent-safety" ], "arxiv_id": "2602.18456", "source": "arxiv", "source_id": "arxiv:2602.18456", "pdf_url": "https://arxiv.org/pdf/2602.18456", "primary_query": "agent-safety" }, { "id": "2601.14652", "title": "MAS-Orchestra: Understanding and Improving Multi-Agent Reasoning Through Holistic Orchestration and Controlled Benchmarks", "url": "https://arxiv.org/abs/2601.14652", "published": "2026-01-21", "updated": "2026-05-21", "authors": [ "Zixuan Ke", "Yifei Ming", "Austin Xu", "Ryan Chin", "Xuan-Phi Nguyen", "Prathyusha Jwalapuram", "Jiayu Wang", "Semih Yavuz", "Caiming Xiong", "Shafiq Joty" ], "categories": [ "cs.AI", "cs.CL", "cs.MA" ], "topics": [ "agent-evaluation", "computer-use", "multi-agent", "reasoning" ], "score": 15, "relevance": "high", "matched_queries": [ "function-calling" ], "arxiv_id": "2601.14652", "source": "arxiv", "source_id": "arxiv:2601.14652", "pdf_url": "https://arxiv.org/pdf/2601.14652", "primary_query": "function-calling" }, { "id": "2512.23647", "title": "Nested Browser-Use Learning for Agentic Information Seeking", "url": "https://arxiv.org/abs/2512.23647", "published": "2025-12-29", "updated": "2025-12-29", "authors": [ "Baixuan Li", "Jialong Wu", "Wenbiao Yin", "Kuan Li", "Zhongwang Zhang", "Huifeng Yin", "Zhengwei Tao", "Liwen Zhang", "Pengjun Xie", "Jingren Zhou", "Yong Jiang" ], "categories": [ "cs.CL", "cs.AI", "cs.IR", "cs.MA" ], "topics": [ "agent-evaluation", "computer-use", "rag", "reasoning", "tool-use" ], "score": 15, "relevance": "high", "matched_queries": [ "function-calling" ], "arxiv_id": "2512.23647", "source": "arxiv", "source_id": "arxiv:2512.23647", "pdf_url": "https://arxiv.org/pdf/2512.23647", "primary_query": "function-calling" }, { "id": "2512.23611", "title": "Close the Loop: Synthesizing Infinite Tool-Use Data via Multi-Agent Role-Playing", "url": "https://arxiv.org/abs/2512.23611", "published": "2025-12-29", "updated": "2025-12-29", "authors": [ "Yuwen Li", "Wei Zhang", "Zelong Huang", "Mason Yang", "Jiajun Wu", "Shawn Guo", "Huahao Hu", "Lingyi Sun", "Jian Yang", "Mingjie Tang", "Byran Dai" ], "categories": [ "cs.CL" ], "topics": [ "agent-evaluation", "multi-agent", "rag", "tool-use", "workflow-agent" ], "score": 15, "relevance": "high", "matched_queries": [ "function-calling" ], "arxiv_id": "2512.23611", "source": "arxiv", "source_id": "arxiv:2512.23611", "pdf_url": "https://arxiv.org/pdf/2512.23611", "primary_query": "function-calling" }, { "id": "2512.02605", "title": "IACT: A Self-Organizing Recursive Model for General AI Agents: A Technical White Paper on the Architecture Behind kragent.ai", "url": "https://arxiv.org/abs/2512.02605", "published": "2025-12-02", "updated": "2025-12-02", "authors": [ "Pengju Lu" ], "categories": [ "cs.AI", "cs.MA", "cs.SE" ], "topics": [ "agent-evaluation", "computer-use", "memory", "rag", "tool-use", "workflow-agent" ], "score": 15, "relevance": "high", "matched_queries": [ "function-calling" ], "arxiv_id": "2512.02605", "source": "arxiv", "source_id": "arxiv:2512.02605", "pdf_url": "https://arxiv.org/pdf/2512.02605", "primary_query": "function-calling" }, { "id": "2510.14548", "title": "LLM Agents Beyond Utility: An Open-Ended Perspective", "url": "https://arxiv.org/abs/2510.14548", "published": "2025-10-16", "updated": "2025-10-16", "authors": [ "Asen Nachkov", "Xi Wang", "Luc Van Gool" ], "categories": [ "cs.AI" ], "topics": [ "memory", "planning", "reasoning", "tool-use" ], "score": 15, "relevance": "high", "matched_queries": [ "function-calling" ], "arxiv_id": "2510.14548", "source": "arxiv", "source_id": "arxiv:2510.14548", "pdf_url": "https://arxiv.org/pdf/2510.14548", "primary_query": "function-calling" }, { "id": "2509.26553", "title": "Towards Reliable Benchmarking: A Contamination Free, Controllable Evaluation Framework for Multi-step LLM Function Calling", "url": "https://arxiv.org/abs/2509.26553", "published": "2025-09-30", "updated": "2026-02-06", "authors": [ "Seiji Maekawa", "Jackson Hassell", "Pouya Pezeshkpour", "Tom Mitchell", "Estevam Hruschka" ], "categories": [ "cs.CL", "cs.PL" ], "topics": [ "agent-evaluation", "reasoning", "tool-use" ], "score": 15, "relevance": "high", "matched_queries": [ "function-calling" ], "arxiv_id": "2509.26553", "source": "arxiv", "source_id": "arxiv:2509.26553", "pdf_url": "https://arxiv.org/pdf/2509.26553", "primary_query": "function-calling" }, { "id": "2509.14477", "title": "Ticket-Bench: A Kickoff for Multilingual and Regionalized Agent Evaluation", "url": "https://arxiv.org/abs/2509.14477", "published": "2025-09-17", "updated": "2025-09-17", "authors": [ "Thales Sales Almeida", "João Guilherme Alves Santos", "Thiago Laitz", "Giovana Kerche Bonás" ], "categories": [ "cs.CL" ], "topics": [ "agent-evaluation", "computer-use", "reasoning" ], "score": 15, "relevance": "high", "matched_queries": [ "function-calling" ], "arxiv_id": "2509.14477", "source": "arxiv", "source_id": "arxiv:2509.14477", "pdf_url": "https://arxiv.org/pdf/2509.14477", "primary_query": "function-calling" }, { "id": "2509.02444", "title": "AppCopilot: Toward General, Accurate, Long-Horizon, and Efficient Mobile Agent", "url": "https://arxiv.org/abs/2509.02444", "published": "2025-09-02", "updated": "2025-10-17", "authors": [ "Jingru Fan", "Yufan Dang", "Jingyao Wu", "Huatao Li", "Runde Yang", "Xiyuan Yang", "Yuheng Wang", "Chen Qian" ], "categories": [ "cs.AI", "cs.CL", "cs.CV", "cs.HC" ], "topics": [ "computer-use", "memory", "multi-agent", "planning", "reasoning", "tool-use" ], "score": 15, "relevance": "high", "matched_queries": [ "function-calling" ], "arxiv_id": "2509.02444", "source": "arxiv", "source_id": "arxiv:2509.02444", "pdf_url": "https://arxiv.org/pdf/2509.02444", "primary_query": "function-calling" }, { "id": "2607.06140", "title": "CurateEvo: Data-Curation Evolving for Agentic Post-Training", "url": "https://arxiv.org/abs/2607.06140", "published": "2026-07-07", "updated": "2026-07-07", "authors": [ "Dingzirui Wang", "Xuanliang Zhang", "Keyan Xu", "Qingfu Zhu", "Wanxiang Che" ], "categories": [ "cs.CL" ], "topics": [ "memory", "planning", "rag" ], "score": 14, "relevance": "high", "matched_queries": [ "llm-agent" ], "arxiv_id": "2607.06140", "source": "arxiv", "source_id": "arxiv:2607.06140", "pdf_url": "https://arxiv.org/pdf/2607.06140", "primary_query": "llm-agent" }, { "id": "2607.06452", "title": "From Voting to Agent Collaboration: Answer-Type-Aware LLM Pipelines for BioASQ 14b", "url": "https://arxiv.org/abs/2607.06452", "published": "2026-07-07", "updated": "2026-07-07", "authors": [ "Taeyun Roh", "Eunha Lee", "Wonjune Jang", "Sohyun Chung", "Junha Jung", "Jaewoo Kang" ], "categories": [ "cs.CL", "cs.AI" ], "topics": [ "agent-evaluation", "multi-agent", "rag", "reasoning", "tool-use" ], "score": 14, "relevance": "high", "matched_queries": [ "multi-agent-llm" ], "arxiv_id": "2607.06452", "source": "arxiv", "source_id": "arxiv:2607.06452", "pdf_url": "https://arxiv.org/pdf/2607.06452", "primary_query": "multi-agent-llm" }, { "id": "2607.05772", "title": "Detecting Vulnerability-Inducing Commits via Multi-Stage Reasoning with LLM-Based Agents", "url": "https://arxiv.org/abs/2607.05772", "published": "2026-07-07", "updated": "2026-07-07", "authors": [ "Liyou Chen", "Hailong Sun", "Xiang Gao", "Yue Pan" ], "categories": [ "cs.SE" ], "topics": [ "agent-safety", "multi-agent", "rag", "reasoning", "tool-use", "workflow-agent" ], "score": 14, "relevance": "high", "matched_queries": [ "multi-agent-llm" ], "arxiv_id": "2607.05772", "source": "arxiv", "source_id": "arxiv:2607.05772", "pdf_url": "https://arxiv.org/pdf/2607.05772", "primary_query": "multi-agent-llm" }, { "id": "2607.05378", "title": "CompactionRL: Reinforcement Learning with Context Compaction for Long-Horizon Agents", "url": "https://arxiv.org/abs/2607.05378", "published": "2026-07-06", "updated": "2026-07-06", "authors": [ "Yujiang Li", "Zhenyu Hou", "Yi Jing", "Jie Tang", "Yuxiao Dong" ], "categories": [ "cs.LG" ], "topics": [ "coding-agent", "planning", "tool-use" ], "score": 14, "relevance": "high", "matched_queries": [ "coding-agent", "llm-agent" ], "arxiv_id": "2607.05378", "source": "arxiv", "source_id": "arxiv:2607.05378", "pdf_url": "https://arxiv.org/pdf/2607.05378", "primary_query": "coding-agent" }, { "id": "2607.05132", "title": "When Agents Lie: Premeditation, Persistence, and Exploitation in Repeated Games", "url": "https://arxiv.org/abs/2607.05132", "published": "2026-07-06", "updated": "2026-07-07", "authors": [ "Jerick Shi", "Terry Jingcheng Zhang", "Bernhard Schölkopf", "Vincent Conitzer", "Zhijing Jin" ], "categories": [ "cs.CY", "cs.CL" ], "topics": [ "agent-evaluation", "agent-safety", "planning", "tool-use" ], "score": 14, "relevance": "high", "matched_queries": [ "autonomous-agent-llm", "llm-agent", "planning-agent" ], "arxiv_id": "2607.05132", "source": "arxiv", "source_id": "arxiv:2607.05132", "pdf_url": "https://arxiv.org/pdf/2607.05132", "primary_query": "autonomous-agent-llm" }, { "id": "2607.04963", "title": "STAPO: Selective Trajectory-Aware Policy Optimization for LLM Agent Training", "url": "https://arxiv.org/abs/2607.04963", "published": "2026-07-06", "updated": "2026-07-06", "authors": [ "Qiuyi Qi", "Tian Liang", "Mutian Bao", "Jinjian Zhang", "Dongnan Liu", "Wei Zhou", "Linjian Mo", "Ming Kong", "Jie Liu", "Feng Zhang", "Qiang Zhu" ], "categories": [ "cs.AI" ], "topics": [ "planning", "rag", "tool-use" ], "score": 14, "relevance": "high", "matched_queries": [ "llm-agent" ], "arxiv_id": "2607.04963", "source": "arxiv", "source_id": "arxiv:2607.04963", "pdf_url": "https://arxiv.org/pdf/2607.04963", "primary_query": "llm-agent" }, { "id": "2607.05677", "title": "From Conversation to Contribution: Characterizing Coding Agent in Open-Source Software", "url": "https://arxiv.org/abs/2607.05677", "published": "2026-07-06", "updated": "2026-07-06", "authors": [ "Zihan Fang", "Yueke Zhang", "Ningzhi Tang", "Collin McMillan", "Toby Jia-Jun Li", "Yu Huang" ], "categories": [ "cs.SE", "cs.HC" ], "topics": [ "agent-evaluation", "coding-agent", "computer-use", "multi-agent", "tool-use", "workflow-agent" ], "score": 14, "relevance": "high", "matched_queries": [ "coding-agent" ], "arxiv_id": "2607.05677", "source": "arxiv", "source_id": "arxiv:2607.05677", "pdf_url": "https://arxiv.org/pdf/2607.05677", "primary_query": "coding-agent" }, { "id": "2607.05188", "title": "Latent Programming Horizons in Coding Agents", "url": "https://arxiv.org/abs/2607.05188", "published": "2026-07-06", "updated": "2026-07-06", "authors": [ "André Silva", "Han Tu", "Martin Monperrus" ], "categories": [ "cs.LG", "cs.SE" ], "topics": [ "agent-evaluation", "coding-agent", "reasoning" ], "score": 14, "relevance": "high", "matched_queries": [ "coding-agent" ], "arxiv_id": "2607.05188", "source": "arxiv", "source_id": "arxiv:2607.05188", "pdf_url": "https://arxiv.org/pdf/2607.05188", "primary_query": "coding-agent" }, { "id": "2607.04623", "title": "Can LLMs Really Recover Microservice Failures? A Recovery-Aware Evaluation of Diagnosis-to-Action Reasoning", "url": "https://arxiv.org/abs/2607.04623", "published": "2026-07-06", "updated": "2026-07-06", "authors": [ "Jiaxing Qi", "Zhongzhi Luan", "Hongyu Zhang", "Shaohan Huang", "Carol Fung", "Yongxin Tong", "Hailong Yang", "Depei Qian" ], "categories": [ "cs.SE", "cs.DC" ], "topics": [ "agent-evaluation", "planning", "rag", "reasoning", "tool-use" ], "score": 14, "relevance": "high", "matched_queries": [ "rag-agent" ], "arxiv_id": "2607.04623", "source": "arxiv", "source_id": "arxiv:2607.04623", "pdf_url": "https://arxiv.org/pdf/2607.04623", "primary_query": "rag-agent" }, { "id": "2607.04470", "title": "Regime-Conditional Stabilisation of LLM-Augmented Cooperative Multi-Agent Reinforcement Learning", "url": "https://arxiv.org/abs/2607.04470", "published": "2026-07-05", "updated": "2026-07-05", "authors": [ "Faid Keddouri", "Sohaib Houhou", "Aissa Boulmerka", "Nadir Farhi" ], "categories": [ "cs.LG", "cs.AI", "math.OC" ], "topics": [ "agent-evaluation", "multi-agent", "rag", "tool-use" ], "score": 14, "relevance": "high", "matched_queries": [ "multi-agent-llm" ], "arxiv_id": "2607.04470", "source": "arxiv", "source_id": "arxiv:2607.04470", "pdf_url": "https://arxiv.org/pdf/2607.04470", "primary_query": "multi-agent-llm" }, { "id": "2607.03853", "title": "CogRad: A Cognitively-Inspired Multi-Agent Framework for Radiology Report Generation", "url": "https://arxiv.org/abs/2607.03853", "published": "2026-07-04", "updated": "2026-07-04", "authors": [ "Saif Ur Rehman Khan", "Hasaan Maqsood", "Sebastian Vollmer", "Andreas Dengel", "Muhammad Nabeel Asim" ], "categories": [ "cs.CV" ], "topics": [ "agent-evaluation", "multi-agent", "rag", "reasoning" ], "score": 14, "relevance": "high", "matched_queries": [ "multi-agent-llm" ], "arxiv_id": "2607.03853", "source": "arxiv", "source_id": "arxiv:2607.03853", "pdf_url": "https://arxiv.org/pdf/2607.03853", "primary_query": "multi-agent-llm" }, { "id": "2607.02879", "title": "MedCalc-Pro: Solving Complex Medical Calculations with LLM Agents", "url": "https://arxiv.org/abs/2607.02879", "published": "2026-07-03", "updated": "2026-07-03", "authors": [ "Siran Zhao", "Ruihui Hou", "Ziyue Huai", "Chennuo Zhang", "Tong Ruan" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "tool-use" ], "score": 14, "relevance": "high", "matched_queries": [ "llm-agent" ], "arxiv_id": "2607.02879", "source": "arxiv", "source_id": "arxiv:2607.02879", "pdf_url": "https://arxiv.org/pdf/2607.02879", "primary_query": "llm-agent" }, { "id": "2607.03423", "title": "Securing Multi-Tool AI Agent Chains With Dynamic, Real-Time Compositional Policies", "url": "https://arxiv.org/abs/2607.03423", "published": "2026-07-03", "updated": "2026-07-03", "authors": [ "Chris Schneider", "Kriti Faujdar", "Philipp Schoenegger", "Ben Bariach" ], "categories": [ "cs.CR", "cs.AI" ], "topics": [ "agent-evaluation", "agent-safety", "coding-agent", "tool-use" ], "score": 14, "relevance": "high", "matched_queries": [ "ai-agent", "coding-agent" ], "arxiv_id": "2607.03423", "source": "arxiv", "source_id": "arxiv:2607.03423", "pdf_url": "https://arxiv.org/pdf/2607.03423", "primary_query": "ai-agent" }, { "id": "2607.03162", "title": "APeB: Benchmarking Personalization Ability of Large Language Model Agents", "url": "https://arxiv.org/abs/2607.03162", "published": "2026-07-03", "updated": "2026-07-03", "authors": [ "Garry Yang", "Zizhe Chen", "Xinru Chen", "Yongqiang Chen", "Jianxiang Wang", "Deyu Zou", "Linyi Ding", "Jialiang Wu", "Yunzhong He", "Yu Gong", "James Cheng", "Huaixiao Tou" ], "categories": [ "cs.AI", "cs.HC" ], "topics": [ "agent-evaluation", "tool-use", "workflow-agent" ], "score": 14, "relevance": "high", "matched_queries": [ "agentic-ai" ], "arxiv_id": "2607.03162", "source": "arxiv", "source_id": "arxiv:2607.03162", "pdf_url": "https://arxiv.org/pdf/2607.03162", "primary_query": "agentic-ai" }, { "id": "2607.02882", "title": "Diagnosis-Driven Automatic Repair for Agentic Workflow via Symbolic Inference", "url": "https://arxiv.org/abs/2607.02882", "published": "2026-07-03", "updated": "2026-07-03", "authors": [ "Xuyan Ma", "Yawen Wang", "Junjie Wang", "Xiaofei Xie", "Boyu Wu", "Mingyang Li", "Dandan Wang", "Qing Wang" ], "categories": [ "cs.SE" ], "topics": [ "agent-evaluation", "planning", "tool-use", "workflow-agent" ], "score": 14, "relevance": "high", "matched_queries": [ "agentic-ai" ], "arxiv_id": "2607.02882", "source": "arxiv", "source_id": "arxiv:2607.02882", "pdf_url": "https://arxiv.org/pdf/2607.02882", "primary_query": "agentic-ai" }, { "id": "2607.03628", "title": "Swarm-Driven Multi-Agent Reasoning for Smart City Security", "url": "https://arxiv.org/abs/2607.03628", "published": "2026-07-03", "updated": "2026-07-03", "authors": [ "Saeid Jamshidi", "Kawser Wazed Nafi", "Carol Fung", "Foutse Khomh" ], "categories": [ "cs.CR", "cs.MA" ], "topics": [ "agent-safety", "multi-agent", "reasoning", "tool-use" ], "score": 14, "relevance": "high", "matched_queries": [ "multi-agent-llm" ], "arxiv_id": "2607.03628", "source": "arxiv", "source_id": "arxiv:2607.03628", "pdf_url": "https://arxiv.org/pdf/2607.03628", "primary_query": "multi-agent-llm" }, { "id": "2607.01600", "title": "BOUNDARY_SYNC: Measuring Communication-Induced Representational Coupling in Multi-Agent LLM Systems", "url": "https://arxiv.org/abs/2607.01600", "published": "2026-07-02", "updated": "2026-07-02", "authors": [ "Zewen Liu" ], "categories": [ "cs.LG", "cs.CL" ], "topics": [ "multi-agent", "tool-use" ], "score": 14, "relevance": "high", "matched_queries": [ "llm-agent", "multi-agent-llm" ], "arxiv_id": "2607.01600", "source": "arxiv", "source_id": "arxiv:2607.01600", "pdf_url": "https://arxiv.org/pdf/2607.01600", "primary_query": "llm-agent" }, { "id": "2607.02210", "title": "Criticality-Based Guard Rail Validation for AI Agent Decisions in Autonomous Telecom Networks", "url": "https://arxiv.org/abs/2607.02210", "published": "2026-07-02", "updated": "2026-07-02", "authors": [ "Ravi Kant Sharma" ], "categories": [ "cs.AI", "cs.NI" ], "topics": [ "agent-evaluation", "agent-safety", "multi-agent", "rag", "tool-use" ], "score": 14, "relevance": "high", "matched_queries": [ "ai-agent" ], "arxiv_id": "2607.02210", "source": "arxiv", "source_id": "arxiv:2607.02210", "pdf_url": "https://arxiv.org/pdf/2607.02210", "primary_query": "ai-agent" }, { "id": "2607.01661", "title": "Diverse Evidence, Better Forecasts: Multi-Agent Deliberation Under Information Asymmetry", "url": "https://arxiv.org/abs/2607.01661", "published": "2026-07-02", "updated": "2026-07-02", "authors": [ "Yuante Li", "Yicheng Tao", "Kate Zhang", "Taozhi Wang", "Gefei Gu", "Yaxin Zhou" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "multi-agent", "reasoning" ], "score": 14, "relevance": "high", "matched_queries": [ "multi-agent-llm" ], "arxiv_id": "2607.01661", "source": "arxiv", "source_id": "arxiv:2607.01661", "pdf_url": "https://arxiv.org/pdf/2607.01661", "primary_query": "multi-agent-llm" }, { "id": "2607.00692", "title": "Self-GC: Self-Governing Context for Long-Horizon LLM Agents", "url": "https://arxiv.org/abs/2607.00692", "published": "2026-07-01", "updated": "2026-07-01", "authors": [ "Xubin Hao", "Hongjin Meng", "Xin Yin", "Jiawei Zhu", "Chenpeng Cao" ], "categories": [ "cs.AI" ], "topics": [ "planning", "rag", "tool-use" ], "score": 14, "relevance": "high", "matched_queries": [ "llm-agent", "planning-agent" ], "arxiv_id": "2607.00692", "source": "arxiv", "source_id": "arxiv:2607.00692", "pdf_url": "https://arxiv.org/pdf/2607.00692", "primary_query": "llm-agent" }, { "id": "2607.00297", "title": "EPC: A Standardized Protocol for Measuring Evaluator Preference Dynamics in LLM Agent Systems", "url": "https://arxiv.org/abs/2607.00297", "published": "2026-07-01", "updated": "2026-07-01", "authors": [ "Zewen Liu" ], "categories": [ "cs.LG", "cs.CL" ], "topics": [ "agent-evaluation", "coding-agent", "computer-use", "tool-use" ], "score": 14, "relevance": "high", "matched_queries": [ "llm-agent" ], "arxiv_id": "2607.00297", "source": "arxiv", "source_id": "arxiv:2607.00297", "pdf_url": "https://arxiv.org/pdf/2607.00297", "primary_query": "llm-agent" }, { "id": "2607.01523", "title": "Multi-Head Recurrent Memory Agents", "url": "https://arxiv.org/abs/2607.01523", "published": "2026-07-01", "updated": "2026-07-01", "authors": [ "Jiatong Li", "Samuel Yeh", "Sharon Li" ], "categories": [ "cs.LG", "cs.AI", "cs.CL" ], "topics": [ "agent-evaluation", "agent-safety", "memory" ], "score": 14, "relevance": "high", "matched_queries": [ "agent-memory" ], "arxiv_id": "2607.01523", "source": "arxiv", "source_id": "arxiv:2607.01523", "pdf_url": "https://arxiv.org/pdf/2607.01523", "primary_query": "agent-memory" }, { "id": "2607.01211", "title": "Are Performance-Optimization Benchmarks Reliably Measuring Coding Agents?", "url": "https://arxiv.org/abs/2607.01211", "published": "2026-07-01", "updated": "2026-07-01", "authors": [ "Zhi Chen", "Zhensu Sun", "Yuling Shi", "David Lo", "Lingxiao Jiang" ], "categories": [ "cs.SE", "cs.AI" ], "topics": [ "agent-evaluation", "coding-agent", "rag" ], "score": 14, "relevance": "high", "matched_queries": [ "coding-agent" ], "arxiv_id": "2607.01211", "source": "arxiv", "source_id": "arxiv:2607.01211", "pdf_url": "https://arxiv.org/pdf/2607.01211", "primary_query": "coding-agent" }, { "id": "2607.00918", "title": "From Personas to Plot: Character-Grounded Multi-Agent Story Generation for Long-Form Narratives", "url": "https://arxiv.org/abs/2607.00918", "published": "2026-07-01", "updated": "2026-07-01", "authors": [ "Aayush Aluru", "Chloe Ho", "Muhammad Hammouri", "Kerry Luo", "Myra Malik", "Ryan Lagasse", "Arjun Bahuguna", "Vasu Sharma" ], "categories": [ "cs.CL", "cs.AI", "cs.MA" ], "topics": [ "agent-evaluation", "multi-agent", "tool-use" ], "score": 14, "relevance": "high", "matched_queries": [ "multi-agent-llm" ], "arxiv_id": "2607.00918", "source": "arxiv", "source_id": "arxiv:2607.00918", "pdf_url": "https://arxiv.org/pdf/2607.00918", "primary_query": "multi-agent-llm" }, { "id": "2607.00604", "title": "Vehicle Routing Problem Meets Large Language Models: An Overview and Perspectives", "url": "https://arxiv.org/abs/2607.00604", "published": "2026-07-01", "updated": "2026-07-01", "authors": [ "Xianchao Xiu", "Chong Shen", "Yanjiao Zhu", "Wanquan Liu" ], "categories": [ "math.OC" ], "topics": [ "agent-evaluation", "multi-agent", "planning", "reasoning", "tool-use", "workflow-agent" ], "score": 14, "relevance": "high", "matched_queries": [ "multi-agent-llm" ], "arxiv_id": "2607.00604", "source": "arxiv", "source_id": "arxiv:2607.00604", "pdf_url": "https://arxiv.org/pdf/2607.00604", "primary_query": "multi-agent-llm" }, { "id": "2607.00972", "title": "Bayesian Uncertainty Propagation for Agentic RAG Pipelines: A Proof-of-Concept Study on Multi-Hop Question Answering", "url": "https://arxiv.org/abs/2607.00972", "published": "2026-07-01", "updated": "2026-07-01", "authors": [ "Louis Donaldson", "Connor Walker", "Koorosh Aslansefat", "Yiannis Papadopoulos" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "planning", "rag", "reasoning", "workflow-agent" ], "score": 14, "relevance": "high", "matched_queries": [ "rag-agent" ], "arxiv_id": "2607.00972", "source": "arxiv", "source_id": "arxiv:2607.00972", "pdf_url": "https://arxiv.org/pdf/2607.00972", "primary_query": "rag-agent" }, { "id": "2607.00422", "title": "KidnapRAG: A Black-Box Attack for Hijacking Reasoning in Agentic Retrieval-Augmented Generation Systems", "url": "https://arxiv.org/abs/2607.00422", "published": "2026-07-01", "updated": "2026-07-01", "authors": [ "Chanwoo Choi", "Euntae Kim", "Kyuho Lee", "Youngsam Chun", "Jinhee Jeong", "Eunmi Kim", "Myunggyo Oh", "Junseo Jang", "Buru Chang" ], "categories": [ "cs.CR" ], "topics": [ "agent-evaluation", "rag", "reasoning" ], "score": 14, "relevance": "high", "matched_queries": [ "rag-agent" ], "arxiv_id": "2607.00422", "source": "arxiv", "source_id": "arxiv:2607.00422", "pdf_url": "https://arxiv.org/pdf/2607.00422", "primary_query": "rag-agent" }, { "id": "2606.31635", "title": "A Tutorial on Autonomous Fault-Tolerant Control Using Knowledge-Grounded LLM Agents", "url": "https://arxiv.org/abs/2606.31635", "published": "2026-06-30", "updated": "2026-06-30", "authors": [ "Javal Vyas", "Milapji Singh Gill", "Artan Markaj", "Felix Gehlhoff", "Mehmet Mercangöz" ], "categories": [ "eess.SY", "cs.AI", "cs.MA" ], "topics": [ "agent-safety", "planning", "tool-use", "world-model" ], "score": 14, "relevance": "high", "matched_queries": [ "llm-agent" ], "arxiv_id": "2606.31635", "source": "arxiv", "source_id": "arxiv:2606.31635", "pdf_url": "https://arxiv.org/pdf/2606.31635", "primary_query": "llm-agent" }, { "id": "2606.31227", "title": "Securing the AI Agent: A Unified Framework for Multi-Layer Agent Red Teaming", "url": "https://arxiv.org/abs/2606.31227", "published": "2026-06-30", "updated": "2026-06-30", "authors": [ "Yong Yang", "Xing Zheng", "Huiyu Wu", "Huangsheng Cheng", "Xiaorong Shi", "Jing Guo", "Bo Yang", "Yi Zhou", "Xiangfan Wu", "Zonghao Ying" ], "categories": [ "cs.CR" ], "topics": [ "agent-safety", "tool-use" ], "score": 14, "relevance": "high", "matched_queries": [ "agent-safety", "ai-agent" ], "arxiv_id": "2606.31227", "source": "arxiv", "source_id": "arxiv:2606.31227", "pdf_url": "https://arxiv.org/pdf/2606.31227", "primary_query": "agent-safety" }, { "id": "2607.00255", "title": "SLM, LLM or Agentic AI? Toward Intelligent UAV-Enabled WPT Systems in Low-Altitude Economy Networks", "url": "https://arxiv.org/abs/2607.00255", "published": "2026-06-30", "updated": "2026-06-30", "authors": [ "Feibo Jiang", "Li Dong", "Lei Mao", "Kezhi Wang", "Xianbin Wang", "Abbas Jamalipour" ], "categories": [ "cs.IT" ], "topics": [ "agent-evaluation", "reasoning", "world-model" ], "score": 14, "relevance": "high", "matched_queries": [ "agentic-ai" ], "arxiv_id": "2607.00255", "source": "arxiv", "source_id": "arxiv:2607.00255", "pdf_url": "https://arxiv.org/pdf/2607.00255", "primary_query": "agentic-ai" }, { "id": "2606.31648", "title": "Think in English, Answer in Korean: Efficient Adaptation of Multilingual Tool-Using Agents", "url": "https://arxiv.org/abs/2606.31648", "published": "2026-06-30", "updated": "2026-06-30", "authors": [ "Utsav Garg", "Sungjin Hong", "Jason Jung", "Justin Lee", "Shaan Desai", "Joon Hee Kim", "Anirudh Shrinivason", "Edmond Wen", "Susie Park" ], "categories": [ "cs.AI", "cs.LG" ], "topics": [ "memory", "multi-agent", "reasoning", "tool-use", "workflow-agent" ], "score": 14, "relevance": "high", "matched_queries": [ "agentic-ai", "function-calling", "tool-use" ], "arxiv_id": "2606.31648", "source": "arxiv", "source_id": "arxiv:2606.31648", "pdf_url": "https://arxiv.org/pdf/2606.31648", "primary_query": "agentic-ai" }, { "id": "2607.02577", "title": "Benchmarking the Benchmarks: A Validity Audit of Tool-Calling Evaluation", "url": "https://arxiv.org/abs/2607.02577", "published": "2026-06-30", "updated": "2026-06-30", "authors": [ "Vishvesh Bhat", "Jay Vaghasiya", "Muhammad Ahmed Mohsin", "Asad Aali" ], "categories": [ "cs.SE" ], "topics": [ "agent-evaluation", "agent-safety", "tool-use" ], "score": 14, "relevance": "high", "matched_queries": [ "tool-use" ], "arxiv_id": "2607.02577", "source": "arxiv", "source_id": "arxiv:2607.02577", "pdf_url": "https://arxiv.org/pdf/2607.02577", "primary_query": "tool-use" }, { "id": "2606.31314", "title": "A Novel Method for Differential-Algebraic Dynamic Model Discovery in Power Systems: An LLM-Based Multi-Agent Collaborative Framework", "url": "https://arxiv.org/abs/2606.31314", "published": "2026-06-30", "updated": "2026-06-30", "authors": [ "Xinming Wang", "Fan Tang", "Yingli Wei", "Yakun He", "Zhe Liu", "Ping Jiang", "Haoyu Wu", "Zihan Guo", "Chao Shen" ], "categories": [ "eess.SY" ], "topics": [ "agent-evaluation", "computer-use", "multi-agent" ], "score": 14, "relevance": "high", "matched_queries": [ "multi-agent-llm" ], "arxiv_id": "2606.31314", "source": "arxiv", "source_id": "arxiv:2606.31314", "pdf_url": "https://arxiv.org/pdf/2606.31314", "primary_query": "multi-agent-llm" }, { "id": "2606.30840", "title": "Contrastive Reflection for Iterative Prompt Optimization", "url": "https://arxiv.org/abs/2606.30840", "published": "2026-06-29", "updated": "2026-06-29", "authors": [ "Derek Koh", "Jinghui Mo", "Benjamin H. Le", "Jiening Zhan", "Baofen Zheng", "Kevin Bevis", "Nathaniel C. Owen", "Lauren Elizabeth Charney", "Wenqiong Liu", "Jingwei Wu" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "computer-use", "rag", "reasoning", "workflow-agent" ], "score": 14, "relevance": "high", "matched_queries": [ "ai-agent", "llm-agent" ], "arxiv_id": "2606.30840", "source": "arxiv", "source_id": "arxiv:2606.30840", "pdf_url": "https://arxiv.org/pdf/2606.30840", "primary_query": "ai-agent" }, { "id": "2606.30454", "title": "Collective cooperation without individual fidelity in LLM agents", "url": "https://arxiv.org/abs/2606.30454", "published": "2026-06-29", "updated": "2026-06-29", "authors": [ "Henrique Ferraz de Arruda", "Carlos Gracia Lázaro", "Alberto Aleta", "Yamir Moreno" ], "categories": [ "physics.soc-ph", "cs.AI" ], "topics": [ "agent-evaluation", "tool-use", "world-model" ], "score": 14, "relevance": "high", "matched_queries": [ "llm-agent" ], "arxiv_id": "2606.30454", "source": "arxiv", "source_id": "arxiv:2606.30454", "pdf_url": "https://arxiv.org/pdf/2606.30454", "primary_query": "llm-agent" }, { "id": "2606.30005", "title": "LLM Agents Are Latent Context Managers: Eliciting Self-Managed Context via a Proprioceptive Dashboard", "url": "https://arxiv.org/abs/2606.30005", "published": "2026-06-29", "updated": "2026-07-05", "authors": [ "Binyan Xu", "Haitao Li", "Kehuan Zhang" ], "categories": [ "cs.CL" ], "topics": [ "memory", "planning", "tool-use" ], "score": 14, "relevance": "high", "matched_queries": [ "llm-agent" ], "arxiv_id": "2606.30005", "source": "arxiv", "source_id": "arxiv:2606.30005", "pdf_url": "https://arxiv.org/pdf/2606.30005", "primary_query": "llm-agent" }, { "id": "2606.29762", "title": "Do Recommendation Algorithms Work When Users Are LLM Agents? A Case Study on Moltbook", "url": "https://arxiv.org/abs/2606.29762", "published": "2026-06-29", "updated": "2026-06-29", "authors": [ "Daming Li", "Simeng Han", "Jialu Zhang" ], "categories": [ "cs.IR" ], "topics": [ "agent-evaluation", "rag" ], "score": 14, "relevance": "high", "matched_queries": [ "ai-agent", "llm-agent" ], "arxiv_id": "2606.29762", "source": "arxiv", "source_id": "arxiv:2606.29762", "pdf_url": "https://arxiv.org/pdf/2606.29762", "primary_query": "ai-agent" }, { "id": "2606.30266", "title": "Towards Continual Motion-Language Agents: LoRA Variants for Incremental Motion Understanding and Generation", "url": "https://arxiv.org/abs/2606.30266", "published": "2026-06-29", "updated": "2026-06-29", "authors": [ "Bertram Taetz", "Hugo Albuquerque Cosme da Silva", "Gabriele Bleser-Taetz" ], "categories": [ "cs.LG", "cs.AI" ], "topics": [ "agent-evaluation" ], "score": 14, "relevance": "high", "matched_queries": [ "autonomous-agent-llm", "language-agent" ], "arxiv_id": "2606.30266", "source": "arxiv", "source_id": "arxiv:2606.30266", "pdf_url": "https://arxiv.org/pdf/2606.30266", "primary_query": "autonomous-agent-llm" }, { "id": "2606.30185", "title": "Dynamo: Dynamic Skill-Tool Evolution for Vision-Language Agents", "url": "https://arxiv.org/abs/2606.30185", "published": "2026-06-29", "updated": "2026-06-29", "authors": [ "Yutao Sun", "Yanting Miao", "Hao-Xuan Ma", "Mengyu Zhou", "Mingshuai Chen", "Tiancheng Zhao", "Dexin Wang", "Lei Lv", "Li Xu", "Xiaoxi Jiang", "Guanjun Jiang" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "reasoning", "tool-use" ], "score": 14, "relevance": "high", "matched_queries": [ "language-agent" ], "arxiv_id": "2606.30185", "source": "arxiv", "source_id": "arxiv:2606.30185", "pdf_url": "https://arxiv.org/pdf/2606.30185", "primary_query": "language-agent" }, { "id": "2606.30755", "title": "Understanding and Evaluating Claw-like Agent Security Through a Computer-Systems Lens", "url": "https://arxiv.org/abs/2606.30755", "published": "2026-06-29", "updated": "2026-06-29", "authors": [ "Peizhi Niu", "Wenjie Qu", "Shangding Gu", "Tianneng Shi", "Yuankai Li", "Ahmad Tawaha", "Hend Alzahrani", "Vincent Siu", "Boyi Li", "Chenguang Wang", "Jiaheng Zhang", "Basel Alomair", "Ming Jin", "Muhao Chen", "Chi Wang", "Costas Spanos", "Dawn Song" ], "categories": [ "cs.CR", "cs.AI" ], "topics": [ "agent-evaluation", "agent-safety", "tool-use" ], "score": 14, "relevance": "high", "matched_queries": [ "agent-safety", "ai-agent" ], "arxiv_id": "2606.30755", "source": "arxiv", "source_id": "arxiv:2606.30755", "pdf_url": "https://arxiv.org/pdf/2606.30755", "primary_query": "agent-safety" }, { "id": "2606.29894", "title": "SABER-Math: Automated Benchmark for Information Retrieval Evaluation in Mathematics", "url": "https://arxiv.org/abs/2606.29894", "published": "2026-06-29", "updated": "2026-06-29", "authors": [ "Nikolay Georgiev", "Maria Drencheva", "Kseniia Ibragimova", "Ivo Petrov", "Dimitar I. Dimitrov", "Martin Vechev" ], "categories": [ "cs.IR", "cs.AI", "cs.CL", "cs.LG" ], "topics": [ "agent-evaluation", "rag" ], "score": 14, "relevance": "high", "matched_queries": [ "agentic-ai" ], "arxiv_id": "2606.29894", "source": "arxiv", "source_id": "arxiv:2606.29894", "pdf_url": "https://arxiv.org/pdf/2606.29894", "primary_query": "agentic-ai" }, { "id": "2606.29961", "title": "DuoMem: Towards Capable On-Device Memory Agents via Dual-Space Distillation", "url": "https://arxiv.org/abs/2606.29961", "published": "2026-06-29", "updated": "2026-06-29", "authors": [ "Peyman Hosseini", "Ondrej Bohdal", "Ahmed Alajrami", "Andrea Maracani", "Ignacio Castro", "Matthew Purver", "Mete Ozay", "Savas Ozkan", "Taha Ceritli" ], "categories": [ "cs.LG", "cs.AI" ], "topics": [ "agent-evaluation", "embodied-agent", "memory" ], "score": 14, "relevance": "high", "matched_queries": [ "agent-memory" ], "arxiv_id": "2606.29961", "source": "arxiv", "source_id": "arxiv:2606.29961", "pdf_url": "https://arxiv.org/pdf/2606.29961", "primary_query": "agent-memory" }, { "id": "2606.30119", "title": "On the Internet, Nobody Knows You're an LLM Bot: Unmasking Web Agents with Multi-Layer Fingerprinting", "url": "https://arxiv.org/abs/2606.30119", "published": "2026-06-29", "updated": "2026-06-29", "authors": [ "Iliana Fayolle", "Sihem Bouhenniche", "Samuel Pélissier", "Pierre Laperdrix", "Clémentine Maurice", "Walter Rudametkin" ], "categories": [ "cs.CR" ], "topics": [ "agent-evaluation", "computer-use", "embodied-agent", "rag", "tool-use", "workflow-agent" ], "score": 14, "relevance": "high", "matched_queries": [ "web-gui-agent" ], "arxiv_id": "2606.30119", "source": "arxiv", "source_id": "arxiv:2606.30119", "pdf_url": "https://arxiv.org/pdf/2606.30119", "primary_query": "web-gui-agent" }, { "id": "2606.30602", "title": "MESA: Prioritizing Vulnerable Communication Channels for Securing Multi-Agent Systems", "url": "https://arxiv.org/abs/2606.30602", "published": "2026-06-29", "updated": "2026-06-29", "authors": [ "Kunyang Li", "Kyle Domico", "Jonathan Gregory", "Patrick McDaniel" ], "categories": [ "cs.CR", "cs.AI" ], "topics": [ "agent-evaluation", "agent-safety", "multi-agent", "workflow-agent" ], "score": 14, "relevance": "high", "matched_queries": [ "multi-agent-llm" ], "arxiv_id": "2606.30602", "source": "arxiv", "source_id": "arxiv:2606.30602", "pdf_url": "https://arxiv.org/pdf/2606.30602", "primary_query": "multi-agent-llm" }, { "id": "2606.29932", "title": "SAGA: Scene-Aware, Goal-Evolving Agents for Long-Horizon CivRealm Strategy Planning", "url": "https://arxiv.org/abs/2606.29932", "published": "2026-06-29", "updated": "2026-07-02", "authors": [ "Tianyu Jin", "Shuo Chen", "Yida Wang", "Liuyu Xiang", "Yingzhuo Liu", "Zhiyao Jiang", "Yexin Li", "Zhaofeng He" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "coding-agent", "multi-agent", "planning", "reasoning", "tool-use" ], "score": 14, "relevance": "high", "matched_queries": [ "multi-agent-llm" ], "arxiv_id": "2606.29932", "source": "arxiv", "source_id": "arxiv:2606.29932", "pdf_url": "https://arxiv.org/pdf/2606.29932", "primary_query": "multi-agent-llm" }, { "id": "2606.29746", "title": "DEEPMED Search: An Open-Source Agentic Platform for Medical Deep Research with Introspective Verification", "url": "https://arxiv.org/abs/2606.29746", "published": "2026-06-29", "updated": "2026-06-29", "authors": [ "Maolin Liu", "Fanyu Xu", "Ruoqing Xu", "Jiahang Zhang", "Hao Wang", "Rui Wang" ], "categories": [ "cs.AI", "cs.HC" ], "topics": [ "computer-use", "multi-agent", "rag", "reasoning", "tool-use" ], "score": 14, "relevance": "high", "matched_queries": [ "rag-agent" ], "arxiv_id": "2606.29746", "source": "arxiv", "source_id": "arxiv:2606.29746", "pdf_url": "https://arxiv.org/pdf/2606.29746", "primary_query": "rag-agent" }, { "id": "2606.29225", "title": "PolicyGuard: A Dialogue-Grounded Sub-Agent Verifier for Policy Adherence in LLM Agents", "url": "https://arxiv.org/abs/2606.29225", "published": "2026-06-28", "updated": "2026-06-28", "authors": [ "Seongjae Kang", "Taehyung Yu", "Sung Ju Hwang" ], "categories": [ "cs.AI", "cs.CL" ], "topics": [ "computer-use", "reasoning", "tool-use", "workflow-agent" ], "score": 14, "relevance": "high", "matched_queries": [ "llm-agent" ], "arxiv_id": "2606.29225", "source": "arxiv", "source_id": "arxiv:2606.29225", "pdf_url": "https://arxiv.org/pdf/2606.29225", "primary_query": "llm-agent" }, { "id": "2606.29142", "title": "Agent Security Meets Regulatory Reality -- A Practitioner Systematization of Autonomous-Agent Threats and Controls in Regulated Financial Systems", "url": "https://arxiv.org/abs/2606.29142", "published": "2026-06-28", "updated": "2026-06-28", "authors": [ "Krishna Mohan", "Guda Nagavenkata Srinivasa" ], "categories": [ "cs.CY", "cs.SE" ], "topics": [ "agent-safety", "computer-use", "rag", "tool-use" ], "score": 14, "relevance": "high", "matched_queries": [ "agent-safety", "autonomous-agent-llm", "rag-agent" ], "arxiv_id": "2606.29142", "source": "arxiv", "source_id": "arxiv:2606.29142", "pdf_url": "https://arxiv.org/pdf/2606.29142", "primary_query": "agent-safety" }, { "id": "2606.28733", "title": "Agentic Abstention: Do Agents Know When to Stop Instead of Act?", "url": "https://arxiv.org/abs/2606.28733", "published": "2026-06-27", "updated": "2026-06-27", "authors": [ "Han Luo", "Bingbing Wen", "Lucy Lu Wang" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "reasoning", "tool-use" ], "score": 14, "relevance": "high", "matched_queries": [ "llm-agent" ], "arxiv_id": "2606.28733", "source": "arxiv", "source_id": "arxiv:2606.28733", "pdf_url": "https://arxiv.org/pdf/2606.28733", "primary_query": "llm-agent" }, { "id": "2606.28679", "title": "Capability Gates Are Not Authorization: Confused-Deputy Failures in LLM Agent Frameworks", "url": "https://arxiv.org/abs/2606.28679", "published": "2026-06-27", "updated": "2026-06-27", "authors": [ "David Mellafe Zuvic" ], "categories": [ "cs.CR", "cs.AI" ], "topics": [ "agent-evaluation", "tool-use" ], "score": 14, "relevance": "high", "matched_queries": [ "llm-agent", "tool-use" ], "arxiv_id": "2606.28679", "source": "arxiv", "source_id": "arxiv:2606.28679", "pdf_url": "https://arxiv.org/pdf/2606.28679", "primary_query": "llm-agent" }, { "id": "2606.29026", "title": "Preventing Error Propagation in Multi-Agent AI through Runtime Monitoring", "url": "https://arxiv.org/abs/2606.29026", "published": "2026-06-27", "updated": "2026-06-27", "authors": [ "Shahnewaz Karim Sakib", "Anindya Bijoy Das" ], "categories": [ "cs.AI", "cs.ET" ], "topics": [ "agent-evaluation", "agent-safety", "multi-agent", "reasoning" ], "score": 14, "relevance": "high", "matched_queries": [ "agentic-ai" ], "arxiv_id": "2606.29026", "source": "arxiv", "source_id": "arxiv:2606.29026", "pdf_url": "https://arxiv.org/pdf/2606.29026", "primary_query": "agentic-ai" }, { "id": "2606.28739", "title": "Agent Safety Is Action Alignment", "url": "https://arxiv.org/abs/2606.28739", "published": "2026-06-27", "updated": "2026-06-27", "authors": [ "Shawn Li", "Yue Zhao" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "agent-safety", "tool-use" ], "score": 14, "relevance": "high", "matched_queries": [ "agent-safety" ], "arxiv_id": "2606.28739", "source": "arxiv", "source_id": "arxiv:2606.28739", "pdf_url": "https://arxiv.org/pdf/2606.28739", "primary_query": "agent-safety" }, { "id": "2606.27632", "title": "Yuvion LLM: An Adversarially-Aware Large Language Model for Content And AI Safety", "url": "https://arxiv.org/abs/2606.27632", "published": "2026-06-26", "updated": "2026-06-26", "authors": [ "Ting Ma", "Xiufeng Huang", "Benlei Cui", "Xiaowen Xu", "Shikai Qiu", "Ruijie Jian", "Hongxing Li", "Guanghui Wang", "Longtao Huang", "Haiwen Hong", "Haolei Xu", "Wenjing Jiang", "Ziwen Xu", "Zhaoyu Fan", "Shaoxuan He", "Chuxi Xiao", "Yujian Li", "Xinyue Chen", "Chunyang Chai", "Wenxuan Liu", "Ziheng Wang", "Dongjie Zhang", "Yangfan Zhou", "Libin Dong", "Yupeng Cao", "Xiaoqian Xia", "Jing Wang", "Zhe Jiang", "Zhenan Ye", "Guang Yang", "Bin Liu", "Wei Peng", "Ziqiang Zhu", "Meihui Lian", "Kaiwen Lv Kacuila", "Haidong Ding", "Bingyu Zhu", "Yan Wang", "Hai Zhao", "Xuan Jin", "Wei Zhao", "Pengfei Sun", "Wei Wang", "Huiming Zhang", "Bin Li", "Hui Xue" ], "categories": [ "cs.CL" ], "topics": [ "agent-evaluation", "agent-safety", "planning", "reasoning", "tool-use" ], "score": 14, "relevance": "high", "matched_queries": [ "tool-use" ], "arxiv_id": "2606.27632", "source": "arxiv", "source_id": "arxiv:2606.27632", "pdf_url": "https://arxiv.org/pdf/2606.27632", "primary_query": "tool-use" }, { "id": "2606.26806", "title": "Memory Depth, Not Memory Access: Selective Parametric Consolidation for Long-Running Language Agents", "url": "https://arxiv.org/abs/2606.26806", "published": "2026-06-25", "updated": "2026-06-25", "authors": [ "Haoliang Han" ], "categories": [ "cs.AI", "cs.LG" ], "topics": [ "agent-evaluation", "memory", "rag" ], "score": 14, "relevance": "high", "matched_queries": [ "language-agent" ], "arxiv_id": "2606.26806", "source": "arxiv", "source_id": "arxiv:2606.26806", "pdf_url": "https://arxiv.org/pdf/2606.26806", "primary_query": "language-agent" }, { "id": "2606.27154", "title": "OpenRCA 2.0: From Outcome Labels to Causal Process Supervision", "url": "https://arxiv.org/abs/2606.27154", "published": "2026-06-25", "updated": "2026-06-30", "authors": [ "Aoyang Fang", "Yifan Yang", "Jin'ao Shang", "Qisheng Lu", "Junjielung Xu", "Rui Wang", "Songhan Zhang", "Yuzhong Zhang", "Boxi Yu", "Pinjia He" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "rag", "reasoning", "tool-use" ], "score": 14, "relevance": "high", "matched_queries": [ "tool-use" ], "arxiv_id": "2606.27154", "source": "arxiv", "source_id": "arxiv:2606.27154", "pdf_url": "https://arxiv.org/pdf/2606.27154", "primary_query": "tool-use" }, { "id": "2606.27492", "title": "QueenBee Planner: Skill-Evolving Communication Topologies for Token-Efficient LLM Multi-Agent Systems", "url": "https://arxiv.org/abs/2606.27492", "published": "2026-06-25", "updated": "2026-06-25", "authors": [ "Congjia Tian", "Yuhang Yao", "Jiaming Cui" ], "categories": [ "cs.MA" ], "topics": [ "agent-evaluation", "multi-agent", "planning", "tool-use" ], "score": 14, "relevance": "high", "matched_queries": [ "multi-agent-llm" ], "arxiv_id": "2606.27492", "source": "arxiv", "source_id": "arxiv:2606.27492", "pdf_url": "https://arxiv.org/pdf/2606.27492", "primary_query": "multi-agent-llm" }, { "id": "2606.26758", "title": "EGG: An Expert-Guided Agent Framework for Kernel Generation", "url": "https://arxiv.org/abs/2606.26758", "published": "2026-06-25", "updated": "2026-06-25", "authors": [ "Yaochen Han", "Ke Fan", "Hongxu Jiang", "Wanqi Xu", "Weiyu Xie", "Runhua Zhang", "Chenhui Zhu", "Yixiang Zhang" ], "categories": [ "cs.AI" ], "topics": [ "computer-use", "memory", "multi-agent", "rag", "workflow-agent" ], "score": 14, "relevance": "high", "matched_queries": [ "multi-agent-llm" ], "arxiv_id": "2606.26758", "source": "arxiv", "source_id": "arxiv:2606.26758", "pdf_url": "https://arxiv.org/pdf/2606.26758", "primary_query": "multi-agent-llm" }, { "id": "2606.26205", "title": "Knowledge-augmented Agentic AI for Mental Health Medication Information Seeking", "url": "https://arxiv.org/abs/2606.26205", "published": "2026-06-24", "updated": "2026-06-24", "authors": [ "Huizi Yu", "Jian Liu", "Wenkong Wang", "Lingyao Li", "Jiayan Zhou", "Zhaoqian Xue", "Xiang Li", "Xinxin Lin", "Zhiying Liang", "Zhuoru Wu", "Siyuan Ma", "Xin Ma", "Lizhou Fan" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "agent-safety", "multi-agent" ], "score": 14, "relevance": "high", "matched_queries": [ "agentic-ai", "multi-agent-llm" ], "arxiv_id": "2606.26205", "source": "arxiv", "source_id": "arxiv:2606.26205", "pdf_url": "https://arxiv.org/pdf/2606.26205", "primary_query": "agentic-ai" }, { "id": "2606.25899", "title": "Manipulation Is Task-Dependent: A Multi-Axis, Multi-Environment Evaluation of Frontier LLMs", "url": "https://arxiv.org/abs/2606.25899", "published": "2026-06-24", "updated": "2026-06-24", "authors": [ "Adeeb Zaman", "Erik Nordby", "Fred Heiding" ], "categories": [ "cs.MA" ], "topics": [ "agent-evaluation", "rag", "tool-use", "workflow-agent" ], "score": 14, "relevance": "high", "matched_queries": [ "agentic-ai" ], "arxiv_id": "2606.25899", "source": "arxiv", "source_id": "arxiv:2606.25899", "pdf_url": "https://arxiv.org/pdf/2606.25899", "primary_query": "agentic-ai" }, { "id": "2606.25484", "title": "From Causal Discovery to Implementation: An Agentic AI Framework for E-Scooter Mobility Hub Planning Across 29 German Cities", "url": "https://arxiv.org/abs/2606.25484", "published": "2026-06-24", "updated": "2026-06-24", "authors": [ "Meng Jin", "Melanie Handrich", "Simone Martinenz", "Nicholas Hoeser", "Ziyue Li" ], "categories": [ "cs.CY", "econ.GN", "stat.AP" ], "topics": [ "coding-agent", "computer-use", "planning", "tool-use" ], "score": 14, "relevance": "high", "matched_queries": [ "agentic-ai" ], "arxiv_id": "2606.25484", "source": "arxiv", "source_id": "arxiv:2606.25484", "pdf_url": "https://arxiv.org/pdf/2606.25484", "primary_query": "agentic-ai" }, { "id": "2606.26300", "title": "The Verification Horizon: No Silver Bullet for Coding Agent Rewards", "url": "https://arxiv.org/abs/2606.26300", "published": "2026-06-24", "updated": "2026-06-29", "authors": [ "Binghai Wang", "Chenlong Zhang", "Dayiheng Liu", "Jiajun Zhang", "Jiawei Chen", "Mingze Li", "Mouxiang Chen", "Rongyao Fang", "Siyuan Zhang", "Xuwu Wang", "Yuheng Jing", "Zeyao Ma", "Zeyu Cui" ], "categories": [ "cs.AI", "cs.CL" ], "topics": [ "agent-evaluation", "coding-agent", "planning", "rag", "reasoning" ], "score": 14, "relevance": "high", "matched_queries": [ "coding-agent" ], "arxiv_id": "2606.26300", "source": "arxiv", "source_id": "arxiv:2606.26300", "pdf_url": "https://arxiv.org/pdf/2606.26300", "primary_query": "coding-agent" }, { "id": "2606.25705", "title": "GUI agent: Guided Exploration of User-Sensitive Screens", "url": "https://arxiv.org/abs/2606.25705", "published": "2026-06-24", "updated": "2026-06-24", "authors": [ "Aradhana Nayak", "Mussadiq Nazeer", "Wang Peng", "Feng Liu" ], "categories": [ "cs.AI" ], "topics": [ "agent-safety", "computer-use", "tool-use" ], "score": 14, "relevance": "high", "matched_queries": [ "web-gui-agent" ], "arxiv_id": "2606.25705", "source": "arxiv", "source_id": "arxiv:2606.25705", "pdf_url": "https://arxiv.org/pdf/2606.25705", "primary_query": "web-gui-agent" }, { "id": "2606.25656", "title": "Is GraphRAG Needed? From Basic RAG to Graph-/Agentic Solutions with Context Optimization", "url": "https://arxiv.org/abs/2606.25656", "published": "2026-06-24", "updated": "2026-06-24", "authors": [ "Long Chen", "Ryan Razkenari", "Yuxuan Zhou", "Yuan Tian", "Rahul Ghosh", "Venkatesh Pappakrishnan", "Disha Ahuja", "Vidya Sagar Ravipati" ], "categories": [ "cs.CL", "cs.AI", "cs.IR" ], "topics": [ "agent-evaluation", "memory", "planning", "rag" ], "score": 14, "relevance": "high", "matched_queries": [ "rag-agent" ], "arxiv_id": "2606.25656", "source": "arxiv", "source_id": "arxiv:2606.25656", "pdf_url": "https://arxiv.org/pdf/2606.25656", "primary_query": "rag-agent" }, { "id": "2606.25189", "title": "ActPlane: Programmable OS-Level Policy Enforcement for Agent Harnesses", "url": "https://arxiv.org/abs/2606.25189", "published": "2026-06-23", "updated": "2026-06-30", "authors": [ "Yusheng Zheng", "Tianyuan Wu", "Quanzhi Fu", "Tong Yu", "Wenan Mao", "Tao Ma", "Dan Williams", "Wei Wang", "Andi Quinn" ], "categories": [ "cs.OS" ], "topics": [ "agent-evaluation", "agent-safety", "coding-agent", "planning", "tool-use" ], "score": 14, "relevance": "high", "matched_queries": [ "ai-agent" ], "arxiv_id": "2606.25189", "source": "arxiv", "source_id": "arxiv:2606.25189", "pdf_url": "https://arxiv.org/pdf/2606.25189", "primary_query": "ai-agent" }, { "id": "2606.24402", "title": "Poisoned Playbooks: Demystifying Knowledge Poisoning Effects on AI Security Agents", "url": "https://arxiv.org/abs/2606.24402", "published": "2026-06-23", "updated": "2026-06-23", "authors": [ "Juho Park", "Hyunmin Choi", "Kevin Nam" ], "categories": [ "cs.CR" ], "topics": [ "agent-evaluation", "agent-safety", "rag", "reasoning", "tool-use" ], "score": 14, "relevance": "high", "matched_queries": [ "ai-agent", "rag-agent" ], "arxiv_id": "2606.24402", "source": "arxiv", "source_id": "arxiv:2606.24402", "pdf_url": "https://arxiv.org/pdf/2606.24402", "primary_query": "ai-agent" }, { "id": "2606.24235", "title": "SP-Mind: An Autonomous Reasoning Agent for Spatial Proteomics Analysis", "url": "https://arxiv.org/abs/2606.24235", "published": "2026-06-23", "updated": "2026-06-23", "authors": [ "Yucheng Yuan", "Yuanfeng Ji", "Zhongxiao Li", "Ruijiang Li" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "computer-use", "rag", "reasoning", "tool-use", "workflow-agent" ], "score": 14, "relevance": "high", "matched_queries": [ "ai-agent" ], "arxiv_id": "2606.24235", "source": "arxiv", "source_id": "arxiv:2606.24235", "pdf_url": "https://arxiv.org/pdf/2606.24235", "primary_query": "ai-agent" }, { "id": "2606.25206", "title": "RAVEN: Long-Horizon Reasoning & Navigation with a Visuo-Spatio-Temporal Memory", "url": "https://arxiv.org/abs/2606.25206", "published": "2026-06-23", "updated": "2026-06-23", "authors": [ "Yixun Hu", "Zhicheng Zheng", "Lihan Zha", "Chunwei Xing", "Rajdeep Singh", "Omar Hossain", "Antonio Loquercio", "Dhruv Shah" ], "categories": [ "cs.RO", "cs.AI", "cs.CL" ], "topics": [ "agent-evaluation", "embodied-agent", "memory", "planning", "rag", "reasoning" ], "score": 14, "relevance": "high", "matched_queries": [ "agent-memory" ], "arxiv_id": "2606.25206", "source": "arxiv", "source_id": "arxiv:2606.25206", "pdf_url": "https://arxiv.org/pdf/2606.25206", "primary_query": "agent-memory" }, { "id": "2606.24515", "title": "Reinforcement Learning for Computer-Use Agents with Autonomous Evaluation", "url": "https://arxiv.org/abs/2606.24515", "published": "2026-06-23", "updated": "2026-06-23", "authors": [ "Marta Sumyk", "Oleksandr Kosovan" ], "categories": [ "cs.AI", "cs.HC" ], "topics": [ "agent-evaluation", "computer-use", "rag" ], "score": 14, "relevance": "high", "matched_queries": [ "web-gui-agent" ], "arxiv_id": "2606.24515", "source": "arxiv", "source_id": "arxiv:2606.24515", "pdf_url": "https://arxiv.org/pdf/2606.24515", "primary_query": "web-gui-agent" }, { "id": "2606.24694", "title": "SupplyNet: Supporting Visual Exploratory Learning in Supply Chain via Contextual Multi-Agent Simulation", "url": "https://arxiv.org/abs/2606.24694", "published": "2026-06-23", "updated": "2026-06-23", "authors": [ "Yanjia Li", "Kelcy Kexin Han", "Tianrui Hu", "Yi-Fan Cao", "Huamin Qu", "Sicheng Song" ], "categories": [ "cs.HC" ], "topics": [ "multi-agent", "rag", "reasoning", "world-model" ], "score": 14, "relevance": "high", "matched_queries": [ "multi-agent-llm" ], "arxiv_id": "2606.24694", "source": "arxiv", "source_id": "arxiv:2606.24694", "pdf_url": "https://arxiv.org/pdf/2606.24694", "primary_query": "multi-agent-llm" }, { "id": "2606.24976", "title": "Diagnosing and Mitigating Compounding Failures in Agentic Persuasion via Taxonomic Strategy Retrieval", "url": "https://arxiv.org/abs/2606.24976", "published": "2026-06-23", "updated": "2026-07-04", "authors": [ "Sana Ayromlou", "Purvi Sehgal", "Pradyumna Narayana" ], "categories": [ "cs.AI", "cs.CL", "cs.LG" ], "topics": [ "agent-evaluation", "multi-agent", "planning", "rag" ], "score": 14, "relevance": "high", "matched_queries": [ "rag-agent" ], "arxiv_id": "2606.24976", "source": "arxiv", "source_id": "arxiv:2606.24976", "pdf_url": "https://arxiv.org/pdf/2606.24976", "primary_query": "rag-agent" }, { "id": "2606.22737", "title": "GroundEval: A Deterministic Replacement for LLM-as-Judge in Stateful Agent Evaluation", "url": "https://arxiv.org/abs/2606.22737", "published": "2026-06-22", "updated": "2026-07-02", "authors": [ "Jeffrey Flynt" ], "categories": [ "cs.AI", "cs.CL", "cs.SE" ], "topics": [ "agent-evaluation", "memory", "tool-use" ], "score": 14, "relevance": "high", "matched_queries": [ "agent-evaluation" ], "arxiv_id": "2606.22737", "source": "arxiv", "source_id": "arxiv:2606.22737", "pdf_url": "https://arxiv.org/pdf/2606.22737", "primary_query": "agent-evaluation" }, { "id": "2606.23283", "title": "Towards Root Memories: Benchmarking and Enhancing Implicit Logical Memory Retrieval for Personalized LLMs", "url": "https://arxiv.org/abs/2606.23283", "published": "2026-06-22", "updated": "2026-06-22", "authors": [ "Hongxun Ding", "Xiang Yu", "Chengbing Wang", "Jianfei Xiao", "Keqin Bao", "Wenjie Wang", "Xiangnan He" ], "categories": [ "cs.CL" ], "topics": [ "agent-evaluation", "memory", "rag" ], "score": 14, "relevance": "high", "matched_queries": [ "agent-memory" ], "arxiv_id": "2606.23283", "source": "arxiv", "source_id": "arxiv:2606.23283", "pdf_url": "https://arxiv.org/pdf/2606.23283", "primary_query": "agent-memory" }, { "id": "2606.23195", "title": "Memory Contagion: Cross-Temporal Propagation of Evaluator Bias via Agent Memory", "url": "https://arxiv.org/abs/2606.23195", "published": "2026-06-22", "updated": "2026-06-24", "authors": [ "Zewen Liu" ], "categories": [ "cs.LG", "cs.AI", "cs.CL" ], "topics": [ "agent-evaluation", "computer-use", "memory", "tool-use" ], "score": 14, "relevance": "high", "matched_queries": [ "agent-memory" ], "arxiv_id": "2606.23195", "source": "arxiv", "source_id": "arxiv:2606.23195", "pdf_url": "https://arxiv.org/pdf/2606.23195", "primary_query": "agent-memory" }, { "id": "2606.22864", "title": "When AUC 0.998 Is Not Enough: A Candidate Evaluation Protocol for Hidden-State Probes of Indirect Prompt Injection in Multimodal Computer-Use Agents", "url": "https://arxiv.org/abs/2606.22864", "published": "2026-06-22", "updated": "2026-06-22", "authors": [ "Yanhang Li", "Zhichao Fan", "Zexin Zhuang" ], "categories": [ "cs.LG" ], "topics": [ "agent-evaluation", "agent-safety", "tool-use" ], "score": 14, "relevance": "high", "matched_queries": [ "web-gui-agent" ], "arxiv_id": "2606.22864", "source": "arxiv", "source_id": "arxiv:2606.22864", "pdf_url": "https://arxiv.org/pdf/2606.22864", "primary_query": "web-gui-agent" }, { "id": "2606.22495", "title": "Grounded Scaling: Why Agentic AI Needs Deterministic Environments", "url": "https://arxiv.org/abs/2606.22495", "published": "2026-06-21", "updated": "2026-06-21", "authors": [ "Liang Ding", "Xintong Wang" ], "categories": [ "cs.AI" ], "topics": [ "agent-safety", "embodied-agent", "multi-agent", "reasoning", "tool-use" ], "score": 14, "relevance": "high", "matched_queries": [ "agentic-ai" ], "arxiv_id": "2606.22495", "source": "arxiv", "source_id": "arxiv:2606.22495", "pdf_url": "https://arxiv.org/pdf/2606.22495", "primary_query": "agentic-ai" }, { "id": "2606.22484", "title": "Governed AI-Assisted Engineering: Graduated Human Oversight for Agentic Code Generation in Regulated Domains", "url": "https://arxiv.org/abs/2606.22484", "published": "2026-06-21", "updated": "2026-07-04", "authors": [ "Richard Kang" ], "categories": [ "cs.HC" ], "topics": [ "agent-evaluation", "agent-safety", "coding-agent", "rag", "workflow-agent" ], "score": 14, "relevance": "high", "matched_queries": [ "agentic-ai" ], "arxiv_id": "2606.22484", "source": "arxiv", "source_id": "arxiv:2606.22484", "pdf_url": "https://arxiv.org/pdf/2606.22484", "primary_query": "agentic-ai" }, { "id": "2606.22030", "title": "Nous: A Predictive World Model for Long-Term Agent Memory", "url": "https://arxiv.org/abs/2606.22030", "published": "2026-06-20", "updated": "2026-06-20", "authors": [ "Pranav Singh" ], "categories": [ "cs.AI", "cs.CL", "cs.IR", "cs.LG" ], "topics": [ "agent-evaluation", "memory", "rag", "world-model" ], "score": 14, "relevance": "high", "matched_queries": [ "agent-memory" ], "arxiv_id": "2606.22030", "source": "arxiv", "source_id": "arxiv:2606.22030", "pdf_url": "https://arxiv.org/pdf/2606.22030", "primary_query": "agent-memory" }, { "id": "2606.22151", "title": "Novelty-Aware Agentic Retrieval: Comparing Research Contributions Through Structured Multi-Step Reasoning", "url": "https://arxiv.org/abs/2606.22151", "published": "2026-06-20", "updated": "2026-06-20", "authors": [ "Shou-Tzu Han" ], "categories": [ "cs.IR" ], "topics": [ "agent-evaluation", "computer-use", "rag", "reasoning", "tool-use" ], "score": 14, "relevance": "high", "matched_queries": [ "rag-agent" ], "arxiv_id": "2606.22151", "source": "arxiv", "source_id": "arxiv:2606.22151", "pdf_url": "https://arxiv.org/pdf/2606.22151", "primary_query": "rag-agent" }, { "id": "2606.21842", "title": "Agent-Assisted Side-Channel Attacks on Non-Prefix KV Cache in RAG", "url": "https://arxiv.org/abs/2606.21842", "published": "2026-06-20", "updated": "2026-06-20", "authors": [ "He Sun", "Shinan Liu", "Siyuan Ma", "Junhao Li", "Mingjun Xiao", "Wenhao Jiang" ], "categories": [ "cs.CR" ], "topics": [ "agent-evaluation", "agent-safety", "memory", "rag", "tool-use" ], "score": 14, "relevance": "high", "matched_queries": [ "rag-agent" ], "arxiv_id": "2606.21842", "source": "arxiv", "source_id": "arxiv:2606.21842", "pdf_url": "https://arxiv.org/pdf/2606.21842", "primary_query": "rag-agent" }, { "id": "2606.21409", "title": "Don't Blindly Trust It: How Unreliable Feedback Breaks Tool-Using LLM Agents", "url": "https://arxiv.org/abs/2606.21409", "published": "2026-06-19", "updated": "2026-06-19", "authors": [ "Chubin Zhang", "Zhenglin Wan", "Xingrui Yu", "Pengfei Zhou", "Wangbo Zhao", "Jingxuan Wu", "Yaxin Zhou", "Ivor Tsang" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "coding-agent", "rag", "tool-use" ], "score": 14, "relevance": "high", "matched_queries": [ "tool-use" ], "arxiv_id": "2606.21409", "source": "arxiv", "source_id": "arxiv:2606.21409", "pdf_url": "https://arxiv.org/pdf/2606.21409", "primary_query": "tool-use" }, { "id": "2606.21553", "title": "Dissecting Agentic RAG: A Component Ablation for Multi-Hop QA with a Local 7B Model", "url": "https://arxiv.org/abs/2606.21553", "published": "2026-06-19", "updated": "2026-06-19", "authors": [ "Sheroz Shaikh" ], "categories": [ "cs.CL", "cs.IR" ], "topics": [ "agent-evaluation", "rag", "reasoning", "tool-use" ], "score": 14, "relevance": "high", "matched_queries": [ "rag-agent" ], "arxiv_id": "2606.21553", "source": "arxiv", "source_id": "arxiv:2606.21553", "pdf_url": "https://arxiv.org/pdf/2606.21553", "primary_query": "rag-agent" }, { "id": "2606.20470", "title": "Analyzing Defensive Misdirection Against Model-Guided Automated Attacks on Agentic AI Systems", "url": "https://arxiv.org/abs/2606.20470", "published": "2026-06-18", "updated": "2026-06-26", "authors": [ "Reza Soosahabi", "Vivek Namsani" ], "categories": [ "cs.CR", "cs.AI" ], "topics": [ "agent-evaluation", "computer-use", "tool-use", "workflow-agent" ], "score": 14, "relevance": "high", "matched_queries": [ "agentic-ai" ], "arxiv_id": "2606.20470", "source": "arxiv", "source_id": "arxiv:2606.20470", "pdf_url": "https://arxiv.org/pdf/2606.20470", "primary_query": "agentic-ai" }, { "id": "2606.19812", "title": "Human-on-the-Loop Orchestration for AI-Assisted Legal Discovery", "url": "https://arxiv.org/abs/2606.19812", "published": "2026-06-18", "updated": "2026-06-18", "authors": [ "Anushree Sinha", "Srivaths Ranganathan", "Abhishek Dharmaratnakar", "Debanshu Das" ], "categories": [ "cs.AI", "cs.LG" ], "topics": [ "agent-evaluation", "agent-safety", "planning", "rag", "reasoning", "workflow-agent", "world-model" ], "score": 14, "relevance": "high", "matched_queries": [ "agentic-ai", "planning-agent" ], "arxiv_id": "2606.19812", "source": "arxiv", "source_id": "arxiv:2606.19812", "pdf_url": "https://arxiv.org/pdf/2606.19812", "primary_query": "agentic-ai" }, { "id": "2606.20515", "title": "S-Agent: Spatial Tool-Use Elicits Reasoning for Spatial Intelligence", "url": "https://arxiv.org/abs/2606.20515", "published": "2026-06-18", "updated": "2026-06-28", "authors": [ "Yalun Dai", "Hao Li", "Shulin Tian", "Runmao Yao", "Yuhao Dong", "Fangzhou Hong", "Zhaoxi Chen", "Fangfu Liu", "Baoliang Tian", "Dingwen Zhang", "Tao Wang", "Kim-Hui Yap", "Ziwei Liu" ], "categories": [ "cs.CV" ], "topics": [ "agent-evaluation", "memory", "planning", "reasoning", "tool-use" ], "score": 14, "relevance": "high", "matched_queries": [ "agent-memory", "tool-use" ], "arxiv_id": "2606.20515", "source": "arxiv", "source_id": "arxiv:2606.20515", "pdf_url": "https://arxiv.org/pdf/2606.20515", "primary_query": "agent-memory" }, { "id": "2606.20023", "title": "When Lower Privileges Suffice: Investigating Over-Privileged Tool Selection in LLM Agents", "url": "https://arxiv.org/abs/2606.20023", "published": "2026-06-18", "updated": "2026-07-07", "authors": [ "Kaiyue Yang", "Yuyan Bu", "Jingwei Yi", "Yuchi Wang", "Biyu Zhou", "Juntao Dai", "Songlin Hu", "Yaodong Yang" ], "categories": [ "cs.SE", "cs.AI", "cs.CL" ], "topics": [ "agent-evaluation", "agent-safety", "tool-use" ], "score": 14, "relevance": "high", "matched_queries": [ "tool-use" ], "arxiv_id": "2606.20023", "source": "arxiv", "source_id": "arxiv:2606.20023", "pdf_url": "https://arxiv.org/pdf/2606.20023", "primary_query": "tool-use" }, { "id": "2606.20785", "title": "Fara-1.5: Scalable Learning Environments for Computer Use Agents", "url": "https://arxiv.org/abs/2606.20785", "published": "2026-06-18", "updated": "2026-06-18", "authors": [ "Ahmed Awadallah", "Sahil Gupta", "Yash Lara", "Yadong Lu", "Hussein Mozannar", "Akshay Nambi", "Zach Nussbaum", "Yash Pandya", "Aravind Rajeswaran", "Corby Rosset", "Alexey Taymanov", "Luiz do Valle", "Vibhav Vineet", "Spencer Whitehead", "Andrew Zhao" ], "categories": [ "cs.AI", "cs.LG" ], "topics": [ "agent-evaluation", "computer-use", "rag", "reasoning", "tool-use" ], "score": 14, "relevance": "high", "matched_queries": [ "web-gui-agent" ], "arxiv_id": "2606.20785", "source": "arxiv", "source_id": "arxiv:2606.20785", "pdf_url": "https://arxiv.org/pdf/2606.20785", "primary_query": "web-gui-agent" }, { "id": "2606.19930", "title": "MobileForge: Annotation-Free Adaptation for Mobile GUI Agents with Hierarchical Feedback-Guided Policy Optimization", "url": "https://arxiv.org/abs/2606.19930", "published": "2026-06-18", "updated": "2026-06-18", "authors": [ "Guangyi Liu", "Pengxiang Zhao", "Gao Wu", "Yiwen Yin", "Mading Li", "Liang Liu", "Congxiao Liu", "Zhang Qi", "Mengyan Wang", "Liang Guo", "Yong Liu" ], "categories": [ "cs.HC" ], "topics": [ "agent-evaluation", "computer-use", "tool-use" ], "score": 14, "relevance": "high", "matched_queries": [ "web-gui-agent" ], "arxiv_id": "2606.19930", "source": "arxiv", "source_id": "arxiv:2606.19930", "pdf_url": "https://arxiv.org/pdf/2606.19930", "primary_query": "web-gui-agent" }, { "id": "2606.18671", "title": "HANSEL: Extracting Breadcrumbs from Web Agent Trajectories for Interactive Verification", "url": "https://arxiv.org/abs/2606.18671", "published": "2026-06-17", "updated": "2026-06-17", "authors": [ "Yujin Zhang", "Daye Nam" ], "categories": [ "cs.HC" ], "topics": [ "agent-evaluation", "computer-use", "embodied-agent", "planning" ], "score": 14, "relevance": "high", "matched_queries": [ "ai-agent", "web-gui-agent" ], "arxiv_id": "2606.18671", "source": "arxiv", "source_id": "arxiv:2606.18671", "pdf_url": "https://arxiv.org/pdf/2606.18671", "primary_query": "ai-agent" }, { "id": "2606.19063", "title": "PYPILINE: Malicious PyPI Package Detection via Suspicious API Knowledge and Agent Workflow", "url": "https://arxiv.org/abs/2606.19063", "published": "2026-06-17", "updated": "2026-06-26", "authors": [ "Siyuan Pang", "Yepeng Yao", "Zhengwei Jiang", "Zijing Fan", "Haozhe Li", "Baoxu Liu" ], "categories": [ "cs.CR" ], "topics": [ "agent-evaluation", "agent-safety", "rag", "tool-use", "workflow-agent" ], "score": 14, "relevance": "high", "matched_queries": [ "agentic-ai", "rag-agent" ], "arxiv_id": "2606.19063", "source": "arxiv", "source_id": "arxiv:2606.19063", "pdf_url": "https://arxiv.org/pdf/2606.19063", "primary_query": "agentic-ai" }, { "id": "2606.19409", "title": "OpenRath: Session-Centered Runtime State for Agent Systems", "url": "https://arxiv.org/abs/2606.19409", "published": "2026-06-17", "updated": "2026-06-17", "authors": [ "Fukang Wen", "Zhijie Wang", "Ruilin Xu" ], "categories": [ "cs.SE", "cs.PL" ], "topics": [ "agent-evaluation", "memory", "multi-agent", "rag", "tool-use", "workflow-agent" ], "score": 14, "relevance": "high", "matched_queries": [ "agent-memory" ], "arxiv_id": "2606.19409", "source": "arxiv", "source_id": "arxiv:2606.19409", "pdf_url": "https://arxiv.org/pdf/2606.19409", "primary_query": "agent-memory" }, { "id": "2606.19613", "title": "StaminaBench: Stress-Testing Coding Agents over 100 Interaction Turns", "url": "https://arxiv.org/abs/2606.19613", "published": "2026-06-17", "updated": "2026-06-17", "authors": [ "Vlad Sobal", "Shuo Yang", "Yuting Zhang", "Wei Xia", "Stefano Soatto" ], "categories": [ "cs.SE", "cs.AI" ], "topics": [ "agent-evaluation", "coding-agent", "tool-use" ], "score": 14, "relevance": "high", "matched_queries": [ "coding-agent" ], "arxiv_id": "2606.19613", "source": "arxiv", "source_id": "arxiv:2606.19613", "pdf_url": "https://arxiv.org/pdf/2606.19613", "primary_query": "coding-agent" }, { "id": "2606.20717", "title": "MIRAGE: Stealthy Visual Prompt Injection for Vulnerability Detection in Web Agents", "url": "https://arxiv.org/abs/2606.20717", "published": "2026-06-16", "updated": "2026-06-16", "authors": [ "Xuelong Dai", "Jianyu Ma", "Boyang Ma", "Biwei Yan", "Yijun Yang", "Yue Zhang" ], "categories": [ "cs.CV", "cs.AI", "cs.CR" ], "topics": [ "agent-evaluation", "agent-safety", "computer-use", "rag", "tool-use", "workflow-agent" ], "score": 14, "relevance": "high", "matched_queries": [ "web-gui-agent" ], "arxiv_id": "2606.20717", "source": "arxiv", "source_id": "arxiv:2606.20717", "pdf_url": "https://arxiv.org/pdf/2606.20717", "primary_query": "web-gui-agent" }, { "id": "2606.16111", "title": "Towards Pareto-Optimal Tool-Integrated Agents with Pareto Ranking Policy Optimization", "url": "https://arxiv.org/abs/2606.16111", "published": "2026-06-15", "updated": "2026-06-15", "authors": [ "Junyi Li", "Xiaowei Qian", "Yingyi Zhang", "Wenlin Zhang", "Guojing Li", "Sheng Zhang", "Xiao Han", "Yichao Wang", "Xiangyu Zhao" ], "categories": [ "cs.CL" ], "topics": [ "agent-safety", "computer-use", "rag", "reasoning", "tool-use" ], "score": 14, "relevance": "high", "matched_queries": [ "language-agent", "tool-use" ], "arxiv_id": "2606.16111", "source": "arxiv", "source_id": "arxiv:2606.16111", "pdf_url": "https://arxiv.org/pdf/2606.16111", "primary_query": "language-agent" }, { "id": "2606.16748", "title": "MyPCBench: A Benchmark for Personally Intelligent Computer-Use Agents", "url": "https://arxiv.org/abs/2606.16748", "published": "2026-06-15", "updated": "2026-06-15", "authors": [ "Lawrence Keunho Jang", "Andrew Keunwoo Jang", "Jing Yu Koh", "Ruslan Salakhutdinov" ], "categories": [ "cs.LG", "cs.CL" ], "topics": [ "agent-evaluation", "computer-use", "tool-use", "workflow-agent" ], "score": 14, "relevance": "high", "matched_queries": [ "agent-evaluation", "web-gui-agent" ], "arxiv_id": "2606.16748", "source": "arxiv", "source_id": "arxiv:2606.16748", "pdf_url": "https://arxiv.org/pdf/2606.16748", "primary_query": "agent-evaluation" }, { "id": "2606.15591", "title": "Agentic Retrieval and Reinforcement Learned Equation Chains: A Controlled Generation Framework for Complex and Novel Physics Word Problems", "url": "https://arxiv.org/abs/2606.15591", "published": "2026-06-14", "updated": "2026-06-14", "authors": [ "Tirthankar Mittra" ], "categories": [ "cs.AI", "cs.CL", "cs.MA" ], "topics": [ "agent-evaluation", "computer-use", "rag" ], "score": 14, "relevance": "high", "matched_queries": [ "rag-agent" ], "arxiv_id": "2606.15591", "source": "arxiv", "source_id": "arxiv:2606.15591", "pdf_url": "https://arxiv.org/pdf/2606.15591", "primary_query": "rag-agent" }, { "id": "2606.15152", "title": "Can Agents Read the Room? Benchmarking Visual Social Intelligence in Multimodal Simulation", "url": "https://arxiv.org/abs/2606.15152", "published": "2026-06-13", "updated": "2026-06-13", "authors": [ "Shijun Wan", "Xuehai Wu", "Jiwen Zhang", "Siyuan Wang", "Zhongyu Wei" ], "categories": [ "cs.CL" ], "topics": [ "agent-evaluation", "computer-use", "tool-use", "world-model" ], "score": 14, "relevance": "high", "matched_queries": [ "agent-evaluation" ], "arxiv_id": "2606.15152", "source": "arxiv", "source_id": "arxiv:2606.15152", "pdf_url": "https://arxiv.org/pdf/2606.15152", "primary_query": "agent-evaluation" }, { "id": "2606.15242", "title": "Benign in Isolation, Harmful in Composition: Security Risks in Agent Skill Ecosystems", "url": "https://arxiv.org/abs/2606.15242", "published": "2026-06-13", "updated": "2026-06-13", "authors": [ "Yi Xie", "Jiawei Du", "Yu Cheng", "Jiuan Zhou", "Zhaoxia Yin" ], "categories": [ "cs.CR", "cs.AI" ], "topics": [ "agent-evaluation", "agent-safety", "planning", "tool-use" ], "score": 14, "relevance": "high", "matched_queries": [ "planning-agent" ], "arxiv_id": "2606.15242", "source": "arxiv", "source_id": "arxiv:2606.15242", "pdf_url": "https://arxiv.org/pdf/2606.15242", "primary_query": "planning-agent" }, { "id": "2606.14502", "title": "From Chatbot to Digital Colleague: The Paradigm Shift Toward Persistent Autonomous AI", "url": "https://arxiv.org/abs/2606.14502", "published": "2026-06-12", "updated": "2026-06-12", "authors": [ "Yongheng Zhang", "Ziang Liu", "Jiaxuan Zhu", "Shuai Wang", "Xiangqi Chen", "Haojing Huang", "Jiayi Kuang", "Siyu Chen", "Ao Shen", "Hao Wu", "Qiufeng Wang", "Qian-Wen Zhang", "Junnan Dong", "Wenhao Jiang", "Ying Shen", "Hai-Tao Zheng", "Yinghui Li", "Di Yin", "Xing Sun", "Philip S. Yu" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "memory", "rag", "reasoning", "tool-use" ], "score": 14, "relevance": "high", "matched_queries": [ "tool-use" ], "arxiv_id": "2606.14502", "source": "arxiv", "source_id": "arxiv:2606.14502", "pdf_url": "https://arxiv.org/pdf/2606.14502", "primary_query": "tool-use" }, { "id": "2606.15017", "title": "Are Online Skill and Memory Modules Always Worth Their Tokens? A Budget-Constrained Study of Web Agents", "url": "https://arxiv.org/abs/2606.15017", "published": "2026-06-12", "updated": "2026-06-12", "authors": [ "Sina Hajimiri", "Masih Aminbeidokhti", "Jose Dolz", "Ismail Ben Ayed", "Issam H. Laradji", "Spandana Gella", "Nicolas Gontier" ], "categories": [ "cs.CL" ], "topics": [ "agent-evaluation", "computer-use", "memory", "reasoning", "workflow-agent" ], "score": 14, "relevance": "high", "matched_queries": [ "web-gui-agent" ], "arxiv_id": "2606.15017", "source": "arxiv", "source_id": "arxiv:2606.15017", "pdf_url": "https://arxiv.org/pdf/2606.15017", "primary_query": "web-gui-agent" }, { "id": "2606.14574", "title": "SIMMER: Benchmarking Latent Failures in LLM Executable Planning with a World Model", "url": "https://arxiv.org/abs/2606.14574", "published": "2026-06-12", "updated": "2026-06-12", "authors": [ "Xiaoxin Lu", "Ranran Haoran Zhang", "Rui Zhang" ], "categories": [ "cs.CL", "cs.AI" ], "topics": [ "agent-evaluation", "planning", "rag", "reasoning", "tool-use", "world-model" ], "score": 14, "relevance": "high", "matched_queries": [ "autonomous-agent-llm" ], "arxiv_id": "2606.14574", "source": "arxiv", "source_id": "arxiv:2606.14574", "pdf_url": "https://arxiv.org/pdf/2606.14574", "primary_query": "autonomous-agent-llm" }, { "id": "2606.13317", "title": "SkillCAT: Contrastive Assessment and Topology-Aware Skill Self-Evolution for LLM Agents", "url": "https://arxiv.org/abs/2606.13317", "published": "2026-06-11", "updated": "2026-06-11", "authors": [ "Kunfeng Chen", "Qihuang Zhong", "Juhua Liu", "Bo Du" ], "categories": [ "cs.CL" ], "topics": [ "agent-evaluation", "rag", "tool-use" ], "score": 14, "relevance": "high", "matched_queries": [ "agent-evaluation" ], "arxiv_id": "2606.13317", "source": "arxiv", "source_id": "arxiv:2606.13317", "pdf_url": "https://arxiv.org/pdf/2606.13317", "primary_query": "agent-evaluation" }, { "id": "2606.14805", "title": "Knowledge-Based Zero-Replay Debugging of Multi-Agent LLM Traces", "url": "https://arxiv.org/abs/2606.14805", "published": "2026-06-11", "updated": "2026-06-11", "authors": [ "Dong Ho Kang", "Hyeonjeong Cha", "Daein Weon" ], "categories": [ "cs.SE", "cs.AI" ], "topics": [ "memory", "multi-agent", "tool-use" ], "score": 14, "relevance": "high", "matched_queries": [ "tool-use" ], "arxiv_id": "2606.14805", "source": "arxiv", "source_id": "arxiv:2606.14805", "pdf_url": "https://arxiv.org/pdf/2606.14805", "primary_query": "tool-use" }, { "id": "2606.13602", "title": "EpiBench: Verifiable Evaluation of AI Agents on Epigenomics Analysis", "url": "https://arxiv.org/abs/2606.13602", "published": "2026-06-11", "updated": "2026-06-11", "authors": [ "Harihara Muralidharan", "Reema Baskar", "Soo Hee Lee", "Tim Proctor", "Kenny Workman" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "workflow-agent" ], "score": 14, "relevance": "high", "matched_queries": [ "web-gui-agent" ], "arxiv_id": "2606.13602", "source": "arxiv", "source_id": "arxiv:2606.13602", "pdf_url": "https://arxiv.org/pdf/2606.13602", "primary_query": "web-gui-agent" }, { "id": "2606.13192", "title": "Reasoning for Mobile User Experience with Multimodal LLMs: Task, Benchmark, and Approach", "url": "https://arxiv.org/abs/2606.13192", "published": "2026-06-11", "updated": "2026-06-11", "authors": [ "Ruichao Mao", "Zhou Fang", "Teng Guo", "Hao Yang", "Yaping Li", "Shaohua Peng", "Maji Huang", "Xiaoyu Lin", "Shuoyang Liu", "Xuepeng Li", "Yuyu Zhang", "Hai Rao" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "computer-use", "rag", "reasoning", "tool-use" ], "score": 14, "relevance": "high", "matched_queries": [ "web-gui-agent" ], "arxiv_id": "2606.13192", "source": "arxiv", "source_id": "arxiv:2606.13192", "pdf_url": "https://arxiv.org/pdf/2606.13192", "primary_query": "web-gui-agent" }, { "id": "2606.12674", "title": "Evoflux: Inference-Time Evolution of Executable Tool Workflows for Compact Agents", "url": "https://arxiv.org/abs/2606.12674", "published": "2026-06-10", "updated": "2026-06-10", "authors": [ "Kushal Raj Bhandari", "Ling Yue", "Ching-Yun Ko", "Dhaval Patel", "Shaowu Pan", "Pin-Yu Chen", "Jianxi Gao" ], "categories": [ "cs.AI" ], "topics": [ "agent-safety", "computer-use", "planning", "tool-use", "workflow-agent" ], "score": 14, "relevance": "high", "matched_queries": [ "function-calling", "tool-use" ], "arxiv_id": "2606.12674", "source": "arxiv", "source_id": "arxiv:2606.12674", "pdf_url": "https://arxiv.org/pdf/2606.12674", "primary_query": "function-calling" }, { "id": "2606.12384", "title": "APPO: Agentic Procedural Policy Optimization", "url": "https://arxiv.org/abs/2606.12384", "published": "2026-06-10", "updated": "2026-06-10", "authors": [ "Xucong Wang", "Ziyu Ma", "Yong Wang", "Yuxiang Ji", "Shidong Yang", "Guanhua Chen", "Pengkun Wang", "Xiangxiang Chu" ], "categories": [ "cs.LG", "cs.AI" ], "topics": [ "agent-evaluation", "tool-use", "workflow-agent" ], "score": 14, "relevance": "high", "matched_queries": [ "tool-use" ], "arxiv_id": "2606.12384", "source": "arxiv", "source_id": "arxiv:2606.12384", "pdf_url": "https://arxiv.org/pdf/2606.12384", "primary_query": "tool-use" }, { "id": "2606.12563", "title": "Arbor: Tree Search as a Cognition Layer for Autonomous Agents", "url": "https://arxiv.org/abs/2606.12563", "published": "2026-06-10", "updated": "2026-06-10", "authors": [ "Neha Prakriya", "Chaojun Hou", "Zheng Gong", "Huasha Zhao", "Xi Zhao", "Mou Li", "Zhenyu Gu", "Emad Barsoum" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "memory", "multi-agent", "tool-use" ], "score": 14, "relevance": "high", "matched_queries": [ "autonomous-agent-llm" ], "arxiv_id": "2606.12563", "source": "arxiv", "source_id": "arxiv:2606.12563", "pdf_url": "https://arxiv.org/pdf/2606.12563", "primary_query": "autonomous-agent-llm" }, { "id": "2606.11079", "title": "VISTA: A Versatile Interactive User Simulation Toolkit for Agent Evaluation", "url": "https://arxiv.org/abs/2606.11079", "published": "2026-06-09", "updated": "2026-06-09", "authors": [ "Yunan Lu", "Ryan Shea", "Yusen Zhang", "Zhou Yu" ], "categories": [ "cs.CL" ], "topics": [ "agent-evaluation", "rag", "tool-use", "world-model" ], "score": 14, "relevance": "high", "matched_queries": [ "agent-evaluation" ], "arxiv_id": "2606.11079", "source": "arxiv", "source_id": "arxiv:2606.11079", "pdf_url": "https://arxiv.org/pdf/2606.11079", "primary_query": "agent-evaluation" }, { "id": "2606.10742", "title": "MemVenom: Triggered Poisoning of Multimodal Memories in Web Agents", "url": "https://arxiv.org/abs/2606.10742", "published": "2026-06-09", "updated": "2026-06-09", "authors": [ "Yv Zhang", "Hao Sun", "Hao Fang", "Kuofeng Gao", "Fan Mo", "Bin Chen", "Shu-Tao Xia", "Yaowei Wang" ], "categories": [ "cs.CR", "cs.LG" ], "topics": [ "agent-evaluation", "computer-use", "memory", "planning", "rag", "reasoning" ], "score": 14, "relevance": "high", "matched_queries": [ "web-gui-agent" ], "arxiv_id": "2606.10742", "source": "arxiv", "source_id": "arxiv:2606.10742", "pdf_url": "https://arxiv.org/pdf/2606.10742", "primary_query": "web-gui-agent" }, { "id": "2606.09774", "title": "Auto-Configuring Scientific Simulators with Lightweight Coding-Agent Adapters", "url": "https://arxiv.org/abs/2606.09774", "published": "2026-06-08", "updated": "2026-06-25", "authors": [ "Matthew Ho", "Brian Liu", "Jixuan Chen", "Audrey Wang", "Lianhui Qin" ], "categories": [ "cs.AI", "cs.CL" ], "topics": [ "agent-evaluation", "agent-safety", "coding-agent", "memory", "rag", "tool-use", "world-model" ], "score": 14, "relevance": "high", "matched_queries": [ "agent-memory" ], "arxiv_id": "2606.09774", "source": "arxiv", "source_id": "arxiv:2606.09774", "pdf_url": "https://arxiv.org/pdf/2606.09774", "primary_query": "agent-memory" }, { "id": "2606.09198", "title": "MASS: Deep Research for Social Sciences with Memory-Augmented Social Simulation", "url": "https://arxiv.org/abs/2606.09198", "published": "2026-06-08", "updated": "2026-06-08", "authors": [ "Yongrui Liu", "Deyi Xiong" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "computer-use", "memory", "planning", "rag", "world-model" ], "score": 14, "relevance": "high", "matched_queries": [ "agent-memory" ], "arxiv_id": "2606.09198", "source": "arxiv", "source_id": "arxiv:2606.09198", "pdf_url": "https://arxiv.org/pdf/2606.09198", "primary_query": "agent-memory" }, { "id": "2606.08790", "title": "RAILS: Verification-Native Clearing For Agentic Commerce", "url": "https://arxiv.org/abs/2606.08790", "published": "2026-06-07", "updated": "2026-06-07", "authors": [ "Adrian de Valois-Franklin", "Alex Bogdan" ], "categories": [ "cs.AI", "cs.CR", "cs.MA" ], "topics": [ "agent-evaluation", "agent-safety", "tool-use" ], "score": 14, "relevance": "high", "matched_queries": [ "autonomous-agent-llm" ], "arxiv_id": "2606.08790", "source": "arxiv", "source_id": "arxiv:2606.08790", "pdf_url": "https://arxiv.org/pdf/2606.08790", "primary_query": "autonomous-agent-llm" }, { "id": "2606.08625", "title": "From Holistic Evaluation to Structured Criteria: Rubrics Across the Evolving LLM Landscape", "url": "https://arxiv.org/abs/2606.08625", "published": "2026-06-07", "updated": "2026-07-01", "authors": [ "Hao Chen", "Ziyu Han", "Yukun Yan", "Qingfu Zhu", "Maosong Sun", "Wanxiang Che" ], "categories": [ "cs.CL" ], "topics": [ "agent-evaluation", "agent-safety", "computer-use", "reasoning", "tool-use" ], "score": 14, "relevance": "high", "matched_queries": [ "autonomous-agent-llm" ], "arxiv_id": "2606.08625", "source": "arxiv", "source_id": "arxiv:2606.08625", "pdf_url": "https://arxiv.org/pdf/2606.08625", "primary_query": "autonomous-agent-llm" }, { "id": "2606.07379", "title": "Do Coding Agents Deceive Us? Detecting and Preventing Cheating via Capped Evaluation with Randomized Tests", "url": "https://arxiv.org/abs/2606.07379", "published": "2026-06-05", "updated": "2026-06-08", "authors": [ "Thanawat Lodkaew", "Johannes Ackermann", "Soichiro Nishimori", "Nontawat Charoenphakdee", "Masashi Sugiyama", "Takashi Ishida" ], "categories": [ "cs.LG", "cs.AI", "cs.CL", "stat.ME" ], "topics": [ "agent-evaluation", "coding-agent", "rag" ], "score": 14, "relevance": "high", "matched_queries": [ "agent-evaluation" ], "arxiv_id": "2606.07379", "source": "arxiv", "source_id": "arxiv:2606.07379", "pdf_url": "https://arxiv.org/pdf/2606.07379", "primary_query": "agent-evaluation" }, { "id": "2606.05548", "title": "ADK Arena: Evaluating Agent Development Kits via LLM-as-a-Developer", "url": "https://arxiv.org/abs/2606.05548", "published": "2026-06-04", "updated": "2026-06-04", "authors": [ "Jintao Huang", "Xiaomin Li", "Gaurav Mittal", "Yu Hu" ], "categories": [ "cs.SE", "cs.AI" ], "topics": [ "agent-evaluation", "coding-agent", "tool-use" ], "score": 14, "relevance": "high", "matched_queries": [ "agent-evaluation", "autonomous-agent-llm" ], "arxiv_id": "2606.05548", "source": "arxiv", "source_id": "arxiv:2606.05548", "pdf_url": "https://arxiv.org/pdf/2606.05548", "primary_query": "agent-evaluation" }, { "id": "2606.06473", "title": "MLEvolve: A Self-Evolving Framework for Automated Machine Learning Algorithm Discovery", "url": "https://arxiv.org/abs/2606.06473", "published": "2026-06-04", "updated": "2026-06-04", "authors": [ "Shangheng Du", "Xiangchao Yan", "Jinxin Shi", "Zongsheng Cao", "Shiyang Feng", "Zichen Liang", "Boyuan Sun", "Tianshuo Peng", "Yifan Zhou", "Xin Li", "Jie Zhou", "Liang He", "Bo Zhang", "Lei Bai" ], "categories": [ "cs.AI", "cs.CL" ], "topics": [ "agent-evaluation", "coding-agent", "memory", "multi-agent", "planning", "rag" ], "score": 14, "relevance": "high", "matched_queries": [ "planning-agent" ], "arxiv_id": "2606.06473", "source": "arxiv", "source_id": "arxiv:2606.06473", "pdf_url": "https://arxiv.org/pdf/2606.06473", "primary_query": "planning-agent" }, { "id": "2606.06462", "title": "Benchmark Everything Everywhere All at Once", "url": "https://arxiv.org/abs/2606.06462", "published": "2026-06-04", "updated": "2026-06-04", "authors": [ "Shiyun Xiong", "Dongming Wu", "Peiwen Sun", "Yuang Ai", "Bokang Yang", "Wencheng Han", "Xiao-Hui Li", "Xiangyu Yue" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "coding-agent", "reasoning", "tool-use" ], "score": 14, "relevance": "high", "matched_queries": [ "autonomous-agent-llm" ], "arxiv_id": "2606.06462", "source": "arxiv", "source_id": "arxiv:2606.06462", "pdf_url": "https://arxiv.org/pdf/2606.06462", "primary_query": "autonomous-agent-llm" }, { "id": "2606.05263", "title": "Policy-Conditioned Counterfactual Credit for Verifiable Reinforcement Learning of Long-Horizon Language Agents", "url": "https://arxiv.org/abs/2606.05263", "published": "2026-06-03", "updated": "2026-06-03", "authors": [ "Renwei Meng" ], "categories": [ "cs.LG", "cs.AI" ], "topics": [ "agent-evaluation", "planning", "rag", "reasoning", "tool-use" ], "score": 14, "relevance": "high", "matched_queries": [ "language-agent" ], "arxiv_id": "2606.05263", "source": "arxiv", "source_id": "arxiv:2606.05263", "pdf_url": "https://arxiv.org/pdf/2606.05263", "primary_query": "language-agent" }, { "id": "2606.04628", "title": "RAMPART: Registry-based Agentic Memory with Priority-Aware Runtime Transformation", "url": "https://arxiv.org/abs/2606.04628", "published": "2026-06-03", "updated": "2026-06-03", "authors": [ "Nikodem Tomczak" ], "categories": [ "cs.CL", "cs.MA" ], "topics": [ "memory" ], "score": 14, "relevance": "high", "matched_queries": [ "agent-memory" ], "arxiv_id": "2606.04628", "source": "arxiv", "source_id": "arxiv:2606.04628", "pdf_url": "https://arxiv.org/pdf/2606.04628", "primary_query": "agent-memory" }, { "id": "2606.05414", "title": "When Evidence is Sparse: Weakly Supervised Early Failure Alerting in Dialogs and LLM-Agent Trajectories", "url": "https://arxiv.org/abs/2606.05414", "published": "2026-06-03", "updated": "2026-06-03", "authors": [ "Avinash Baidya", "Xinran Liang", "Ruocheng Guo", "Xiang Gao", "Kamalika Das" ], "categories": [ "cs.CL", "cs.AI", "cs.HC", "cs.LG" ], "topics": [ "agent-evaluation", "agent-safety", "planning", "rag", "tool-use" ], "score": 14, "relevance": "high", "matched_queries": [ "planning-agent" ], "arxiv_id": "2606.05414", "source": "arxiv", "source_id": "arxiv:2606.05414", "pdf_url": "https://arxiv.org/pdf/2606.05414", "primary_query": "planning-agent" }, { "id": "2606.03329", "title": "InfoMem: Training Long-Context Memory Agents with Answer-Conditioned Information Gain", "url": "https://arxiv.org/abs/2606.03329", "published": "2026-06-02", "updated": "2026-06-02", "authors": [ "Tiancheng Han", "Yong Li", "Wuzhou Yu", "Qiaosheng Zhang", "Wenqi Shao" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "memory", "rag", "tool-use" ], "score": 14, "relevance": "high", "matched_queries": [ "agent-memory" ], "arxiv_id": "2606.03329", "source": "arxiv", "source_id": "arxiv:2606.03329", "pdf_url": "https://arxiv.org/pdf/2606.03329", "primary_query": "agent-memory" }, { "id": "2606.02965", "title": "What Benchmarks Don't Measure: The Case for Evaluating Abstention Competence in Autonomous Agents", "url": "https://arxiv.org/abs/2606.02965", "published": "2026-06-01", "updated": "2026-06-01", "authors": [ "Victor Ojewale", "Suresh Venkatasubramanian" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "agent-safety", "computer-use", "tool-use", "workflow-agent" ], "score": 14, "relevance": "high", "matched_queries": [ "agent-evaluation" ], "arxiv_id": "2606.02965", "source": "arxiv", "source_id": "arxiv:2606.02965", "pdf_url": "https://arxiv.org/pdf/2606.02965", "primary_query": "agent-evaluation" }, { "id": "2606.02404", "title": "K-BrowseComp: A Web Browsing Agent Benchmark Grounded in Korean Contexts", "url": "https://arxiv.org/abs/2606.02404", "published": "2026-06-01", "updated": "2026-06-01", "authors": [ "Nahyun Lee", "Dongkeun Yoon", "Guijin Son", "Geewook Kim", "Dayoon Ko", "Jeonghun Park", "Haneul Yoo", "Jaewon Cho", "Junghun Park", "Changyoon Lee", "Kyochul Jang", "Jaeyeon Kim", "Eunsu Kim", "Woojin Cho", "Seungone Kim" ], "categories": [ "cs.CL" ], "topics": [ "agent-evaluation", "reasoning" ], "score": 14, "relevance": "high", "matched_queries": [ "agent-evaluation" ], "arxiv_id": "2606.02404", "source": "arxiv", "source_id": "arxiv:2606.02404", "pdf_url": "https://arxiv.org/pdf/2606.02404", "primary_query": "agent-evaluation" }, { "id": "2606.09863", "title": "From Confident Closing to Silent Failure: Characterizing False Success in LLM Agents", "url": "https://arxiv.org/abs/2606.09863", "published": "2026-06-01", "updated": "2026-06-01", "authors": [ "Laksh Advani" ], "categories": [ "cs.LG" ], "topics": [ "agent-evaluation", "coding-agent", "tool-use" ], "score": 14, "relevance": "high", "matched_queries": [ "agent-evaluation" ], "arxiv_id": "2606.09863", "source": "arxiv", "source_id": "arxiv:2606.09863", "pdf_url": "https://arxiv.org/pdf/2606.09863", "primary_query": "agent-evaluation" }, { "id": "2606.00611", "title": "TRACE: Trajectory Risk-Aware Compression for Long-Horizon Agent Safety", "url": "https://arxiv.org/abs/2606.00611", "published": "2026-05-30", "updated": "2026-05-30", "authors": [ "Zhepei Hong", "Lin Wang", "Liting Li", "Haokai Ma", "Junfeng Fang", "Fei Shen", "Dan Zhang", "Xiang Wang" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "agent-safety", "planning" ], "score": 14, "relevance": "high", "matched_queries": [ "agent-safety" ], "arxiv_id": "2606.00611", "source": "arxiv", "source_id": "arxiv:2606.00611", "pdf_url": "https://arxiv.org/pdf/2606.00611", "primary_query": "agent-safety" }, { "id": "2606.00198", "title": "BAGEN: Are LLM Agents Budget-Aware?", "url": "https://arxiv.org/abs/2606.00198", "published": "2026-05-29", "updated": "2026-05-29", "authors": [ "Yuxiang Lin", "Zihan Wang", "Mengyang Liu", "Yuxuan Shan", "Longju Bai", "Junyao Zhang", "Xing Jin", "Boshan Chen", "Jinyan Su", "Xingyao Wang", "Jiaxin Pei", "Manling Li" ], "categories": [ "cs.LG", "cs.AI", "cs.CL" ], "topics": [ "agent-evaluation", "planning", "rag", "tool-use" ], "score": 14, "relevance": "high", "matched_queries": [ "planning-agent" ], "arxiv_id": "2606.00198", "source": "arxiv", "source_id": "arxiv:2606.00198", "pdf_url": "https://arxiv.org/pdf/2606.00198", "primary_query": "planning-agent" }, { "id": "2605.29790", "title": "Evolve as a Team: Collaborative Self-Evolution for LLM-based Multi-Agent Systems", "url": "https://arxiv.org/abs/2605.29790", "published": "2026-05-28", "updated": "2026-05-28", "authors": [ "Zhezheng Hao", "Tianfu Wang", "Huanshuo Dong", "Ziyan Liu", "Hong Wang", "Xiankun Lin", "Qiang Lin", "Can Wang", "Hande Dong", "Jiawei Chen" ], "categories": [ "cs.MA", "cs.AI" ], "topics": [ "agent-evaluation", "multi-agent", "planning" ], "score": 14, "relevance": "high", "matched_queries": [ "agent-evaluation" ], "arxiv_id": "2605.29790", "source": "arxiv", "source_id": "arxiv:2605.29790", "pdf_url": "https://arxiv.org/pdf/2605.29790", "primary_query": "agent-evaluation" }, { "id": "2605.29640", "title": "VikingMem: A Memory Base Management System for Stateful LLM-based Applications", "url": "https://arxiv.org/abs/2605.29640", "published": "2026-05-28", "updated": "2026-06-12", "authors": [ "Jiajie Fu", "Junwen Chen", "Mengzhao Wang", "Aoxiang He", "Maojia Sheng", "Xiangyu Ke", "Yifan Zhu", "Yunjun Gao" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "memory", "rag", "tool-use" ], "score": 14, "relevance": "high", "matched_queries": [ "agent-memory" ], "arxiv_id": "2605.29640", "source": "arxiv", "source_id": "arxiv:2605.29640", "pdf_url": "https://arxiv.org/pdf/2605.29640", "primary_query": "agent-memory" }, { "id": "2605.29801", "title": "AgentDoG 1.5: A Lightweight and Scalable Alignment Framework for AI Agent Safety and Security", "url": "https://arxiv.org/abs/2605.29801", "published": "2026-05-28", "updated": "2026-05-28", "authors": [ "Dongrui Liu", "Yu Li", "Zhonghao Yang", "Peng Wang", "Guanxu Chen", "Yuejin Xie", "Qinghua Mao", "Wanying Qu", "Yanxu Zhu", "Tianyi Zhou", "Leitao Yuan", "Zhijie Zheng", "Qihao Lin", "Yimin Wang", "Haoyu Luo", "Shuai Shao", "Chen Qian", "Qingyu Liu", "Ling Tang", "Ruiyang Qin", "Qihan Ren", "Junxiao Yang", "Kun Wang", "Zhiheng Xi", "Linfeng Zhang", "Ranjie Duan", "Bo Zhang", "Wenjie Wang", "Wen Shen", "Qiaosheng Zhang", "Yan Teng", "Chaochao Lu", "Rui Mei", "Man Li", "Jialing Tao", "Xi Lin", "Tianhang Zheng", "Yong Liu", "Quanshi Zhang", "Lei Zhu", "Xingjun Ma", "Junhua Liu", "Hui Xue", "Xiaoxiang Zuo", "Xiangnan He", "Chao Shen", "Xianglong Liu", "Minlie Huang", "Jing Shao", "Xia Hu" ], "categories": [ "cs.AI", "cs.CL", "cs.CR", "cs.CV", "cs.LG" ], "topics": [ "agent-safety", "computer-use" ], "score": 14, "relevance": "high", "matched_queries": [ "agent-safety" ], "arxiv_id": "2605.29801", "source": "arxiv", "source_id": "arxiv:2605.29801", "pdf_url": "https://arxiv.org/pdf/2605.29801", "primary_query": "agent-safety" }, { "id": "2605.30407", "title": "Exploring Autonomous Agentic Data Engineering for Model Specialization", "url": "https://arxiv.org/abs/2605.30407", "published": "2026-05-28", "updated": "2026-06-08", "authors": [ "Yujie Luo", "Xiangyuan Ru", "Jingsheng Zheng", "Jingjing Wang", "Yuqi Zhu", "Jintian Zhang", "Runnan Fang", "Kewei Xu", "Ye Liu", "Zheng Wei", "Jiang Bian", "Zang Li", "Shumin Deng" ], "categories": [ "cs.CL", "cs.AI", "cs.IR", "cs.LG" ], "topics": [ "agent-evaluation", "computer-use", "planning", "workflow-agent" ], "score": 14, "relevance": "high", "matched_queries": [ "autonomous-agent-llm" ], "arxiv_id": "2605.30407", "source": "arxiv", "source_id": "arxiv:2605.30407", "pdf_url": "https://arxiv.org/pdf/2605.30407", "primary_query": "autonomous-agent-llm" }, { "id": "2605.27825", "title": "MRMMIA: Membership Inference Attacks on Memory in Chat Agents", "url": "https://arxiv.org/abs/2605.27825", "published": "2026-05-27", "updated": "2026-05-27", "authors": [ "Kai Chen", "Yan Pang", "Tianhao Wang" ], "categories": [ "cs.CR", "cs.LG" ], "topics": [ "agent-evaluation", "agent-safety", "memory", "rag", "tool-use" ], "score": 14, "relevance": "high", "matched_queries": [ "agent-memory" ], "arxiv_id": "2605.27825", "source": "arxiv", "source_id": "arxiv:2605.27825", "pdf_url": "https://arxiv.org/pdf/2605.27825", "primary_query": "agent-memory" }, { "id": "2605.28175", "title": "Mixture-of-Experts Knowledge Graph Retrieval-Augmented Generation for Multi-Agent LLM-based Recommendation", "url": "https://arxiv.org/abs/2605.28175", "published": "2026-05-27", "updated": "2026-05-29", "authors": [ "Shijie Wang", "Chengyi Liu", "Yujuan Ding", "Shanru Lin", "See-Kiong Ng", "Xu Xin", "Wenqi Fan" ], "categories": [ "cs.IR" ], "topics": [ "agent-evaluation", "agent-safety", "multi-agent", "rag" ], "score": 14, "relevance": "high", "matched_queries": [ "rag-agent" ], "arxiv_id": "2605.28175", "source": "arxiv", "source_id": "arxiv:2605.28175", "pdf_url": "https://arxiv.org/pdf/2605.28175", "primary_query": "rag-agent" }, { "id": "2605.27935", "title": "Do Agents Think Deeper? A Mechanistic Investigation of Layer-Wise Dynamics in Sequential Planning", "url": "https://arxiv.org/abs/2605.27935", "published": "2026-05-27", "updated": "2026-05-27", "authors": [ "Zhenyu Cui", "Xiangzhong Luo" ], "categories": [ "cs.AI" ], "topics": [ "planning", "reasoning", "tool-use" ], "score": 14, "relevance": "high", "matched_queries": [ "autonomous-agent-llm", "planning-agent" ], "arxiv_id": "2605.27935", "source": "arxiv", "source_id": "arxiv:2605.27935", "pdf_url": "https://arxiv.org/pdf/2605.27935", "primary_query": "autonomous-agent-llm" }, { "id": "2605.25920", "title": "Can LLMs Time Travel? Enhancing Temporal Consistency in Legal Agentic Search through Reinforcement Learning", "url": "https://arxiv.org/abs/2605.25920", "published": "2026-05-25", "updated": "2026-05-25", "authors": [ "Wei Fan", "Yining Zhou", "Mufan Zhang", "Yanbing Weng", "Yiran HU", "Tianshi Zheng", "Baixuan Xu", "Chunyang Li", "Jianhui Yang", "Haoran Li", "Yangqiu Song" ], "categories": [ "cs.CL", "cs.AI" ], "topics": [ "agent-evaluation", "rag", "reasoning" ], "score": 14, "relevance": "high", "matched_queries": [ "rag-agent" ], "arxiv_id": "2605.25920", "source": "arxiv", "source_id": "arxiv:2605.25920", "pdf_url": "https://arxiv.org/pdf/2605.25920", "primary_query": "rag-agent" }, { "id": "2605.25393", "title": "Decision-Making with Lightweight Confidence-Aware Language Model for Autonomous Driving", "url": "https://arxiv.org/abs/2605.25393", "published": "2026-05-25", "updated": "2026-05-25", "authors": [ "Ruoyu Yao", "Ruiguo Zhong", "Pei Liu", "Mingxing Peng", "Rui Yang", "Jun Ma" ], "categories": [ "cs.RO" ], "topics": [ "agent-evaluation", "multi-agent", "planning", "rag", "reasoning", "tool-use", "workflow-agent" ], "score": 14, "relevance": "high", "matched_queries": [ "rag-agent" ], "arxiv_id": "2605.25393", "source": "arxiv", "source_id": "arxiv:2605.25393", "pdf_url": "https://arxiv.org/pdf/2605.25393", "primary_query": "rag-agent" }, { "id": "2605.25310", "title": "Tool-Call Dependency Structure is Linearly Decodable in LLM Agent Residual Streams", "url": "https://arxiv.org/abs/2605.25310", "published": "2026-05-25", "updated": "2026-05-25", "authors": [ "Tianda Sun", "Dimitar Kazakov" ], "categories": [ "cs.CL" ], "topics": [ "agent-evaluation", "planning", "tool-use" ], "score": 14, "relevance": "high", "matched_queries": [ "planning-agent" ], "arxiv_id": "2605.25310", "source": "arxiv", "source_id": "arxiv:2605.25310", "pdf_url": "https://arxiv.org/pdf/2605.25310", "primary_query": "planning-agent" }, { "id": "2605.24309", "title": "Reframing LLM Agent Security as an Agent-Human Interaction Problem", "url": "https://arxiv.org/abs/2605.24309", "published": "2026-05-23", "updated": "2026-05-23", "authors": [ "Peiran Wang", "Ying Li", "Yuan Tian" ], "categories": [ "cs.CR" ], "topics": [ "agent-evaluation", "agent-safety", "tool-use" ], "score": 14, "relevance": "high", "matched_queries": [ "agent-safety" ], "arxiv_id": "2605.24309", "source": "arxiv", "source_id": "arxiv:2605.24309", "pdf_url": "https://arxiv.org/pdf/2605.24309", "primary_query": "agent-safety" }, { "id": "2605.19604", "title": "Formal Skill: Programmable Runtime Skills for Efficient and Accurate LLM Agents", "url": "https://arxiv.org/abs/2605.19604", "published": "2026-05-19", "updated": "2026-05-19", "authors": [ "Xi Zhang", "Meijun Gao", "Yuntian Zhao", "Xinyu Tan", "Yilun Yao", "Feiyu Wang", "Yanshu Wang", "Dingsiyi", "Tong Yang" ], "categories": [ "cs.AI" ], "topics": [ "rag", "reasoning", "tool-use", "workflow-agent" ], "score": 14, "relevance": "high", "matched_queries": [ "function-calling" ], "arxiv_id": "2605.19604", "source": "arxiv", "source_id": "arxiv:2605.19604", "pdf_url": "https://arxiv.org/pdf/2605.19604", "primary_query": "function-calling" }, { "id": "2605.18502", "title": "The distance-based formation controller design for multi-agent systems in port-Hamiltonian form", "url": "https://arxiv.org/abs/2605.18502", "published": "2026-05-18", "updated": "2026-05-18", "authors": [ "Jingyi Zhao", "Yongxin Wu", "Héctor García de Marina", "Yuhu Wu", "Yann Le Gorrec" ], "categories": [ "math.OC" ], "topics": [ "agent-evaluation", "agent-safety", "multi-agent", "tool-use", "world-model" ], "score": 14, "relevance": "high", "matched_queries": [ "agent-safety" ], "arxiv_id": "2605.18502", "source": "arxiv", "source_id": "arxiv:2605.18502", "pdf_url": "https://arxiv.org/pdf/2605.18502", "primary_query": "agent-safety" }, { "id": "2605.15701", "title": "H-Mem: A Novel Memory Mechanism for Evolving and Retrieving Agent Memory via a Hybrid Structure", "url": "https://arxiv.org/abs/2605.15701", "published": "2026-05-15", "updated": "2026-05-15", "authors": [ "Jiawei Yu", "Yixiang Fang", "Xilin Liu", "Yuchi Ma" ], "categories": [ "cs.CL", "cs.AI" ], "topics": [ "agent-evaluation", "memory", "rag" ], "score": 14, "relevance": "high", "matched_queries": [ "agent-memory" ], "arxiv_id": "2605.15701", "source": "arxiv", "source_id": "arxiv:2605.15701", "pdf_url": "https://arxiv.org/pdf/2605.15701", "primary_query": "agent-memory" }, { "id": "2605.15625", "title": "ColPackAgent: Agent-Skill-Guided Hard-Particle Monte Carlo Workflows for Colloidal Packing", "url": "https://arxiv.org/abs/2605.15625", "published": "2026-05-15", "updated": "2026-05-15", "authors": [ "Lijie Ding", "Changwoo Do" ], "categories": [ "cs.AI", "cond-mat.soft" ], "topics": [ "agent-evaluation", "computer-use", "planning", "tool-use", "workflow-agent", "world-model" ], "score": 14, "relevance": "high", "matched_queries": [ "planning-agent" ], "arxiv_id": "2605.15625", "source": "arxiv", "source_id": "arxiv:2605.15625", "pdf_url": "https://arxiv.org/pdf/2605.15625", "primary_query": "planning-agent" }, { "id": "2605.14290", "title": "Web Agents Should Adopt the Plan-Then-Execute Paradigm", "url": "https://arxiv.org/abs/2605.14290", "published": "2026-05-14", "updated": "2026-05-14", "authors": [ "Julien Piet", "Annabella Chow", "Yiwei Hou", "Muxi Lyu", "Sylvie Venuto", "Jinhao Zhu", "Raluca Ada Popa", "David Wagner" ], "categories": [ "cs.CR", "cs.AI", "cs.CL", "cs.SE" ], "topics": [ "agent-evaluation", "agent-safety", "computer-use", "planning", "tool-use" ], "score": 14, "relevance": "high", "matched_queries": [ "planning-agent" ], "arxiv_id": "2605.14290", "source": "arxiv", "source_id": "arxiv:2605.14290", "pdf_url": "https://arxiv.org/pdf/2605.14290", "primary_query": "planning-agent" }, { "id": "2605.13716", "title": "SkillOps: Managing LLM Agent Skill Libraries as Self-Maintaining Software Ecosystems", "url": "https://arxiv.org/abs/2605.13716", "published": "2026-05-13", "updated": "2026-05-13", "authors": [ "Hongji Pu", "Xinyuan Song", "Liang Zhao" ], "categories": [ "cs.SE", "cs.MA" ], "topics": [ "agent-evaluation", "agent-safety", "planning", "rag" ], "score": 14, "relevance": "high", "matched_queries": [ "planning-agent" ], "arxiv_id": "2605.13716", "source": "arxiv", "source_id": "arxiv:2605.13716", "pdf_url": "https://arxiv.org/pdf/2605.13716", "primary_query": "planning-agent" }, { "id": "2605.13618", "title": "OpenAaaS: An Open Agent-as-a-Service Framework for Distributed Materials-Informatics Research", "url": "https://arxiv.org/abs/2605.13618", "published": "2026-05-13", "updated": "2026-05-13", "authors": [ "Peng Kang", "Bixuan Li", "Xiaoya Huang", "Shuo Shi", "Weiqiao Zhou", "Zhen Li", "Yu Liu", "Lei Zheng" ], "categories": [ "cond-mat.mtrl-sci", "cs.AI" ], "topics": [ "memory", "multi-agent", "planning", "rag", "reasoning", "workflow-agent" ], "score": 14, "relevance": "high", "matched_queries": [ "autonomous-agent-llm" ], "arxiv_id": "2605.13618", "source": "arxiv", "source_id": "arxiv:2605.13618", "pdf_url": "https://arxiv.org/pdf/2605.13618", "primary_query": "autonomous-agent-llm" }, { "id": "2605.07112", "title": "Switchcraft: AI Model Router for Agentic Tool Calling", "url": "https://arxiv.org/abs/2605.07112", "published": "2026-05-08", "updated": "2026-05-08", "authors": [ "Sharad Agarwal", "Pooria Namyar", "Alec Wolman", "Rahul Ambavat", "Ankur Gupta", "Qizheng Zhang" ], "categories": [ "cs.AI", "cs.MA" ], "topics": [ "agent-evaluation", "reasoning", "tool-use" ], "score": 14, "relevance": "high", "matched_queries": [ "function-calling" ], "arxiv_id": "2605.07112", "source": "arxiv", "source_id": "arxiv:2605.07112", "pdf_url": "https://arxiv.org/pdf/2605.07112", "primary_query": "function-calling" }, { "id": "2605.06992", "title": "Why Does Agentic Safety Fail to Generalize Across Tasks?", "url": "https://arxiv.org/abs/2605.06992", "published": "2026-05-07", "updated": "2026-05-07", "authors": [ "Yonatan Slutzky", "Yotam Alexander", "Tomer Slor", "Yoav Nagel", "Nadav Cohen" ], "categories": [ "cs.LG", "stat.ML" ], "topics": [ "agent-safety", "embodied-agent" ], "score": 14, "relevance": "high", "matched_queries": [ "agent-safety" ], "arxiv_id": "2605.06992", "source": "arxiv", "source_id": "arxiv:2605.06992", "pdf_url": "https://arxiv.org/pdf/2605.06992", "primary_query": "agent-safety" }, { "id": "2605.06957", "title": "Learning and Reusing Policy Decompositions for Hierarchical Generalized Planning with LLM Agents", "url": "https://arxiv.org/abs/2605.06957", "published": "2026-05-07", "updated": "2026-05-07", "authors": [ "Shirin Sohrabi", "Haritha Ananthakrishnan", "Harsha Kokel", "Kavitha Srinivas", "Michael Katz" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "planning", "rag" ], "score": 14, "relevance": "high", "matched_queries": [ "planning-agent" ], "arxiv_id": "2605.06957", "source": "arxiv", "source_id": "arxiv:2605.06957", "pdf_url": "https://arxiv.org/pdf/2605.06957", "primary_query": "planning-agent" }, { "id": "2605.06737", "title": "A Self-Healing Framework for Reliable LLM-Based Autonomous Agents", "url": "https://arxiv.org/abs/2605.06737", "published": "2026-05-07", "updated": "2026-05-07", "authors": [ "Cheonsu Jeong", "Younggun Shin" ], "categories": [ "cs.SE", "cs.AI" ], "topics": [ "agent-evaluation", "computer-use", "multi-agent", "planning", "reasoning", "workflow-agent" ], "score": 14, "relevance": "high", "matched_queries": [ "autonomous-agent-llm" ], "arxiv_id": "2605.06737", "source": "arxiv", "source_id": "arxiv:2605.06737", "pdf_url": "https://arxiv.org/pdf/2605.06737", "primary_query": "autonomous-agent-llm" }, { "id": "2604.27464", "title": "Security Attack and Defense Strategies for Autonomous Agent Frameworks: A Layered Review with OpenClaw as a Case Study", "url": "https://arxiv.org/abs/2604.27464", "published": "2026-04-30", "updated": "2026-04-30", "authors": [ "Luyao Xu", "Xiang Chen" ], "categories": [ "cs.CR", "cs.AI" ], "topics": [ "agent-evaluation", "agent-safety", "planning", "tool-use", "workflow-agent" ], "score": 14, "relevance": "high", "matched_queries": [ "agent-safety", "autonomous-agent-llm" ], "arxiv_id": "2604.27464", "source": "arxiv", "source_id": "arxiv:2604.27464", "pdf_url": "https://arxiv.org/pdf/2604.27464", "primary_query": "agent-safety" }, { "id": "2604.28157", "title": "FlashRT: Towards Computationally and Memory Efficient Red-Teaming for Prompt Injection and Knowledge Corruption", "url": "https://arxiv.org/abs/2604.28157", "published": "2026-04-30", "updated": "2026-04-30", "authors": [ "Yanting Wang", "Chenlong Yin", "Ying Chen", "Jinyuan Jia" ], "categories": [ "cs.CR" ], "topics": [ "agent-evaluation", "agent-safety", "memory", "rag", "tool-use" ], "score": 14, "relevance": "high", "matched_queries": [ "autonomous-agent-llm" ], "arxiv_id": "2604.28157", "source": "arxiv", "source_id": "arxiv:2604.28157", "pdf_url": "https://arxiv.org/pdf/2604.28157", "primary_query": "autonomous-agent-llm" }, { "id": "2604.27859", "title": "Rethinking Agentic Reinforcement Learning In Large Language Models", "url": "https://arxiv.org/abs/2604.27859", "published": "2026-04-30", "updated": "2026-05-15", "authors": [ "Fangming Cui", "Ruixiao Zhu", "Cheng Fang", "Sunan Li", "Jiahong Li" ], "categories": [ "cs.AI", "cs.ET" ], "topics": [ "memory", "planning", "reasoning", "tool-use" ], "score": 14, "relevance": "high", "matched_queries": [ "autonomous-agent-llm" ], "arxiv_id": "2604.27859", "source": "arxiv", "source_id": "arxiv:2604.27859", "pdf_url": "https://arxiv.org/pdf/2604.27859", "primary_query": "autonomous-agent-llm" }, { "id": "2606.20573", "title": "AONA: A Comprehensive Architecture and Workflow Design for Global Agentic Collaboration", "url": "https://arxiv.org/abs/2606.20573", "published": "2026-04-30", "updated": "2026-04-30", "authors": [ "Jinliang Xu", "Runkai Zhu", "Bingqi Li", "Fanjie Nie", "Jin Li", "Jiagui Xie" ], "categories": [ "cs.NI", "cs.MA" ], "topics": [ "multi-agent", "tool-use", "workflow-agent" ], "score": 14, "relevance": "high", "matched_queries": [ "autonomous-agent-llm" ], "arxiv_id": "2606.20573", "source": "arxiv", "source_id": "arxiv:2606.20573", "pdf_url": "https://arxiv.org/pdf/2606.20573", "primary_query": "autonomous-agent-llm" }, { "id": "2604.25555", "title": "From CRUD to Autonomous Agents: Formal Validation and Zero-Trust Security for Semantic Gateways in AI-Native Enterprise Systems", "url": "https://arxiv.org/abs/2604.25555", "published": "2026-04-28", "updated": "2026-04-28", "authors": [ "Ignacio Peyrano" ], "categories": [ "cs.CR", "cs.AI" ], "topics": [ "agent-evaluation", "agent-safety", "coding-agent", "tool-use", "workflow-agent" ], "score": 14, "relevance": "high", "matched_queries": [ "autonomous-agent-llm" ], "arxiv_id": "2604.25555", "source": "arxiv", "source_id": "arxiv:2604.25555", "pdf_url": "https://arxiv.org/pdf/2604.25555", "primary_query": "autonomous-agent-llm" }, { "id": "2604.21190", "title": "SpatiO: Adaptive Test-Time Orchestration of Vision-Language Agents for Spatial Reasoning", "url": "https://arxiv.org/abs/2604.21190", "published": "2026-04-23", "updated": "2026-06-27", "authors": [ "Chan Yeong Hwang", "Miso Choi", "Sunghyun On", "Jinkyu Kim", "Jungbeom Lee" ], "categories": [ "cs.CV" ], "topics": [ "agent-evaluation", "multi-agent", "rag", "reasoning" ], "score": 14, "relevance": "high", "matched_queries": [ "language-agent" ], "arxiv_id": "2604.21190", "source": "arxiv", "source_id": "arxiv:2604.21190", "pdf_url": "https://arxiv.org/pdf/2604.21190", "primary_query": "language-agent" }, { "id": "2604.16706", "title": "Evaluating Tool-Using Language Agents: Judge Reliability, Propagation Cascades, and Runtime Mitigation in AgentProp-Bench", "url": "https://arxiv.org/abs/2604.16706", "published": "2026-04-17", "updated": "2026-04-17", "authors": [ "Bhaskar Gurram" ], "categories": [ "cs.AI", "cs.CL", "cs.MA" ], "topics": [ "agent-evaluation", "tool-use" ], "score": 14, "relevance": "high", "matched_queries": [ "language-agent" ], "arxiv_id": "2604.16706", "source": "arxiv", "source_id": "arxiv:2604.16706", "pdf_url": "https://arxiv.org/pdf/2604.16706", "primary_query": "language-agent" }, { "id": "2604.15579", "title": "Don't Make Models Guess Security and Safety: Symbolic Guardrails for Domain-Specific AI Agents", "url": "https://arxiv.org/abs/2604.15579", "published": "2026-04-16", "updated": "2026-07-05", "authors": [ "Yining Hong", "Yining She", "Eunsuk Kang", "Christopher S. Timperley", "Christian Kästner" ], "categories": [ "cs.SE", "cs.AI", "cs.CR" ], "topics": [ "agent-evaluation", "agent-safety", "coding-agent", "tool-use" ], "score": 14, "relevance": "high", "matched_queries": [ "agent-safety" ], "arxiv_id": "2604.15579", "source": "arxiv", "source_id": "arxiv:2604.15579", "pdf_url": "https://arxiv.org/pdf/2604.15579", "primary_query": "agent-safety" }, { "id": "2604.15415", "title": "HarmfulSkillBench: How Do Harmful Skills Weaponize Your Agents?", "url": "https://arxiv.org/abs/2604.15415", "published": "2026-04-16", "updated": "2026-04-16", "authors": [ "Yukun Jiang", "Yage Zhang", "Michael Backes", "Xinyue Shen", "Yang Zhang" ], "categories": [ "cs.CR", "cs.AI" ], "topics": [ "agent-evaluation", "agent-safety", "rag", "tool-use" ], "score": 14, "relevance": "high", "matched_queries": [ "agent-safety" ], "arxiv_id": "2604.15415", "source": "arxiv", "source_id": "arxiv:2604.15415", "pdf_url": "https://arxiv.org/pdf/2604.15415", "primary_query": "agent-safety" }, { "id": "2604.14399", "title": "SpaceMind: A Modular and Self-Evolving Embodied Vision-Language Agent Framework for Autonomous On-orbit Servicing", "url": "https://arxiv.org/abs/2604.14399", "published": "2026-04-15", "updated": "2026-04-15", "authors": [ "Aodi Wu", "Haodong Han", "Xubo Luo", "Ruisuo Wang", "Shan He", "Xue Wan" ], "categories": [ "cs.RO", "cs.AI", "eess.SY" ], "topics": [ "embodied-agent", "reasoning", "tool-use", "world-model" ], "score": 14, "relevance": "high", "matched_queries": [ "language-agent" ], "arxiv_id": "2604.14399", "source": "arxiv", "source_id": "arxiv:2604.14399", "pdf_url": "https://arxiv.org/pdf/2604.14399", "primary_query": "language-agent" }, { "id": "2604.08388", "title": "Awakening the Sleeping Agent: Lean-Specific Agentic Data Reactivates General Tool Use in Goedel Prover", "url": "https://arxiv.org/abs/2604.08388", "published": "2026-04-09", "updated": "2026-04-09", "authors": [ "Jui-Hui Chung", "Hongzhou Lin", "Lai Jiang", "Shange Tang", "Chi Jin" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "tool-use" ], "score": 14, "relevance": "high", "matched_queries": [ "function-calling" ], "arxiv_id": "2604.08388", "source": "arxiv", "source_id": "arxiv:2604.08388", "pdf_url": "https://arxiv.org/pdf/2604.08388", "primary_query": "function-calling" }, { "id": "2604.06762", "title": "ARuleCon: Agentic Security Rule Conversion", "url": "https://arxiv.org/abs/2604.06762", "published": "2026-04-08", "updated": "2026-04-08", "authors": [ "Ming Xu", "Hongtai Wang", "Yanpei Guo", "Zhengmin Yu", "Weili Han", "Hoon Wei Lim", "Jin Song Dong", "Jiaheng Zhang" ], "categories": [ "cs.CR" ], "topics": [ "agent-evaluation", "agent-safety", "rag" ], "score": 14, "relevance": "high", "matched_queries": [ "agent-safety" ], "arxiv_id": "2604.06762", "source": "arxiv", "source_id": "arxiv:2604.06762", "pdf_url": "https://arxiv.org/pdf/2604.06762", "primary_query": "agent-safety" }, { "id": "2604.02155", "title": "Brief Is Better: Non-Monotonic Chain-of-Thought Budget Effects in Function-Calling Language Agents", "url": "https://arxiv.org/abs/2604.02155", "published": "2026-04-02", "updated": "2026-04-02", "authors": [ "Xuan Qi" ], "categories": [ "cs.CL" ], "topics": [ "agent-evaluation", "rag", "reasoning", "tool-use" ], "score": 14, "relevance": "high", "matched_queries": [ "function-calling", "language-agent" ], "arxiv_id": "2604.02155", "source": "arxiv", "source_id": "arxiv:2604.02155", "pdf_url": "https://arxiv.org/pdf/2604.02155", "primary_query": "function-calling" }, { "id": "2603.27148", "title": "SafetyDrift: Predicting When AI Agents Cross the Line Before They Actually Do", "url": "https://arxiv.org/abs/2603.27148", "published": "2026-03-28", "updated": "2026-03-28", "authors": [ "Aditya Dhodapkar", "Farhaan Pishori" ], "categories": [ "cs.CR", "cs.AI" ], "topics": [ "agent-safety", "tool-use" ], "score": 14, "relevance": "high", "matched_queries": [ "agent-safety" ], "arxiv_id": "2603.27148", "source": "arxiv", "source_id": "arxiv:2603.27148", "pdf_url": "https://arxiv.org/pdf/2603.27148", "primary_query": "agent-safety" }, { "id": "2603.19469", "title": "A Framework for Formalizing LLM Agent Security", "url": "https://arxiv.org/abs/2603.19469", "published": "2026-03-19", "updated": "2026-03-19", "authors": [ "Vincent Siu", "Jingxuan He", "Kyle Montgomery", "Zhun Wang", "Neil Gong", "Chenguang Wang", "Dawn Song" ], "categories": [ "cs.CR", "cs.AI" ], "topics": [ "agent-safety", "memory", "tool-use" ], "score": 14, "relevance": "high", "matched_queries": [ "agent-safety" ], "arxiv_id": "2603.19469", "source": "arxiv", "source_id": "arxiv:2603.19469", "pdf_url": "https://arxiv.org/pdf/2603.19469", "primary_query": "agent-safety" }, { "id": "2603.11088", "title": "The Attack and Defense Landscape of Agentic AI: A Comprehensive Survey", "url": "https://arxiv.org/abs/2603.11088", "published": "2026-03-11", "updated": "2026-03-11", "authors": [ "Juhee Kim", "Xiaoyuan Liu", "Zhun Wang", "Shi Qiu", "Bo Li", "Wenbo Guo", "Dawn Song" ], "categories": [ "cs.CR", "cs.AI" ], "topics": [ "agent-safety", "tool-use", "workflow-agent" ], "score": 14, "relevance": "high", "matched_queries": [ "agent-safety" ], "arxiv_id": "2603.11088", "source": "arxiv", "source_id": "arxiv:2603.11088", "pdf_url": "https://arxiv.org/pdf/2603.11088", "primary_query": "agent-safety" }, { "id": "2603.01438", "title": "Enhancing Persona Following at Decoding Time via Dynamic Importance Estimation for Role-Playing Agents", "url": "https://arxiv.org/abs/2603.01438", "published": "2026-03-02", "updated": "2026-03-02", "authors": [ "Yuxin Liu", "Mingye Zhu", "Siyuan Liu", "Bo Hu", "Lei Zhang" ], "categories": [ "cs.CL", "cs.AI" ], "topics": [ "agent-safety", "coding-agent", "computer-use", "planning", "rag", "world-model" ], "score": 14, "relevance": "high", "matched_queries": [ "language-agent" ], "arxiv_id": "2603.01438", "source": "arxiv", "source_id": "arxiv:2603.01438", "pdf_url": "https://arxiv.org/pdf/2603.01438", "primary_query": "language-agent" }, { "id": "2602.23320", "title": "ParamMem: Augmenting Language Agents with Parametric Reflective Memory", "url": "https://arxiv.org/abs/2602.23320", "published": "2026-02-26", "updated": "2026-02-27", "authors": [ "Tianjun Yao", "Yongqiang Chen", "Yujia Zheng", "Pan Li", "Zhiqiang Shen", "Kun Zhang" ], "categories": [ "cs.LG", "cs.MA" ], "topics": [ "agent-evaluation", "memory", "reasoning" ], "score": 14, "relevance": "high", "matched_queries": [ "language-agent" ], "arxiv_id": "2602.23320", "source": "arxiv", "source_id": "arxiv:2602.23320", "pdf_url": "https://arxiv.org/pdf/2602.23320", "primary_query": "language-agent" }, { "id": "2602.16931", "title": "Narrow Fine-Tuning Erodes Safety Alignment in Vision-Language Agents", "url": "https://arxiv.org/abs/2602.16931", "published": "2026-02-18", "updated": "2026-03-15", "authors": [ "Idhant Gulati", "Shivam Raval" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "agent-safety" ], "score": 14, "relevance": "high", "matched_queries": [ "language-agent" ], "arxiv_id": "2602.16931", "source": "arxiv", "source_id": "arxiv:2602.16931", "pdf_url": "https://arxiv.org/pdf/2602.16931", "primary_query": "language-agent" }, { "id": "2602.07391", "title": "NAAMSE: Framework for Evolutionary Security Evaluation of Agents", "url": "https://arxiv.org/abs/2602.07391", "published": "2026-02-07", "updated": "2026-03-08", "authors": [ "Kunal Pai", "Parth Shah", "Harshil Patel" ], "categories": [ "cs.AI", "cs.MA" ], "topics": [ "agent-evaluation", "agent-safety" ], "score": 14, "relevance": "high", "matched_queries": [ "agent-safety" ], "arxiv_id": "2602.07391", "source": "arxiv", "source_id": "arxiv:2602.07391", "pdf_url": "https://arxiv.org/pdf/2602.07391", "primary_query": "agent-safety" }, { "id": "2601.05467", "title": "STELP: Secure Transpilation and Execution of LLM-Generated Programs", "url": "https://arxiv.org/abs/2601.05467", "published": "2026-01-09", "updated": "2026-01-15", "authors": [ "Swapnil Shinde", "Sahil Wadhwa", "Andy Luo", "Akshay Gupta", "Mohammad Shahed Sorower" ], "categories": [ "cs.SE", "cs.AI" ], "topics": [ "agent-evaluation", "agent-safety", "multi-agent", "planning", "reasoning", "tool-use" ], "score": 14, "relevance": "high", "matched_queries": [ "function-calling" ], "arxiv_id": "2601.05467", "source": "arxiv", "source_id": "arxiv:2601.05467", "pdf_url": "https://arxiv.org/pdf/2601.05467", "primary_query": "function-calling" }, { "id": "2510.26167", "title": "ToolRM: Towards Agentic Tool-Use Reward Modeling", "url": "https://arxiv.org/abs/2510.26167", "published": "2025-10-30", "updated": "2026-01-13", "authors": [ "Renhao Li", "Jianhong Tu", "Yang Su", "Yantao Liu", "Fei Huang", "Hamid Alinejad-Rokny", "Derek F. Wong", "Junyang Lin", "Min Yang" ], "categories": [ "cs.AI", "cs.CL" ], "topics": [ "agent-evaluation", "tool-use" ], "score": 14, "relevance": "high", "matched_queries": [ "function-calling" ], "arxiv_id": "2510.26167", "source": "arxiv", "source_id": "arxiv:2510.26167", "pdf_url": "https://arxiv.org/pdf/2510.26167", "primary_query": "function-calling" }, { "id": "2510.22768", "title": "Seeing is Believing? Evaluating Vision-Language Model Susceptibility in Agent-to-Agent Multimodal Persuasion", "url": "https://arxiv.org/abs/2510.22768", "published": "2025-10-26", "updated": "2026-06-02", "authors": [ "Haoyi Qiu", "Yilun Zhou", "Pranav Narayanan Venkit", "Kung-Hsiang Huang", "Jiaxin Zhang", "Nanyun Peng", "Chien-Sheng Wu" ], "categories": [ "cs.CL" ], "topics": [ "agent-evaluation", "agent-safety", "multi-agent", "rag", "tool-use" ], "score": 14, "relevance": "high", "matched_queries": [ "function-calling" ], "arxiv_id": "2510.22768", "source": "arxiv", "source_id": "arxiv:2510.22768", "pdf_url": "https://arxiv.org/pdf/2510.22768", "primary_query": "function-calling" }, { "id": "2607.05794", "title": "From Passive Retrieval to Active Memory Navigation: Learning to Use Memory as a Structured Action Space", "url": "https://arxiv.org/abs/2607.05794", "published": "2026-07-07", "updated": "2026-07-07", "authors": [ "Yue Xu", "Yutao Sun", "Yihao Liu", "Mengyu Zhou", "Jiayi Qiao", "Lu Ma", "Kai Tang", "Wenjie Wang", "Xiaoxi Jiang", "Guanjun Jiang" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "embodied-agent", "memory", "rag", "reasoning", "tool-use" ], "score": 13, "relevance": "high", "matched_queries": [ "tool-use" ], "arxiv_id": "2607.05794", "source": "arxiv", "source_id": "arxiv:2607.05794", "pdf_url": "https://arxiv.org/pdf/2607.05794", "primary_query": "tool-use" }, { "id": "2607.06341", "title": "Harnessing Code Agents for Automatic Software Verification", "url": "https://arxiv.org/abs/2607.06341", "published": "2026-07-07", "updated": "2026-07-07", "authors": [ "Shuangxiang Kan", "Shuanglong Kan", "Sebastian Ertel" ], "categories": [ "cs.FL", "cs.AI", "cs.SE" ], "topics": [ "agent-evaluation", "coding-agent", "memory", "rag", "tool-use" ], "score": 13, "relevance": "high", "matched_queries": [ "coding-agent" ], "arxiv_id": "2607.06341", "source": "arxiv", "source_id": "arxiv:2607.06341", "pdf_url": "https://arxiv.org/pdf/2607.06341", "primary_query": "coding-agent" }, { "id": "2607.05001", "title": "TACTIC-KG: Toward Small Agent Teams for Cyber Threat Intelligence Knowledge Graph Construction", "url": "https://arxiv.org/abs/2607.05001", "published": "2026-07-06", "updated": "2026-07-07", "authors": [ "Mouhamed Amine Bouchiha", "Gregory Blanc" ], "categories": [ "cs.CR", "cs.AI", "cs.LG", "cs.MA" ], "topics": [ "agent-evaluation", "agent-safety", "reasoning", "tool-use" ], "score": 13, "relevance": "high", "matched_queries": [ "llm-agent" ], "arxiv_id": "2607.05001", "source": "arxiv", "source_id": "arxiv:2607.05001", "pdf_url": "https://arxiv.org/pdf/2607.05001", "primary_query": "llm-agent" }, { "id": "2607.05666", "title": "What Do AI Agents Actually Change? An Empirical Taxonomy of Mutation Patterns in Performance-Improving Pull Requests", "url": "https://arxiv.org/abs/2607.05666", "published": "2026-07-06", "updated": "2026-07-06", "authors": [ "Illia Dovhoshliubnyi", "Nima Soroush", "Ashkan Sami", "Alexander Brownlee" ], "categories": [ "cs.SE", "cs.AI" ], "topics": [ "coding-agent" ], "score": 13, "relevance": "high", "matched_queries": [ "ai-agent", "coding-agent" ], "arxiv_id": "2607.05666", "source": "arxiv", "source_id": "arxiv:2607.05666", "pdf_url": "https://arxiv.org/pdf/2607.05666", "primary_query": "ai-agent" }, { "id": "2607.05518", "title": "aiAuthZ: Off-Host, Identity-Bound Authorization for AI Agents", "url": "https://arxiv.org/abs/2607.05518", "published": "2026-07-06", "updated": "2026-07-06", "authors": [ "Sai Varun Kodathala" ], "categories": [ "cs.CR", "cs.AI" ], "topics": [ "agent-evaluation", "agent-safety", "tool-use" ], "score": 13, "relevance": "high", "matched_queries": [ "ai-agent" ], "arxiv_id": "2607.05518", "source": "arxiv", "source_id": "arxiv:2607.05518", "pdf_url": "https://arxiv.org/pdf/2607.05518", "primary_query": "ai-agent" }, { "id": "2607.04697", "title": "AI Agent Pull Requests on GitHub: Frequency, Structure, and Merge Conflict Rates", "url": "https://arxiv.org/abs/2607.04697", "published": "2026-07-06", "updated": "2026-07-07", "authors": [ "George Xu", "Arjun Subramanian", "Nithilan Karthik" ], "categories": [ "cs.SE" ], "topics": [ "agent-evaluation", "coding-agent", "multi-agent" ], "score": 13, "relevance": "high", "matched_queries": [ "ai-agent", "coding-agent" ], "arxiv_id": "2607.04697", "source": "arxiv", "source_id": "arxiv:2607.04697", "pdf_url": "https://arxiv.org/pdf/2607.04697", "primary_query": "ai-agent" }, { "id": "2607.05391", "title": "LLM-as-a-Verifier: A General-Purpose Verification Framework", "url": "https://arxiv.org/abs/2607.05391", "published": "2026-07-06", "updated": "2026-07-07", "authors": [ "Jacky Kwok", "Shulu Li", "Pranav Atreya", "Yuejiang Liu", "Yixing Jiang", "Chelsea Finn", "Marco Pavone", "Ion Stoica", "Azalia Mirhoseini" ], "categories": [ "cs.AI", "cs.CL", "cs.LG", "cs.MA", "cs.RO" ], "topics": [ "agent-evaluation", "coding-agent", "embodied-agent", "reasoning" ], "score": 13, "relevance": "high", "matched_queries": [ "coding-agent" ], "arxiv_id": "2607.05391", "source": "arxiv", "source_id": "arxiv:2607.05391", "pdf_url": "https://arxiv.org/pdf/2607.05391", "primary_query": "coding-agent" }, { "id": "2607.04334", "title": "Do GUI Agents Believe Their Eyes? Diagnosing State-Belief Reliance on Pixels versus Structure", "url": "https://arxiv.org/abs/2607.04334", "published": "2026-07-05", "updated": "2026-07-05", "authors": [ "Guijia Zhang", "Harry Yang" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "computer-use", "rag", "tool-use" ], "score": 13, "relevance": "high", "matched_queries": [ "web-gui-agent" ], "arxiv_id": "2607.04334", "source": "arxiv", "source_id": "arxiv:2607.04334", "pdf_url": "https://arxiv.org/pdf/2607.04334", "primary_query": "web-gui-agent" }, { "id": "2607.04394", "title": "MechMath Agent Team: LLM Driven Agents for Mathematical Research", "url": "https://arxiv.org/abs/2607.04394", "published": "2026-07-05", "updated": "2026-07-05", "authors": [ "Yichuan Cao", "Ruichen Qiu", "Junqi Liu", "Jiaqi Wang", "Dakai Guo", "Ruyong Feng", "Lihong Zhi", "Xiao-Shan Gao" ], "categories": [ "cs.AI", "cs.SC" ], "topics": [ "agent-evaluation", "multi-agent", "planning", "rag", "reasoning" ], "score": 13, "relevance": "high", "matched_queries": [ "multi-agent-llm" ], "arxiv_id": "2607.04394", "source": "arxiv", "source_id": "arxiv:2607.04394", "pdf_url": "https://arxiv.org/pdf/2607.04394", "primary_query": "multi-agent-llm" }, { "id": "2607.02942", "title": "A Workflow-Aware Serving Layer for Agentic Applications", "url": "https://arxiv.org/abs/2607.02942", "published": "2026-07-03", "updated": "2026-07-03", "authors": [ "Jiayi Qian", "Zishen Wan", "Hanchen Yang", "Chun Tao", "Souvik Kundu", "Tushar Krishna" ], "categories": [ "cs.DC", "cs.MA" ], "topics": [ "reasoning", "tool-use", "workflow-agent" ], "score": 13, "relevance": "high", "matched_queries": [ "agentic-ai" ], "arxiv_id": "2607.02942", "source": "arxiv", "source_id": "arxiv:2607.02942", "pdf_url": "https://arxiv.org/pdf/2607.02942", "primary_query": "agentic-ai" }, { "id": "2607.03105", "title": "ORBIT-Q: Dual-axis benchmarking of autonomous agents in scientific quantum programming", "url": "https://arxiv.org/abs/2607.03105", "published": "2026-07-03", "updated": "2026-07-03", "authors": [ "Shi-Xin Zhang", "Yu-Qin Chen" ], "categories": [ "quant-ph" ], "topics": [ "agent-evaluation", "coding-agent", "workflow-agent" ], "score": 13, "relevance": "high", "matched_queries": [ "coding-agent" ], "arxiv_id": "2607.03105", "source": "arxiv", "source_id": "arxiv:2607.03105", "pdf_url": "https://arxiv.org/pdf/2607.03105", "primary_query": "coding-agent" }, { "id": "2607.03316", "title": "Is Agentic Code Review Helpful? Mining Developers' Feedback to CodeRabbit Reviews in the Wild", "url": "https://arxiv.org/abs/2607.03316", "published": "2026-07-03", "updated": "2026-07-03", "authors": [ "Hong Yi Lin", "Mingzhao Liang", "Kla Tantithamthavorn", "Patanamon Thongtanunam" ], "categories": [ "cs.SE", "cs.AI" ], "topics": [ "agent-safety", "coding-agent", "workflow-agent" ], "score": 13, "relevance": "high", "matched_queries": [ "autonomous-agent-llm" ], "arxiv_id": "2607.03316", "source": "arxiv", "source_id": "arxiv:2607.03316", "pdf_url": "https://arxiv.org/pdf/2607.03316", "primary_query": "autonomous-agent-llm" }, { "id": "2607.03220", "title": "CONTRA: Red-Teaming Configurations of Personalizable Agents", "url": "https://arxiv.org/abs/2607.03220", "published": "2026-07-03", "updated": "2026-07-03", "authors": [ "Jonathan Nöther", "Adish Singla", "Goran Radanovic" ], "categories": [ "cs.CR", "cs.AI" ], "topics": [ "agent-evaluation", "agent-safety", "coding-agent", "reasoning", "tool-use", "workflow-agent" ], "score": 13, "relevance": "high", "matched_queries": [ "autonomous-agent-llm" ], "arxiv_id": "2607.03220", "source": "arxiv", "source_id": "arxiv:2607.03220", "pdf_url": "https://arxiv.org/pdf/2607.03220", "primary_query": "autonomous-agent-llm" }, { "id": "2607.01812", "title": "TO-Master: an LLM-agent framework for automated topology optimization", "url": "https://arxiv.org/abs/2607.01812", "published": "2026-07-02", "updated": "2026-07-02", "authors": [ "Haoju Lin", "Wenchang Zhang", "Weipeng Xu", "Xiang Li", "Tian Xu", "Tianju Xue" ], "categories": [ "cs.CE" ], "topics": [ "agent-evaluation", "computer-use", "reasoning", "tool-use", "workflow-agent" ], "score": 13, "relevance": "high", "matched_queries": [ "llm-agent" ], "arxiv_id": "2607.01812", "source": "arxiv", "source_id": "arxiv:2607.01812", "pdf_url": "https://arxiv.org/pdf/2607.01812", "primary_query": "llm-agent" }, { "id": "2607.02453", "title": "Adoption and Ecosystem Health: A Longitudinal Analysis of Open-Source Multi-Agent Frameworks", "url": "https://arxiv.org/abs/2607.02453", "published": "2026-07-02", "updated": "2026-07-02", "authors": [ "Xi Zhang", "Papi Menon", "Vivian Chu", "Koray Cosguner" ], "categories": [ "cs.MA" ], "topics": [ "agent-evaluation", "multi-agent" ], "score": 13, "relevance": "high", "matched_queries": [ "ai-agent" ], "arxiv_id": "2607.02453", "source": "arxiv", "source_id": "arxiv:2607.02453", "pdf_url": "https://arxiv.org/pdf/2607.02453", "primary_query": "ai-agent" }, { "id": "2607.02245", "title": "Copewell: A Multi-Agent Swarm Architecture for Equitable Mental Wellness Support", "url": "https://arxiv.org/abs/2607.02245", "published": "2026-07-02", "updated": "2026-07-02", "authors": [ "Seren Yenikent", "Jack Vinijtrongjit", "Katherine Ng" ], "categories": [ "cs.AI", "cs.CY", "cs.HC" ], "topics": [ "agent-evaluation", "agent-safety", "multi-agent" ], "score": 13, "relevance": "high", "matched_queries": [ "ai-agent" ], "arxiv_id": "2607.02245", "source": "arxiv", "source_id": "arxiv:2607.02245", "pdf_url": "https://arxiv.org/pdf/2607.02245", "primary_query": "ai-agent" }, { "id": "2607.02381", "title": "HULAT2 at MER-TRANS 2026: Governed Multi-Agent Simplification for Spanish Easy-to-Read Generation", "url": "https://arxiv.org/abs/2607.02381", "published": "2026-07-02", "updated": "2026-07-02", "authors": [ "Lourdes Moreno", "Paloma Martínez", "Marco Antonio Sanchez-Escudero", "Miguel Domínguez-Gómez" ], "categories": [ "cs.CL" ], "topics": [ "agent-evaluation", "computer-use", "multi-agent", "tool-use", "workflow-agent" ], "score": 13, "relevance": "high", "matched_queries": [ "agentic-ai" ], "arxiv_id": "2607.02381", "source": "arxiv", "source_id": "arxiv:2607.02381", "pdf_url": "https://arxiv.org/pdf/2607.02381", "primary_query": "agentic-ai" }, { "id": "2607.01531", "title": "OPINE-World: Programmatic World Modeling with Ontology-error-Prioritized Interactive Exploration", "url": "https://arxiv.org/abs/2607.01531", "published": "2026-07-01", "updated": "2026-07-01", "authors": [ "David Courtis", "Wenhao Li", "Scott Sanner" ], "categories": [ "cs.AI", "cs.LG" ], "topics": [ "agent-evaluation", "computer-use", "planning", "tool-use", "world-model" ], "score": 13, "relevance": "high", "matched_queries": [ "llm-agent", "planning-agent" ], "arxiv_id": "2607.01531", "source": "arxiv", "source_id": "arxiv:2607.01531", "pdf_url": "https://arxiv.org/pdf/2607.01531", "primary_query": "llm-agent" }, { "id": "2607.01047", "title": "Conversable Complexity: Agentic LLM Collectives as Interpretable Substrates", "url": "https://arxiv.org/abs/2607.01047", "published": "2026-07-01", "updated": "2026-07-01", "authors": [ "Elias Najarro", "Ane Espeseth", "Eleni Nisioti", "Sebastian Risi", "Stefano Nichele" ], "categories": [ "cs.CL" ], "topics": [ "memory", "tool-use" ], "score": 13, "relevance": "high", "matched_queries": [ "llm-agent" ], "arxiv_id": "2607.01047", "source": "arxiv", "source_id": "arxiv:2607.01047", "pdf_url": "https://arxiv.org/pdf/2607.01047", "primary_query": "llm-agent" }, { "id": "2607.01510", "title": "Janus: a Playground for User-Involved Agentic Permission Management", "url": "https://arxiv.org/abs/2607.01510", "published": "2026-07-01", "updated": "2026-07-01", "authors": [ "Natalie Grace Brigham", "Eugene Bagdasarian", "Tadayoshi Kohno", "Franziska Roesner" ], "categories": [ "cs.AI", "cs.CR" ], "topics": [ "agent-evaluation", "agent-safety", "tool-use" ], "score": 13, "relevance": "high", "matched_queries": [ "ai-agent" ], "arxiv_id": "2607.01510", "source": "arxiv", "source_id": "arxiv:2607.01510", "pdf_url": "https://arxiv.org/pdf/2607.01510", "primary_query": "ai-agent" }, { "id": "2607.00407", "title": "Personalization as Inverse Planning: Learning Latent Design Intents for Agentic Slide Generation via Structural Denoising", "url": "https://arxiv.org/abs/2607.00407", "published": "2026-07-01", "updated": "2026-07-01", "authors": [ "Tianci Liu", "Zihan Dong", "Linjun Zhang", "Haoyu Wang", "jing Gao", "Emre Kiciman", "Ranveer Chandra", "Wei-Ting Chen" ], "categories": [ "cs.AI" ], "topics": [ "multi-agent", "planning", "tool-use" ], "score": 13, "relevance": "high", "matched_queries": [ "ai-agent" ], "arxiv_id": "2607.00407", "source": "arxiv", "source_id": "arxiv:2607.00407", "pdf_url": "https://arxiv.org/pdf/2607.00407", "primary_query": "ai-agent" }, { "id": "2607.01366", "title": "Auto-FL-Research: Agentic Search for Federated Learning Algorithms", "url": "https://arxiv.org/abs/2607.01366", "published": "2026-07-01", "updated": "2026-07-01", "authors": [ "Holger R. Roth", "Ziyue Xu", "Chester Chen", "Daguang Xu", "Peter Cnudde", "Andrew Feng" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "coding-agent", "workflow-agent" ], "score": 13, "relevance": "high", "matched_queries": [ "agentic-ai", "coding-agent" ], "arxiv_id": "2607.01366", "source": "arxiv", "source_id": "arxiv:2607.01366", "pdf_url": "https://arxiv.org/pdf/2607.01366", "primary_query": "agentic-ai" }, { "id": "2607.00990", "title": "SWE-Doctor: Guiding Software Engineering Agents with Runtime Diagnosis from Multi-Faceted Bug Reproduction Tests", "url": "https://arxiv.org/abs/2607.00990", "published": "2026-07-01", "updated": "2026-07-01", "authors": [ "Yaoqi Guo", "Yang Liu", "Jie M. Zhang", "Yun Ma", "Yiling Lou", "Zhenpeng Chen" ], "categories": [ "cs.SE", "cs.AI" ], "topics": [ "agent-evaluation", "coding-agent", "computer-use", "rag" ], "score": 13, "relevance": "high", "matched_queries": [ "coding-agent" ], "arxiv_id": "2607.00990", "source": "arxiv", "source_id": "arxiv:2607.00990", "pdf_url": "https://arxiv.org/pdf/2607.00990", "primary_query": "coding-agent" }, { "id": "2606.31471", "title": "Think While You Map: Asynchronous Vision-Language Agents for Incremental 3D Scene Graphs", "url": "https://arxiv.org/abs/2606.31471", "published": "2026-06-30", "updated": "2026-06-30", "authors": [ "Deniz Bickici", "Michael Pabst", "Shohei Mori", "Dieter Schmalstieg" ], "categories": [ "cs.CV" ], "topics": [ "agent-evaluation", "rag" ], "score": 13, "relevance": "high", "matched_queries": [ "language-agent" ], "arxiv_id": "2606.31471", "source": "arxiv", "source_id": "arxiv:2606.31471", "pdf_url": "https://arxiv.org/pdf/2606.31471", "primary_query": "language-agent" }, { "id": "2606.31831", "title": "An Agentic AI Framework to Accelerate Scientific Discovery in Plant Phenotyping", "url": "https://arxiv.org/abs/2606.31831", "published": "2026-06-30", "updated": "2026-06-30", "authors": [ "Renan Souza", "Daniel Rosendo", "Kelsey Carter", "John Lagergren", "Frédéric Suter", "Shelaine L. Curd", "Gerald A. Tuskan", "Rafael Ferreira da Silva", "David Weston" ], "categories": [ "cs.AI" ], "topics": [ "agent-safety", "planning", "tool-use" ], "score": 13, "relevance": "high", "matched_queries": [ "agentic-ai", "ai-agent" ], "arxiv_id": "2606.31831", "source": "arxiv", "source_id": "arxiv:2606.31831", "pdf_url": "https://arxiv.org/pdf/2606.31831", "primary_query": "agentic-ai" }, { "id": "2606.31916", "title": "Theory of Mind and Persuasion Beyond Conversation: Assessing the Capacity of LLMs to Induce Belief States via Planning and Action", "url": "https://arxiv.org/abs/2606.31916", "published": "2026-06-30", "updated": "2026-06-30", "authors": [ "Ben Slater", "Matteo G. Mecattaf", "Lucy G. Cheke", "John Burden", "Winnie Street" ], "categories": [ "cs.CL" ], "topics": [ "agent-evaluation", "agent-safety", "planning", "reasoning", "tool-use" ], "score": 13, "relevance": "high", "matched_queries": [ "agent-evaluation" ], "arxiv_id": "2606.31916", "source": "arxiv", "source_id": "arxiv:2606.31916", "pdf_url": "https://arxiv.org/pdf/2606.31916", "primary_query": "agent-evaluation" }, { "id": "2606.31767", "title": "JETO-Bench: A Reproducible Benchmark for Execution Time Improvement Patches in Java", "url": "https://arxiv.org/abs/2606.31767", "published": "2026-06-30", "updated": "2026-06-30", "authors": [ "Khashayar Etemadi", "Zhendong Su" ], "categories": [ "cs.SE" ], "topics": [ "agent-evaluation", "coding-agent", "tool-use" ], "score": 13, "relevance": "high", "matched_queries": [ "coding-agent" ], "arxiv_id": "2606.31767", "source": "arxiv", "source_id": "arxiv:2606.31767", "pdf_url": "https://arxiv.org/pdf/2606.31767", "primary_query": "coding-agent" }, { "id": "2606.31665", "title": "ForecastAgentSearch: Towards a Multi-Expert Agent Search System for Geopolitical Event Forecasting", "url": "https://arxiv.org/abs/2606.31665", "published": "2026-06-30", "updated": "2026-06-30", "authors": [ "Miaomiao Cai", "He Chang", "Yunshan Ma", "See-kiong Ng" ], "categories": [ "cs.MA" ], "topics": [ "agent-evaluation", "multi-agent", "planning", "rag" ], "score": 13, "relevance": "high", "matched_queries": [ "multi-agent-llm" ], "arxiv_id": "2606.31665", "source": "arxiv", "source_id": "arxiv:2606.31665", "pdf_url": "https://arxiv.org/pdf/2606.31665", "primary_query": "multi-agent-llm" }, { "id": "2606.29719", "title": "A Diagnostic Framework and Multi-Evaluator Audit of Evaluator-Driven Preference Dynamics in Self-Adapting LLM Agents", "url": "https://arxiv.org/abs/2606.29719", "published": "2026-06-29", "updated": "2026-06-29", "authors": [ "Liu Zewen" ], "categories": [ "cs.LG", "cs.CL" ], "topics": [ "agent-evaluation", "rag" ], "score": 13, "relevance": "high", "matched_queries": [ "llm-agent" ], "arxiv_id": "2606.29719", "source": "arxiv", "source_id": "arxiv:2606.29719", "pdf_url": "https://arxiv.org/pdf/2606.29719", "primary_query": "llm-agent" }, { "id": "2606.29745", "title": "ECHO: Learning Epistemically Adaptive Language Agents with Turn-Level Credit", "url": "https://arxiv.org/abs/2606.29745", "published": "2026-06-29", "updated": "2026-06-29", "authors": [ "Abhijnan Nath", "Nikhil Krishnaswamy" ], "categories": [ "cs.MA" ], "topics": [ "agent-evaluation", "rag", "reasoning", "tool-use" ], "score": 13, "relevance": "high", "matched_queries": [ "language-agent" ], "arxiv_id": "2606.29745", "source": "arxiv", "source_id": "arxiv:2606.29745", "pdf_url": "https://arxiv.org/pdf/2606.29745", "primary_query": "language-agent" }, { "id": "2606.30970", "title": "Behavioral Governance for Autonomous AI Agents: The AgentBound Framework", "url": "https://arxiv.org/abs/2606.30970", "published": "2026-06-29", "updated": "2026-07-01", "authors": [ "Anuj Kaul", "Qianlong Lan", "Pranay Gupta" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "agent-safety", "tool-use", "workflow-agent" ], "score": 13, "relevance": "high", "matched_queries": [ "ai-agent" ], "arxiv_id": "2606.30970", "source": "arxiv", "source_id": "arxiv:2606.30970", "pdf_url": "https://arxiv.org/pdf/2606.30970", "primary_query": "ai-agent" }, { "id": "2606.29788", "title": "MemLeak: Diagnosing Information Leaks in Multimodal Agent Memory", "url": "https://arxiv.org/abs/2606.29788", "published": "2026-06-29", "updated": "2026-06-29", "authors": [ "Kuan Wang", "Chao Zhang" ], "categories": [ "cs.LG" ], "topics": [ "agent-evaluation", "memory" ], "score": 13, "relevance": "high", "matched_queries": [ "agent-memory", "ai-agent" ], "arxiv_id": "2606.29788", "source": "arxiv", "source_id": "arxiv:2606.29788", "pdf_url": "https://arxiv.org/pdf/2606.29788", "primary_query": "agent-memory" }, { "id": "2606.30616", "title": "Scaling the Horizon, Not the Parameters: Reaching Trillion-Parameter Performance with a 35B Agent", "url": "https://arxiv.org/abs/2606.30616", "published": "2026-06-29", "updated": "2026-06-29", "authors": [ "Lei Bai", "Zongsheng Cao", "Yang Chen", "Zhiyao Cui", "Shangheng Du", "Yue Fan", "Shiyang Feng", "Zijie Guo", "Haonan He", "Liang He", "Xiaohan He", "Shuyue Hu", "Yusong Hu", "Songtao Huang", "Yichen Jiang", "Hao Li", "Xin Li", "Dahua Lin", "Weihao Lin", "Fenghua Ling", "Dongrui Liu", "Zhuo Liu", "Runmin Ma", "Chunjiang Mu", "Haoyang Peng", "Tianshuo Peng", "Jinxin Shi", "Luohe Shi", "Boyuan Sun", "Zelin Tan", "Shengji Tang", "Qianyi Wang", "Yiming Wu", "Yi Xie", "Xiangchao Yan", "Jingqi Ye", "Peng Ye", "Fangchen Yu", "Jiakang Yuan", "Bihao Zhan", "Bo Zhang", "Chen Zhang", "Shufei Zhang", "Shuaiyu Zhang", "Wenlong Zhang", "Yiqun Zhang", "Junpeng Zhao", "Zhijie Zhong", "Bowen Zhou", "Yuhao Zhou" ], "categories": [ "cs.CL" ], "topics": [ "agent-evaluation", "agent-safety", "planning", "rag", "reasoning", "tool-use" ], "score": 13, "relevance": "high", "matched_queries": [ "agent-evaluation" ], "arxiv_id": "2606.30616", "source": "arxiv", "source_id": "arxiv:2606.30616", "pdf_url": "https://arxiv.org/pdf/2606.30616", "primary_query": "agent-evaluation" }, { "id": "2606.30111", "title": "Automating the Design of Embodied Agent Architectures", "url": "https://arxiv.org/abs/2606.30111", "published": "2026-06-29", "updated": "2026-07-03", "authors": [ "Jian Zhou", "Sihao Lin", "Jin Li", "Shuai Fu", "Gengze Zhou", "Qi Wu" ], "categories": [ "cs.RO", "cs.AI", "cs.LG" ], "topics": [ "agent-evaluation", "coding-agent", "embodied-agent", "memory", "planning", "reasoning", "tool-use" ], "score": 13, "relevance": "high", "matched_queries": [ "coding-agent" ], "arxiv_id": "2606.30111", "source": "arxiv", "source_id": "arxiv:2606.30111", "pdf_url": "https://arxiv.org/pdf/2606.30111", "primary_query": "coding-agent" }, { "id": "2606.29495", "title": "Cognitive World Models for Process-Level Social Influence Evaluation", "url": "https://arxiv.org/abs/2606.29495", "published": "2026-06-28", "updated": "2026-06-28", "authors": [ "Minghui Ma", "Bin Guo", "Han Wang", "Mengqi Chen", "Jingqi Liu", "Yan Liu", "Zhiwen Yu" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "computer-use", "multi-agent", "world-model" ], "score": 13, "relevance": "high", "matched_queries": [ "multi-agent-llm" ], "arxiv_id": "2606.29495", "source": "arxiv", "source_id": "arxiv:2606.29495", "pdf_url": "https://arxiv.org/pdf/2606.29495", "primary_query": "multi-agent-llm" }, { "id": "2606.28896", "title": "A Task-Driven and Quality-Assured Agent Framework for SAR Data Generation", "url": "https://arxiv.org/abs/2606.28896", "published": "2026-06-27", "updated": "2026-06-27", "authors": [ "Xuanting Wu", "Fan Zhanga", "Fei Ma", "Ling Guan", "Guochun Ma", "Yongsheng Zhou" ], "categories": [ "eess.IV", "cs.AI", "cs.LG" ], "topics": [ "agent-evaluation", "planning", "rag", "workflow-agent" ], "score": 13, "relevance": "high", "matched_queries": [ "agent-evaluation" ], "arxiv_id": "2606.28896", "source": "arxiv", "source_id": "arxiv:2606.28896", "pdf_url": "https://arxiv.org/pdf/2606.28896", "primary_query": "agent-evaluation" }, { "id": "2606.28279", "title": "Agentic Hardware Design as Repository-Level Code Evolution", "url": "https://arxiv.org/abs/2606.28279", "published": "2026-06-26", "updated": "2026-06-26", "authors": [ "Cunxi Yu", "Chenhui Deng", "Nathaniel Pinckney", "Brucek Khailany" ], "categories": [ "cs.AR", "cs.AI" ], "topics": [ "agent-evaluation", "coding-agent" ], "score": 13, "relevance": "high", "matched_queries": [ "agentic-ai" ], "arxiv_id": "2606.28279", "source": "arxiv", "source_id": "arxiv:2606.28279", "pdf_url": "https://arxiv.org/pdf/2606.28279", "primary_query": "agentic-ai" }, { "id": "2606.28430", "title": "Building to the Test: Coding Agents Deliver What You Check, Not What You Requested", "url": "https://arxiv.org/abs/2606.28430", "published": "2026-06-26", "updated": "2026-06-26", "authors": [ "Yanuo Ma", "Ben Kereopa-Yorke", "Ben Schultz" ], "categories": [ "cs.SE", "cs.AI" ], "topics": [ "agent-evaluation", "coding-agent" ], "score": 13, "relevance": "high", "matched_queries": [ "coding-agent" ], "arxiv_id": "2606.28430", "source": "arxiv", "source_id": "arxiv:2606.28430", "pdf_url": "https://arxiv.org/pdf/2606.28430", "primary_query": "coding-agent" }, { "id": "2606.28187", "title": "GBC: Gradient-Based Connections for Optimizing Multi-Agent Systems", "url": "https://arxiv.org/abs/2606.28187", "published": "2026-06-26", "updated": "2026-06-26", "authors": [ "Xiaocheng Yang", "Abdulrahman Alrabah", "Dilek Hakkani-Tür", "Gokhan Tur" ], "categories": [ "cs.MA" ], "topics": [ "multi-agent", "rag", "tool-use" ], "score": 13, "relevance": "high", "matched_queries": [ "multi-agent-llm" ], "arxiv_id": "2606.28187", "source": "arxiv", "source_id": "arxiv:2606.28187", "pdf_url": "https://arxiv.org/pdf/2606.28187", "primary_query": "multi-agent-llm" }, { "id": "2606.27416", "title": "Glite ARF: Verifier-Driven Research with Parallel LLM Coding Agents", "url": "https://arxiv.org/abs/2606.27416", "published": "2026-06-25", "updated": "2026-06-25", "authors": [ "Vassili Philippov", "Pavel Katunin", "Dmitry Andreev", "Igor Ostanin", "Anton Nikolaev" ], "categories": [ "cs.MA", "cs.SE" ], "topics": [ "coding-agent", "reasoning", "tool-use" ], "score": 13, "relevance": "high", "matched_queries": [ "coding-agent" ], "arxiv_id": "2606.27416", "source": "arxiv", "source_id": "arxiv:2606.27416", "pdf_url": "https://arxiv.org/pdf/2606.27416", "primary_query": "coding-agent" }, { "id": "2606.26649", "title": "Autoformalization of Agent Instructions into Policy-as-Code", "url": "https://arxiv.org/abs/2606.26649", "published": "2026-06-25", "updated": "2026-06-25", "authors": [ "Adam Mondl", "Matthew Maisel", "John H. Brock" ], "categories": [ "cs.AI", "cs.CR" ], "topics": [ "agent-evaluation", "agent-safety", "coding-agent", "tool-use" ], "score": 13, "relevance": "high", "matched_queries": [ "agent-safety" ], "arxiv_id": "2606.26649", "source": "arxiv", "source_id": "arxiv:2606.26649", "pdf_url": "https://arxiv.org/pdf/2606.26649", "primary_query": "agent-safety" }, { "id": "2606.26057", "title": "The Unfireable Safety Kernel: Execution-Time AI Alignment for AI Agents and Other Escapable AI Systems", "url": "https://arxiv.org/abs/2606.26057", "published": "2026-06-24", "updated": "2026-06-24", "authors": [ "Seth Dobrin", "Łukasz Chmiel" ], "categories": [ "cs.AI", "cs.CR", "cs.LG" ], "topics": [ "agent-evaluation", "agent-safety", "planning", "tool-use", "world-model" ], "score": 13, "relevance": "high", "matched_queries": [ "ai-agent" ], "arxiv_id": "2606.26057", "source": "arxiv", "source_id": "arxiv:2606.26057", "pdf_url": "https://arxiv.org/pdf/2606.26057", "primary_query": "ai-agent" }, { "id": "2606.26356", "title": "Instruction Bleed: Cross-Module Interference in Prompt-Composed Agentic Systems", "url": "https://arxiv.org/abs/2606.26356", "published": "2026-06-24", "updated": "2026-06-24", "authors": [ "Ching-Yu Lin", "Yifan Liu" ], "categories": [ "cs.AI", "cs.IR", "cs.MA" ], "topics": [ "agent-evaluation", "multi-agent" ], "score": 13, "relevance": "high", "matched_queries": [ "agent-evaluation", "agentic-ai" ], "arxiv_id": "2606.26356", "source": "arxiv", "source_id": "arxiv:2606.26356", "pdf_url": "https://arxiv.org/pdf/2606.26356", "primary_query": "agent-evaluation" }, { "id": "2606.25519", "title": "Quantization Inflates Reasoning: Token Inflation as a Hidden Cost of Low-Bit Reasoning Models", "url": "https://arxiv.org/abs/2606.25519", "published": "2026-06-24", "updated": "2026-06-29", "authors": [ "Xinyu Lian", "Walid Krichene", "Beichen Huang", "Masahiro Tanaka", "Olatunji Ruwase", "Li Zhang", "Minjia Zhang" ], "categories": [ "cs.AI", "cs.LG" ], "topics": [ "agent-evaluation", "coding-agent", "rag", "reasoning", "tool-use" ], "score": 13, "relevance": "high", "matched_queries": [ "tool-use" ], "arxiv_id": "2606.25519", "source": "arxiv", "source_id": "arxiv:2606.25519", "pdf_url": "https://arxiv.org/pdf/2606.25519", "primary_query": "tool-use" }, { "id": "2606.25195", "title": "SoK: AI Secure Code Generation: Progress, Pitfalls, and Paths Forward", "url": "https://arxiv.org/abs/2606.25195", "published": "2026-06-23", "updated": "2026-06-23", "authors": [ "Rupam Patir", "Keyan Guo", "Haipeng Cai", "Hongxin Hu" ], "categories": [ "cs.CR", "cs.AI" ], "topics": [ "agent-evaluation", "agent-safety", "coding-agent", "computer-use", "workflow-agent" ], "score": 13, "relevance": "high", "matched_queries": [ "agentic-ai", "coding-agent" ], "arxiv_id": "2606.25195", "source": "arxiv", "source_id": "arxiv:2606.25195", "pdf_url": "https://arxiv.org/pdf/2606.25195", "primary_query": "agentic-ai" }, { "id": "2606.24416", "title": "Agentic AI for Bilevel Long-Term Optimization of Policy-Driven Physical Layer Systems", "url": "https://arxiv.org/abs/2606.24416", "published": "2026-06-23", "updated": "2026-06-23", "authors": [ "Bingnan Xiao", "Chenhao Yang", "Wei Ni", "Xin Wang", "Tony Q. S. Quek" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "memory", "multi-agent", "rag" ], "score": 13, "relevance": "high", "matched_queries": [ "agentic-ai" ], "arxiv_id": "2606.24416", "source": "arxiv", "source_id": "arxiv:2606.24416", "pdf_url": "https://arxiv.org/pdf/2606.24416", "primary_query": "agentic-ai" }, { "id": "2606.24855", "title": "OpenThoughts-Agent: Data Recipes for Agentic Models", "url": "https://arxiv.org/abs/2606.24855", "published": "2026-06-23", "updated": "2026-06-23", "authors": [ "Negin Raoof", "Richard Zhuang", "Marianna Nezhurina", "Etash Guha", "Atula Tejaswi", "Ryan Marten", "Charlie F. Ruan", "Tyler Griggs", "Alexander Glenn Shaw", "Hritik Bansal", "E. Kelly Buchanan", "Artem Gazizov", "Reinhard Heckel", "Chinmay Hegde", "Sankalp Jajee", "Daanish Khazi", "Emmanouil Koukoumidis", "Xiangyi Li", "Hange Liu", "Shlok Natarajan", "Harsh Raj", "Nicholas Roberts", "Ethan Shen", "Nishad Singhi", "Michael Siu", "Ashima Suvarna", "Hanwen Xing", "Patrick Yubeaton", "Robert Zhang", "Leon Liangyu Chen", "Xiaokun Chen", "Steven Dillmann", "Saadia Gabriel", "Xunyi Jiang", "Anurag Kashyap", "Boxuan Li", "Yein Park", "Minh Pham", "Sujay Sanghavi", "Lin Shi", "Ke Sun", "Yixin Wang", "Zhiwei Xu", "Erica Zhang", "Siyan Zhao", "Wanjia Zhao", "Jenia Jitsev", "Alex Dimakis", "Benjamin Feuer", "Ludwig Schmidt" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "rag" ], "score": 13, "relevance": "high", "matched_queries": [ "agent-evaluation" ], "arxiv_id": "2606.24855", "source": "arxiv", "source_id": "arxiv:2606.24855", "pdf_url": "https://arxiv.org/pdf/2606.24855", "primary_query": "agent-evaluation" }, { "id": "2606.25191", "title": "To Isolate or to Score? Model-Adaptive Assessment for Cost-Efficient Multi-Agent RAG", "url": "https://arxiv.org/abs/2606.25191", "published": "2026-06-23", "updated": "2026-06-23", "authors": [ "Jungseob Lee", "Chanjun Park", "Heuiseok Lim" ], "categories": [ "cs.AI", "cs.CL" ], "topics": [ "agent-evaluation", "multi-agent", "rag", "reasoning" ], "score": 13, "relevance": "high", "matched_queries": [ "rag-agent" ], "arxiv_id": "2606.25191", "source": "arxiv", "source_id": "arxiv:2606.25191", "pdf_url": "https://arxiv.org/pdf/2606.25191", "primary_query": "rag-agent" }, { "id": "2606.23130", "title": "Understanding the (In)Security of Vibe-Coded Applications", "url": "https://arxiv.org/abs/2606.23130", "published": "2026-06-22", "updated": "2026-06-23", "authors": [ "Junquan Deng", "Zhiyu Fan", "Ruijie Meng" ], "categories": [ "cs.CR", "cs.SE" ], "topics": [ "agent-evaluation", "agent-safety", "coding-agent", "memory", "tool-use", "workflow-agent" ], "score": 13, "relevance": "high", "matched_queries": [ "ai-agent" ], "arxiv_id": "2606.23130", "source": "arxiv", "source_id": "arxiv:2606.23130", "pdf_url": "https://arxiv.org/pdf/2606.23130", "primary_query": "ai-agent" }, { "id": "2606.22953", "title": "Plans Don't Persist: Why Context Management Is Load Bearing for LLM Agents", "url": "https://arxiv.org/abs/2606.22953", "published": "2026-06-22", "updated": "2026-06-22", "authors": [ "Aman Mehta", "Anupam Datta" ], "categories": [ "cs.AI", "cs.CL" ], "topics": [ "planning", "reasoning", "tool-use" ], "score": 13, "relevance": "high", "matched_queries": [ "planning-agent" ], "arxiv_id": "2606.22953", "source": "arxiv", "source_id": "arxiv:2606.22953", "pdf_url": "https://arxiv.org/pdf/2606.22953", "primary_query": "planning-agent" }, { "id": "2606.21836", "title": "AgentDSE: Reasoning-Augmented Architectural Design Space Exploration", "url": "https://arxiv.org/abs/2606.21836", "published": "2026-06-20", "updated": "2026-06-20", "authors": [ "Chenyu Wang", "Jiahe Caroline Shi", "David Kong", "Duane S. Boning", "Zishen Wan", "Yilun Du", "Vijay Janapa Reddi" ], "categories": [ "cs.AR", "cs.AI" ], "topics": [ "agent-evaluation", "coding-agent", "computer-use", "reasoning" ], "score": 13, "relevance": "high", "matched_queries": [ "coding-agent" ], "arxiv_id": "2606.21836", "source": "arxiv", "source_id": "arxiv:2606.21836", "pdf_url": "https://arxiv.org/pdf/2606.21836", "primary_query": "coding-agent" }, { "id": "2606.21401", "title": "SwarmX: Agentic Scheduling for Low-Latency Agentic Systems", "url": "https://arxiv.org/abs/2606.21401", "published": "2026-06-19", "updated": "2026-06-28", "authors": [ "Yeqi Huang", "Yanwei Ye", "Guomin Chen", "Wenhao Su", "Bin Gong", "Jialian Li", "Zhan Lu", "Yangshen Deng", "Xuan Sun", "Le Xu", "Luo Mai" ], "categories": [ "cs.DC", "cs.AI" ], "topics": [ "agent-evaluation", "multi-agent", "tool-use", "workflow-agent" ], "score": 13, "relevance": "high", "matched_queries": [ "agentic-ai" ], "arxiv_id": "2606.21401", "source": "arxiv", "source_id": "arxiv:2606.21401", "pdf_url": "https://arxiv.org/pdf/2606.21401", "primary_query": "agentic-ai" }, { "id": "2606.21228", "title": "Sakana Fugu Technical Report", "url": "https://arxiv.org/abs/2606.21228", "published": "2026-06-19", "updated": "2026-06-23", "authors": [ "Yujin Tang", "Edoardo Cetin", "Jinglue Xu", "Qi Sun", "Stefan Nielsen", "Vincent Richard", "Haruto Goda", "Iaroslav Tymchenko", "Nhan Nguyen", "Hyunin Lee", "Mari Ashiga", "Shashank Kotyan", "So Kuroki", "Tarin Clanuwat" ], "categories": [ "cs.LG" ], "topics": [ "coding-agent", "multi-agent", "rag", "reasoning" ], "score": 13, "relevance": "high", "matched_queries": [ "coding-agent" ], "arxiv_id": "2606.21228", "source": "arxiv", "source_id": "arxiv:2606.21228", "pdf_url": "https://arxiv.org/pdf/2606.21228", "primary_query": "coding-agent" }, { "id": "2606.20510", "title": "Efficient and Sound Probabilistic Verification for AI Agents", "url": "https://arxiv.org/abs/2606.20510", "published": "2026-06-18", "updated": "2026-06-18", "authors": [ "Alaia Solko-Breslin", "Pramod Kaushik Mudrakarta", "Mihai Christodorescu", "Somesh Jha", "Krishnamurthy Dj Dvijotham" ], "categories": [ "cs.CR", "cs.AI" ], "topics": [ "agent-evaluation", "agent-safety", "computer-use", "tool-use" ], "score": 13, "relevance": "high", "matched_queries": [ "ai-agent" ], "arxiv_id": "2606.20510", "source": "arxiv", "source_id": "arxiv:2606.20510", "pdf_url": "https://arxiv.org/pdf/2606.20510", "primary_query": "ai-agent" }, { "id": "2606.19242", "title": "Runtime Compliance Verification for AI Agents", "url": "https://arxiv.org/abs/2606.19242", "published": "2026-06-17", "updated": "2026-06-17", "authors": [ "Nafiseh Kahani", "Masoud Barati", "Diana Addae" ], "categories": [ "cs.SE" ], "topics": [ "agent-evaluation", "tool-use" ], "score": 13, "relevance": "high", "matched_queries": [ "ai-agent", "function-calling", "tool-use" ], "arxiv_id": "2606.19242", "source": "arxiv", "source_id": "arxiv:2606.19242", "pdf_url": "https://arxiv.org/pdf/2606.19242", "primary_query": "ai-agent" }, { "id": "2606.28374", "title": "Recursive Self-Evolving Agents via Held-Out Selection", "url": "https://arxiv.org/abs/2606.28374", "published": "2026-06-17", "updated": "2026-06-17", "authors": [ "Michael Nguyen", "Quoc Nguyen", "Paul Vuong" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "reasoning", "tool-use", "workflow-agent" ], "score": 13, "relevance": "high", "matched_queries": [ "tool-use" ], "arxiv_id": "2606.28374", "source": "arxiv", "source_id": "arxiv:2606.28374", "pdf_url": "https://arxiv.org/pdf/2606.28374", "primary_query": "tool-use" }, { "id": "2606.18363", "title": "Guava: An Effective and Universal Harness for Embodied Manipulation", "url": "https://arxiv.org/abs/2606.18363", "published": "2026-06-16", "updated": "2026-06-16", "authors": [ "Haowen Liu", "Xirui Li", "Shaoxiong Yao", "Peng Shi", "Tianyi Zhou", "Jia-Bin Huang", "Furong Huang", "Jiayuan Mao" ], "categories": [ "cs.RO", "cs.AI" ], "topics": [ "embodied-agent", "planning", "reasoning", "tool-use", "workflow-agent", "world-model" ], "score": 13, "relevance": "high", "matched_queries": [ "agentic-ai", "tool-use" ], "arxiv_id": "2606.18363", "source": "arxiv", "source_id": "arxiv:2606.18363", "pdf_url": "https://arxiv.org/pdf/2606.18363", "primary_query": "agentic-ai" }, { "id": "2606.17453", "title": "MapSatisfyBench: Benchmarking Satisfaction-Aware Map Agents through Behavior-Grounded Implicit Decision Factors", "url": "https://arxiv.org/abs/2606.17453", "published": "2026-06-16", "updated": "2026-06-17", "authors": [ "Lubin Bai", "Mengyu Cao", "Sixue Wang", "Zhongwei Wan", "Yue Pan", "Jiale Hou", "Xiang Li", "Xiuyuan Zhang" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "tool-use" ], "score": 13, "relevance": "high", "matched_queries": [ "agent-evaluation" ], "arxiv_id": "2606.17453", "source": "arxiv", "source_id": "arxiv:2606.17453", "pdf_url": "https://arxiv.org/pdf/2606.17453", "primary_query": "agent-evaluation" }, { "id": "2606.18023", "title": "LoopCoder-v2: Only Loop Once for Efficient Test-Time Computation Scaling", "url": "https://arxiv.org/abs/2606.18023", "published": "2026-06-16", "updated": "2026-06-16", "authors": [ "Jian Yang", "Shawn Guo", "Wei Zhang", "Tianyu Zheng", "Yaxin Du", "Haau-Sing Li", "Jiajun Wu", "Yue Song", "Yan Xing", "Qingsong Cai", "Zelong Huang", "Chuan Hao", "Ran Tao", "Xianglong Liu", "Wayne Xin Zhao", "Mingjie Tang", "Weifeng Lv", "Ming Zhou", "Bryan Dai" ], "categories": [ "cs.LG", "cs.AI" ], "topics": [ "agent-evaluation", "coding-agent", "memory", "reasoning", "tool-use" ], "score": 13, "relevance": "high", "matched_queries": [ "tool-use" ], "arxiv_id": "2606.18023", "source": "arxiv", "source_id": "arxiv:2606.18023", "pdf_url": "https://arxiv.org/pdf/2606.18023", "primary_query": "tool-use" }, { "id": "2606.17383", "title": "Model Validation of Agentic AI Systems: A POMDP-Based Framework for Belief-State, Forecast, and Policy Validation", "url": "https://arxiv.org/abs/2606.17383", "published": "2026-06-16", "updated": "2026-06-16", "authors": [ "Matthew Francis Dixon" ], "categories": [ "q-fin.RM", "cs.AI", "cs.LG", "stat.ML" ], "topics": [ "agent-safety", "rag", "tool-use" ], "score": 13, "relevance": "high", "matched_queries": [ "autonomous-agent-llm" ], "arxiv_id": "2606.17383", "source": "arxiv", "source_id": "arxiv:2606.17383", "pdf_url": "https://arxiv.org/pdf/2606.17383", "primary_query": "autonomous-agent-llm" }, { "id": "2606.16813", "title": "GIST-CMTF: Goal-State Inference for Causal Minimal Tool Filtering in LLM Agents", "url": "https://arxiv.org/abs/2606.16813", "published": "2026-06-15", "updated": "2026-06-15", "authors": [ "Rahul Suresh Babu", "Rohit Shukla" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "computer-use", "tool-use" ], "score": 13, "relevance": "high", "matched_queries": [ "tool-use" ], "arxiv_id": "2606.16813", "source": "arxiv", "source_id": "arxiv:2606.16813", "pdf_url": "https://arxiv.org/pdf/2606.16813", "primary_query": "tool-use" }, { "id": "2606.16839", "title": "Towards LLM Accelerated Rapid Reviews for Software Tool Discovery -- Case for Log Anomaly Detection", "url": "https://arxiv.org/abs/2606.16839", "published": "2026-06-15", "updated": "2026-06-15", "authors": [ "Jesse Nyyssölä", "Hamza Bin Mazhar", "Alexander Bakhtin", "Matteo Esposito", "Nana Reinikainen", "Yuqing Wang", "Ying Song", "Davide Taibi", "Mika Mäntylä" ], "categories": [ "cs.SE" ], "topics": [ "agent-evaluation", "coding-agent", "planning", "tool-use", "workflow-agent" ], "score": 13, "relevance": "high", "matched_queries": [ "planning-agent" ], "arxiv_id": "2606.16839", "source": "arxiv", "source_id": "arxiv:2606.16839", "pdf_url": "https://arxiv.org/pdf/2606.16839", "primary_query": "planning-agent" }, { "id": "2606.16481", "title": "Steering Emotional Dynamics for Art Therapy: Controllable Narrative Script Generation through Hierarchically Guided LLM Agents", "url": "https://arxiv.org/abs/2606.16481", "published": "2026-06-15", "updated": "2026-06-15", "authors": [ "Suqing Wang", "Qinghai Miao", "Chao Guo", "Yisheng Lv" ], "categories": [ "cs.AI" ], "topics": [ "computer-use", "planning", "tool-use" ], "score": 13, "relevance": "high", "matched_queries": [ "planning-agent" ], "arxiv_id": "2606.16481", "source": "arxiv", "source_id": "arxiv:2606.16481", "pdf_url": "https://arxiv.org/pdf/2606.16481", "primary_query": "planning-agent" }, { "id": "2606.17368", "title": "Distributed General-Purpose Agent Networks: Architecture, Key Mechanisms, and Prototypes", "url": "https://arxiv.org/abs/2606.17368", "published": "2026-06-15", "updated": "2026-06-15", "authors": [ "Shengli Zhang", "Deen Ma", "Zibin Lin", "Taotao Wang" ], "categories": [ "cs.AI", "cs.NI" ], "topics": [ "computer-use", "multi-agent", "planning", "tool-use", "world-model" ], "score": 13, "relevance": "high", "matched_queries": [ "autonomous-agent-llm" ], "arxiv_id": "2606.17368", "source": "arxiv", "source_id": "arxiv:2606.17368", "pdf_url": "https://arxiv.org/pdf/2606.17368", "primary_query": "autonomous-agent-llm" }, { "id": "2606.15709", "title": "AI-Driven Framework for Adaptive Water Network Management with Proof-of-Concept Implementation: Addressing Non-Revenue Water in Jordan", "url": "https://arxiv.org/abs/2606.15709", "published": "2026-06-14", "updated": "2026-06-14", "authors": [ "Mohammed Fasha", "Nahel Al-Maayta", "Bilal Sowan", "Mohammad Athamneh", "Husam Barham" ], "categories": [ "cs.AI", "cs.MA" ], "topics": [ "agent-evaluation", "agent-safety", "rag", "tool-use", "workflow-agent", "world-model" ], "score": 13, "relevance": "high", "matched_queries": [ "function-calling", "rag-agent" ], "arxiv_id": "2606.15709", "source": "arxiv", "source_id": "arxiv:2606.15709", "pdf_url": "https://arxiv.org/pdf/2606.15709", "primary_query": "function-calling" }, { "id": "2606.15874", "title": "LLM-as-Code: Agentic Programming for Agent Harness", "url": "https://arxiv.org/abs/2606.15874", "published": "2026-06-14", "updated": "2026-06-22", "authors": [ "Junjia Qi", "Zichuan Fu", "Jingtong Gao", "Wenlin Zhang", "Hanyu Yan", "Xian Wu", "Xiangyu Zhao" ], "categories": [ "cs.AI", "cs.SE" ], "topics": [ "reasoning", "tool-use" ], "score": 13, "relevance": "high", "matched_queries": [ "web-gui-agent" ], "arxiv_id": "2606.15874", "source": "arxiv", "source_id": "arxiv:2606.15874", "pdf_url": "https://arxiv.org/pdf/2606.15874", "primary_query": "web-gui-agent" }, { "id": "2606.15906", "title": "MAGE-RAG: Multigranular Adaptive Graph Evidence for Agentic Multimodal RAG in Long-Document QA", "url": "https://arxiv.org/abs/2606.15906", "published": "2026-06-14", "updated": "2026-06-14", "authors": [ "Yilong Zuo", "Xunkai Li", "Jing Yuan", "Qiangqiang Dai", "Hongchao Qin", "Ronghua Li" ], "categories": [ "cs.IR", "cs.AI", "cs.CL", "cs.DB", "cs.MM" ], "topics": [ "agent-evaluation", "coding-agent", "rag" ], "score": 13, "relevance": "high", "matched_queries": [ "rag-agent" ], "arxiv_id": "2606.15906", "source": "arxiv", "source_id": "arxiv:2606.15906", "pdf_url": "https://arxiv.org/pdf/2606.15906", "primary_query": "rag-agent" }, { "id": "2606.15994", "title": "Agentic Framework for Deep Learning workload migration via In-Context Learning", "url": "https://arxiv.org/abs/2606.15994", "published": "2026-06-14", "updated": "2026-06-14", "authors": [ "Qiyue Liang", "Steven Ingram", "George Vanica", "Andi Gavrilescu", "Newfel Harrat", "Hassan Sipra", "Sethuraman Sankaran" ], "categories": [ "cs.AI", "cs.LG" ], "topics": [ "agent-safety", "rag", "tool-use" ], "score": 13, "relevance": "high", "matched_queries": [ "autonomous-agent-llm" ], "arxiv_id": "2606.15994", "source": "arxiv", "source_id": "arxiv:2606.15994", "pdf_url": "https://arxiv.org/pdf/2606.15994", "primary_query": "autonomous-agent-llm" }, { "id": "2606.15034", "title": "OSGuard: A Benchmark for Safety in Computer-Use Agents", "url": "https://arxiv.org/abs/2606.15034", "published": "2026-06-13", "updated": "2026-06-13", "authors": [ "Mina Mohammadmirzaei", "Jeffrey Flanigan" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "agent-safety", "computer-use", "tool-use" ], "score": 13, "relevance": "high", "matched_queries": [ "web-gui-agent" ], "arxiv_id": "2606.15034", "source": "arxiv", "source_id": "arxiv:2606.15034", "pdf_url": "https://arxiv.org/pdf/2606.15034", "primary_query": "web-gui-agent" }, { "id": "2606.12837", "title": "LoHoSearch: Benchmarking Long-Horizon Search Agents Beyond the Human Difficulty Ceiling", "url": "https://arxiv.org/abs/2606.12837", "published": "2026-06-11", "updated": "2026-06-17", "authors": [ "Jiarui Zhao", "Rongzhi Zhang", "Lingchuan Liu", "Hao Yang", "Xunliang Cai", "Xi Su" ], "categories": [ "cs.CL" ], "topics": [ "agent-evaluation", "planning", "reasoning", "tool-use" ], "score": 13, "relevance": "high", "matched_queries": [ "agent-evaluation" ], "arxiv_id": "2606.12837", "source": "arxiv", "source_id": "arxiv:2606.12837", "pdf_url": "https://arxiv.org/pdf/2606.12837", "primary_query": "agent-evaluation" }, { "id": "2606.13663", "title": "HyperTool: Beyond Step-Wise Tool Calls for Tool-Augmented Agents", "url": "https://arxiv.org/abs/2606.13663", "published": "2026-06-11", "updated": "2026-06-11", "authors": [ "Yaxin Du", "Yifan Zhou", "Yujie Ge", "Jiajun Wang", "Xianghe Pang", "Shuo Tang", "Tuney Zheng", "Bryan Dai", "Jian Yang", "Siheng Chen" ], "categories": [ "cs.CL" ], "topics": [ "rag", "reasoning", "tool-use", "workflow-agent" ], "score": 13, "relevance": "high", "matched_queries": [ "tool-use" ], "arxiv_id": "2606.13663", "source": "arxiv", "source_id": "arxiv:2606.13663", "pdf_url": "https://arxiv.org/pdf/2606.13663", "primary_query": "tool-use" }, { "id": "2606.12634", "title": "Keep Policy Gradient in Charge: Sibling-Guided Credit Distillation for Long-Horizon Tool-Use Agents", "url": "https://arxiv.org/abs/2606.12634", "published": "2026-06-10", "updated": "2026-06-29", "authors": [ "Tianyu Ding", "Jianhong Xin", "Juan Pablo De la Cruz Weinstein" ], "categories": [ "cs.LG", "cs.AI", "cs.CL" ], "topics": [ "agent-evaluation", "computer-use", "planning", "reasoning", "tool-use" ], "score": 13, "relevance": "high", "matched_queries": [ "tool-use" ], "arxiv_id": "2606.12634", "source": "arxiv", "source_id": "arxiv:2606.12634", "pdf_url": "https://arxiv.org/pdf/2606.12634", "primary_query": "tool-use" }, { "id": "2606.11688", "title": "Goal-Autopilot: A Verifiable Anti-Fabrication Firewall for Unattended Long-Horizon Agents", "url": "https://arxiv.org/abs/2606.11688", "published": "2026-06-10", "updated": "2026-06-10", "authors": [ "Youwang Deng" ], "categories": [ "cs.CL", "cs.AI" ], "topics": [ "agent-evaluation", "coding-agent", "planning", "rag" ], "score": 13, "relevance": "high", "matched_queries": [ "planning-agent" ], "arxiv_id": "2606.11688", "source": "arxiv", "source_id": "arxiv:2606.11688", "pdf_url": "https://arxiv.org/pdf/2606.11688", "primary_query": "planning-agent" }, { "id": "2606.10394", "title": "STAGE-Claw: Automated State-based Agent Benchmarking for Realistic Scenarios", "url": "https://arxiv.org/abs/2606.10394", "published": "2026-06-09", "updated": "2026-06-09", "authors": [ "Sirui Liang", "Bohan Yu", "Peiyu Wang", "Shiguang Guo", "Wenxing Hu", "Pengfei Cao", "Jian Zhao", "Cao Liu", "Ke Zeng", "Xunliang Cai", "Kang Liu" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "tool-use" ], "score": 13, "relevance": "high", "matched_queries": [ "agent-evaluation" ], "arxiv_id": "2606.10394", "source": "arxiv", "source_id": "arxiv:2606.10394", "pdf_url": "https://arxiv.org/pdf/2606.10394", "primary_query": "agent-evaluation" }, { "id": "2606.10532", "title": "ActiveMem: Distributed Active Memory for Long-Horizon LLM Reasoning", "url": "https://arxiv.org/abs/2606.10532", "published": "2026-06-09", "updated": "2026-06-09", "authors": [ "Yunhan Jiang", "Wenbin Duan", "Shasha Guo", "Liang Pang", "Xiaoqian Sun", "Huawei Shen" ], "categories": [ "cs.AI" ], "topics": [ "agent-safety", "memory", "planning", "reasoning", "tool-use" ], "score": 13, "relevance": "high", "matched_queries": [ "agent-memory" ], "arxiv_id": "2606.10532", "source": "arxiv", "source_id": "arxiv:2606.10532", "pdf_url": "https://arxiv.org/pdf/2606.10532", "primary_query": "agent-memory" }, { "id": "2606.11176", "title": "Data Journalist Agent: Transforming Data into Verifiable Multimodal Stories", "url": "https://arxiv.org/abs/2606.11176", "published": "2026-06-09", "updated": "2026-06-09", "authors": [ "Kevin Qinghong Lin", "Batu EI", "Yuhong Shi", "Pan Lu", "Philip Torr", "James Zou" ], "categories": [ "cs.CV", "cs.CL", "cs.CY", "cs.HC" ], "topics": [ "agent-evaluation", "coding-agent", "multi-agent", "rag", "reasoning", "tool-use" ], "score": 13, "relevance": "high", "matched_queries": [ "web-gui-agent" ], "arxiv_id": "2606.11176", "source": "arxiv", "source_id": "arxiv:2606.11176", "pdf_url": "https://arxiv.org/pdf/2606.11176", "primary_query": "web-gui-agent" }, { "id": "2606.08960", "title": "Hardening Agent Benchmarks with Adversarial Hacker-Fixer Loops", "url": "https://arxiv.org/abs/2606.08960", "published": "2026-06-08", "updated": "2026-06-08", "authors": [ "Ziqian Zhong", "Ivgeni Segal", "Ivan Bercovich", "Shashwat Saxena", "Kexun Zhang", "Aditi Raghunathan" ], "categories": [ "cs.CR", "cs.AI", "cs.LG", "cs.MA" ], "topics": [ "agent-evaluation", "reasoning" ], "score": 13, "relevance": "high", "matched_queries": [ "agent-evaluation" ], "arxiv_id": "2606.08960", "source": "arxiv", "source_id": "arxiv:2606.08960", "pdf_url": "https://arxiv.org/pdf/2606.08960", "primary_query": "agent-evaluation" }, { "id": "2606.09447", "title": "AliyunConsoleAgent: Training Web Agents in Real-World Cloud Environments via Distillation and Reinforcement Learning", "url": "https://arxiv.org/abs/2606.09447", "published": "2026-06-08", "updated": "2026-06-08", "authors": [ "Bojie Rong", "Zheyu Shen", "Qiaoping Wang", "Pengfei Kang", "Yang Xu", "Yawen Wei", "Hanyu Wu", "Zhi Zhao", "Leihao Pei", "Linquan Jiang" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "computer-use", "rag", "tool-use" ], "score": 13, "relevance": "high", "matched_queries": [ "web-gui-agent" ], "arxiv_id": "2606.09447", "source": "arxiv", "source_id": "arxiv:2606.09447", "pdf_url": "https://arxiv.org/pdf/2606.09447", "primary_query": "web-gui-agent" }, { "id": "2606.09426", "title": "WeaveBench: A Long-Horizon, Real-World Benchmark for Computer-Use Agents with Hybrid Interfaces", "url": "https://arxiv.org/abs/2606.09426", "published": "2026-06-08", "updated": "2026-07-06", "authors": [ "Wanli Li", "Bowen Zhou", "Yunyao Yu", "Zhou Xu", "Yifan Yang", "Dongsheng Li", "Caihua Shan" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "computer-use", "planning", "tool-use" ], "score": 13, "relevance": "high", "matched_queries": [ "web-gui-agent" ], "arxiv_id": "2606.09426", "source": "arxiv", "source_id": "arxiv:2606.09426", "pdf_url": "https://arxiv.org/pdf/2606.09426", "primary_query": "web-gui-agent" }, { "id": "2606.09316", "title": "Anything2Skill: Compiling External Knowledge into Reusable Skills for Agents", "url": "https://arxiv.org/abs/2606.09316", "published": "2026-06-08", "updated": "2026-06-19", "authors": [ "Qianjun Pan", "Yutao Yang", "Junsong Li", "Jie Zhou", "Kai Chen", "Xin Li", "Qin Chen", "Liang He" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "computer-use", "memory", "planning", "rag", "tool-use", "workflow-agent" ], "score": 13, "relevance": "high", "matched_queries": [ "rag-agent" ], "arxiv_id": "2606.09316", "source": "arxiv", "source_id": "arxiv:2606.09316", "pdf_url": "https://arxiv.org/pdf/2606.09316", "primary_query": "rag-agent" }, { "id": "2606.09961", "title": "3SPO: State-Score-Supervised Policy Optimization for LLM Agents", "url": "https://arxiv.org/abs/2606.09961", "published": "2026-06-08", "updated": "2026-06-08", "authors": [ "Yu Han", "Kailing Li", "Yang Jiao", "Yulin Dai", "Yuqian Fu", "Linhai Zhuo", "Tianwen Qian" ], "categories": [ "cs.LG", "cs.AI" ], "topics": [ "computer-use", "planning", "tool-use" ], "score": 13, "relevance": "high", "matched_queries": [ "autonomous-agent-llm" ], "arxiv_id": "2606.09961", "source": "arxiv", "source_id": "arxiv:2606.09961", "pdf_url": "https://arxiv.org/pdf/2606.09961", "primary_query": "autonomous-agent-llm" }, { "id": "2606.08172", "title": "The Governance of Human-LLM Interaction: Safety Gating, Civility Steering, and Affective Default Lock-In", "url": "https://arxiv.org/abs/2606.08172", "published": "2026-06-06", "updated": "2026-06-06", "authors": [ "Manuele Reani", "Hongjian Zhang", "Hongyu Tian" ], "categories": [ "cs.HC", "cs.AI", "cs.CY" ], "topics": [ "agent-evaluation", "agent-safety", "computer-use", "multi-agent", "planning", "tool-use" ], "score": 13, "relevance": "high", "matched_queries": [ "agent-evaluation" ], "arxiv_id": "2606.08172", "source": "arxiv", "source_id": "arxiv:2606.08172", "pdf_url": "https://arxiv.org/pdf/2606.08172", "primary_query": "agent-evaluation" }, { "id": "2606.08162", "title": "Silent Failure in LLM Agent Systems: The Entropy Principle and the Inevitable Disorder of Autonomous Agents", "url": "https://arxiv.org/abs/2606.08162", "published": "2026-06-06", "updated": "2026-06-06", "authors": [ "Dexing Liu" ], "categories": [ "cs.MA" ], "topics": [ "memory", "tool-use" ], "score": 13, "relevance": "high", "matched_queries": [ "autonomous-agent-llm" ], "arxiv_id": "2606.08162", "source": "arxiv", "source_id": "arxiv:2606.08162", "pdf_url": "https://arxiv.org/pdf/2606.08162", "primary_query": "autonomous-agent-llm" }, { "id": "2606.07836", "title": "Agentic multi-fidelity learning of quasiparticle and excitonic properties", "url": "https://arxiv.org/abs/2606.07836", "published": "2026-06-05", "updated": "2026-06-05", "authors": [ "Arnab Neogi", "Aaron Forde", "Christopher A. Lane", "Sergei Tretiak", "Jian-Xin Zhu" ], "categories": [ "cond-mat.mtrl-sci", "cond-mat.stat-mech", "cs.AI", "physics.comp-ph", "quant-ph" ], "topics": [ "agent-evaluation", "computer-use", "rag", "workflow-agent", "world-model" ], "score": 13, "relevance": "high", "matched_queries": [ "agent-evaluation" ], "arxiv_id": "2606.07836", "source": "arxiv", "source_id": "arxiv:2606.07836", "pdf_url": "https://arxiv.org/pdf/2606.07836", "primary_query": "agent-evaluation" }, { "id": "2606.18272", "title": "Mitigating Anchoring Bias in LLM-Based Agents for Energy-Efficient 6G Autonomous Networks", "url": "https://arxiv.org/abs/2606.18272", "published": "2026-06-05", "updated": "2026-06-18", "authors": [ "Hatim Chergui", "Claudia Carballo González", "Farhad Rezazadeh", "Merouane Debbah" ], "categories": [ "cs.NI", "cs.AI", "eess.SY" ], "topics": [ "agent-safety", "computer-use", "multi-agent", "reasoning" ], "score": 13, "relevance": "high", "matched_queries": [ "autonomous-agent-llm" ], "arxiv_id": "2606.18272", "source": "arxiv", "source_id": "arxiv:2606.18272", "pdf_url": "https://arxiv.org/pdf/2606.18272", "primary_query": "autonomous-agent-llm" }, { "id": "2606.05658", "title": "Agent-Orchestrated Adaptive RAG: A Comparative Study on Structured and Multi-Hop Retrieval", "url": "https://arxiv.org/abs/2606.05658", "published": "2026-06-04", "updated": "2026-06-04", "authors": [ "Anuj Maharjan", "Devinder Kaur", "Richard Molyet" ], "categories": [ "cs.IR", "cs.AI" ], "topics": [ "agent-evaluation", "rag", "reasoning" ], "score": 13, "relevance": "high", "matched_queries": [ "rag-agent" ], "arxiv_id": "2606.05658", "source": "arxiv", "source_id": "arxiv:2606.05658", "pdf_url": "https://arxiv.org/pdf/2606.05658", "primary_query": "rag-agent" }, { "id": "2606.05622", "title": "AdaPlanBench: Evaluating Adaptive Planning in Large Language Model Agents under World and User Constraints", "url": "https://arxiv.org/abs/2606.05622", "published": "2026-06-04", "updated": "2026-06-04", "authors": [ "Jiayu Liu", "Cheng Qian", "Zhenhailong Wang", "Bingxuan Li", "Jiateng Liu", "Heng Wang", "Jeonghwan Kim", "Yumeng Wang", "Xiusi Chen", "Yi R. Fung", "Heng Ji" ], "categories": [ "cs.CL" ], "topics": [ "agent-evaluation", "planning", "rag", "tool-use" ], "score": 13, "relevance": "high", "matched_queries": [ "planning-agent" ], "arxiv_id": "2606.05622", "source": "arxiv", "source_id": "arxiv:2606.05622", "pdf_url": "https://arxiv.org/pdf/2606.05622", "primary_query": "planning-agent" }, { "id": "2606.05436", "title": "Ten Headache Specialists versus Artificial Intelligence for Clinical Literature Summarization: A Critical Evaluation and Comparison", "url": "https://arxiv.org/abs/2606.05436", "published": "2026-06-03", "updated": "2026-06-03", "authors": [ "Alejandro Lozano", "Keiko Ihara", "Ping-Hao Yang", "Carrie E. Robertson", "Jennifer Stern", "Allan Purdy", "Hsiangkuo Yuan", "Pengfei Zhang", "Yulia Orlova", "Olga Fermo", "Jennifer Hranilovich", "Fred Cohen", "Todd J. Schwedt", "Jenelle A. Jindal", "Serena Yeung-Levy", "Chia-Chun Chiang" ], "categories": [ "cs.AI", "cs.CL", "cs.IR" ], "topics": [ "agent-evaluation", "computer-use", "rag", "tool-use" ], "score": 13, "relevance": "high", "matched_queries": [ "rag-agent" ], "arxiv_id": "2606.05436", "source": "arxiv", "source_id": "arxiv:2606.05436", "pdf_url": "https://arxiv.org/pdf/2606.05436", "primary_query": "rag-agent" }, { "id": "2606.03544", "title": "SAGE: A Quantitative Evaluation of Socialized Evolution in Agent Ecosystems", "url": "https://arxiv.org/abs/2606.03544", "published": "2026-06-02", "updated": "2026-06-02", "authors": [ "Linyue Pan", "Yaoming Zhu", "Lin Qiu", "Xuezhi Cao", "Xunliang Cai" ], "categories": [ "cs.AI", "cs.CL" ], "topics": [ "agent-evaluation", "planning", "reasoning", "tool-use" ], "score": 13, "relevance": "high", "matched_queries": [ "language-agent" ], "arxiv_id": "2606.03544", "source": "arxiv", "source_id": "arxiv:2606.03544", "pdf_url": "https://arxiv.org/pdf/2606.03544", "primary_query": "language-agent" }, { "id": "2606.03157", "title": "ClinicalMC: A Benchmark for Multi-Course Clinical Decision-Making with Large Language Models", "url": "https://arxiv.org/abs/2606.03157", "published": "2026-06-02", "updated": "2026-06-02", "authors": [ "Ruihui Hou", "Siyi Zhu", "Ziyue Huai", "Guangya Yu", "Yongqi Fan", "Chunming Wang", "Tong Ruan" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "multi-agent", "rag" ], "score": 13, "relevance": "high", "matched_queries": [ "agent-evaluation" ], "arxiv_id": "2606.03157", "source": "arxiv", "source_id": "arxiv:2606.03157", "pdf_url": "https://arxiv.org/pdf/2606.03157", "primary_query": "agent-evaluation" }, { "id": "2606.01961", "title": "AutoMedBench: Towards Medical AutoResearch with Agentic AI Models", "url": "https://arxiv.org/abs/2606.01961", "published": "2026-06-01", "updated": "2026-06-03", "authors": [ "Junqi Liu", "Selena Song", "Yuhan Wang", "Jiawei Mao", "Hardy Chen", "Xiaoke Huang", "Tianhao Qi", "Pengfei Guo", "Yucheng Tang", "Yufan He", "Can Zhao", "Andriy Myronenko", "Dong Yang", "Daguang Xu", "Yuyin Zhou" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "planning", "rag", "workflow-agent" ], "score": 13, "relevance": "high", "matched_queries": [ "agent-evaluation" ], "arxiv_id": "2606.01961", "source": "arxiv", "source_id": "arxiv:2606.01961", "pdf_url": "https://arxiv.org/pdf/2606.01961", "primary_query": "agent-evaluation" }, { "id": "2606.01185", "title": "\"Skill issues'': data-centric optimization of lakehouse agents", "url": "https://arxiv.org/abs/2606.01185", "published": "2026-05-31", "updated": "2026-05-31", "authors": [ "Nicole Rose Schneider", "Davide Ghilardi", "Giacomo Piccinini", "Jacopo Tagliabue" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "coding-agent", "planning", "reasoning", "tool-use", "workflow-agent" ], "score": 13, "relevance": "high", "matched_queries": [ "agent-evaluation" ], "arxiv_id": "2606.01185", "source": "arxiv", "source_id": "arxiv:2606.01185", "pdf_url": "https://arxiv.org/pdf/2606.01185", "primary_query": "agent-evaluation" }, { "id": "2606.01138", "title": "memorywire: A Vendor-Neutral Wire Format for Agent Memory Operations", "url": "https://arxiv.org/abs/2606.01138", "published": "2026-05-31", "updated": "2026-06-03", "authors": [ "Thamilvendhan Munirathinam" ], "categories": [ "cs.CR", "cs.AI", "cs.DC" ], "topics": [ "agent-evaluation", "memory", "rag", "workflow-agent" ], "score": 13, "relevance": "high", "matched_queries": [ "agent-memory" ], "arxiv_id": "2606.01138", "source": "arxiv", "source_id": "arxiv:2606.01138", "pdf_url": "https://arxiv.org/pdf/2606.01138", "primary_query": "agent-memory" }, { "id": "2606.01166", "title": "BraveGuard: From Open-World Threats to Safer Computer-Use Agents", "url": "https://arxiv.org/abs/2606.01166", "published": "2026-05-31", "updated": "2026-06-02", "authors": [ "Yunhao Feng", "Xiaohu Du", "Xinhao Deng", "Yifan Ding", "Ming Wen", "Yixu Wang", "Yuxiang Xie", "Baihui Zheng", "Yingshui Tan", "Yige Li", "Yutao Wu", "Kerui Cao", "Wenke Huang", "Yanming Guo", "Xingjun Ma", "Yu-Gang Jiang" ], "categories": [ "cs.CR", "cs.CL" ], "topics": [ "agent-evaluation", "agent-safety", "computer-use", "rag", "tool-use" ], "score": 13, "relevance": "high", "matched_queries": [ "agent-safety" ], "arxiv_id": "2606.01166", "source": "arxiv", "source_id": "arxiv:2606.01166", "pdf_url": "https://arxiv.org/pdf/2606.01166", "primary_query": "agent-safety" }, { "id": "2606.00644", "title": "ForeSci: Evaluating LLM Agents for Forward-Looking AI Research Judgment", "url": "https://arxiv.org/abs/2606.00644", "published": "2026-05-30", "updated": "2026-06-04", "authors": [ "Qiuyu Tian", "Haojie Yin", "Yingce Xia", "Youyong Kong", "Zequn Liu" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "rag" ], "score": 13, "relevance": "high", "matched_queries": [ "rag-agent" ], "arxiv_id": "2606.00644", "source": "arxiv", "source_id": "arxiv:2606.00644", "pdf_url": "https://arxiv.org/pdf/2606.00644", "primary_query": "rag-agent" }, { "id": "2605.31308", "title": "TraceGraph: Shared Decision Landscapes for Diagnosing and Improving Agent Trajectories", "url": "https://arxiv.org/abs/2605.31308", "published": "2026-05-29", "updated": "2026-05-29", "authors": [ "Junjie Nian", "Kang Chen", "Ge Zhang", "Yixin Cao", "Yugang Jiang" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "coding-agent", "computer-use", "embodied-agent", "tool-use" ], "score": 13, "relevance": "high", "matched_queries": [ "agent-evaluation" ], "arxiv_id": "2605.31308", "source": "arxiv", "source_id": "arxiv:2605.31308", "pdf_url": "https://arxiv.org/pdf/2605.31308", "primary_query": "agent-evaluation" }, { "id": "2605.31075", "title": "Task-Focused Memorization for Multimodal Agents", "url": "https://arxiv.org/abs/2605.31075", "published": "2026-05-29", "updated": "2026-05-29", "authors": [ "Tao Zou", "Yichen He", "Tian Qiu", "Yuan Lin", "Hang Li" ], "categories": [ "cs.CV" ], "topics": [ "agent-evaluation", "computer-use", "embodied-agent", "memory" ], "score": 13, "relevance": "high", "matched_queries": [ "agent-memory" ], "arxiv_id": "2605.31075", "source": "arxiv", "source_id": "arxiv:2605.31075", "pdf_url": "https://arxiv.org/pdf/2605.31075", "primary_query": "agent-memory" }, { "id": "2605.31268", "title": "Mellum2 Technical Report", "url": "https://arxiv.org/abs/2605.31268", "published": "2026-05-29", "updated": "2026-05-29", "authors": [ "Marko Kojic", "Ivan Bondyrev", "Aral de Moor", "Joseph Shtok", "Petr Borovlev", "Kseniia Lysaniuk", "Madeeswaran Kannan", "Ivan Dolgov", "Nikita Pavlichenko" ], "categories": [ "cs.CL" ], "topics": [ "agent-evaluation", "agent-safety", "coding-agent", "reasoning", "tool-use" ], "score": 13, "relevance": "high", "matched_queries": [ "function-calling" ], "arxiv_id": "2605.31268", "source": "arxiv", "source_id": "arxiv:2605.31268", "pdf_url": "https://arxiv.org/pdf/2605.31268", "primary_query": "function-calling" }, { "id": "2605.29653", "title": "PTCG-Bench: Can LLM Agents Master Pokémon Trading Card Game?", "url": "https://arxiv.org/abs/2605.29653", "published": "2026-05-28", "updated": "2026-05-28", "authors": [ "Dongdong Hua", "Yifei Sun", "Renhong Huang", "Feng Gao", "Chunping Wang", "Yang Yang" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation" ], "score": 13, "relevance": "high", "matched_queries": [ "agent-evaluation", "autonomous-agent-llm" ], "arxiv_id": "2605.29653", "source": "arxiv", "source_id": "arxiv:2605.29653", "pdf_url": "https://arxiv.org/pdf/2605.29653", "primary_query": "agent-evaluation" }, { "id": "2605.29630", "title": "Entity-Collision: A Stratified Protocol for Attributing Retrieval Lift in Agent Memory", "url": "https://arxiv.org/abs/2605.29630", "published": "2026-05-28", "updated": "2026-05-28", "authors": [ "Youwang Deng" ], "categories": [ "cs.CL", "cs.AI", "cs.IR" ], "topics": [ "agent-evaluation", "memory", "rag", "tool-use" ], "score": 13, "relevance": "high", "matched_queries": [ "agent-memory" ], "arxiv_id": "2605.29630", "source": "arxiv", "source_id": "arxiv:2605.29630", "pdf_url": "https://arxiv.org/pdf/2605.29630", "primary_query": "agent-memory" }, { "id": "2606.07591", "title": "ResearchClawBench: A Benchmark for End-to-End Autonomous Scientific Research", "url": "https://arxiv.org/abs/2606.07591", "published": "2026-05-28", "updated": "2026-07-03", "authors": [ "Wanghan Xu", "Shuo Li", "Tianlin Ye", "Qinglong Cao", "Yixin Chen", "Hengjian Gao", "Yiheng Wang", "Qi Li", "Kun Li", "Sheng Xu", "Shengdu Chai", "Fangchen Yu", "Xiangyu Zhao", "Zhangrui Zhao", "Weijie Ma", "Zijie Guo", "Koutian Wu", "Haoyu Zhou", "Haoxiang Yin", "Lixue Cheng", "Chaofan Hu", "Haoxuan Li", "Lu Mi", "Xuxuan Xie", "Yifan Zhou", "Ruizhe Chen", "Zhiwang Zhou", "Xingjian Guo", "Yuhao Zhou", "Xuming He", "Shengyuan Xu", "Xinyu Gu", "Jiamin Wu", "Mianxin Liu", "Chunfeng Song", "Fenghua Ling", "Dongzhan Zhou", "Shixiang Tang", "Yuqiang Li", "Mao Su", "Peng Ye", "Siqi Sun", "Bin Wang", "Xue Yang", "Zhenfei Yin", "Tianfan Fu", "Guangtao Zhai", "Wanli Ouyang", "Bo Zhang", "Lei Bai", "Wenlong Zhang" ], "categories": [ "cs.LG", "cs.AI", "cs.CL" ], "topics": [ "agent-evaluation", "coding-agent", "rag" ], "score": 13, "relevance": "high", "matched_queries": [ "autonomous-agent-llm" ], "arxiv_id": "2606.07591", "source": "arxiv", "source_id": "arxiv:2606.07591", "pdf_url": "https://arxiv.org/pdf/2606.07591", "primary_query": "autonomous-agent-llm" }, { "id": "2605.28617", "title": "LACUNA: Safe Agents as Recursive Program Holes", "url": "https://arxiv.org/abs/2605.28617", "published": "2026-05-27", "updated": "2026-05-27", "authors": [ "Yaoyu Zhao", "Yichen Xu", "Oliver Bračevac", "Cao Nguyen Pham", "Frank Zhengqing Wu", "Martin Odersky" ], "categories": [ "cs.AI", "cs.PL" ], "topics": [ "agent-evaluation", "agent-safety", "planning", "rag", "tool-use" ], "score": 13, "relevance": "high", "matched_queries": [ "planning-agent" ], "arxiv_id": "2605.28617", "source": "arxiv", "source_id": "arxiv:2605.28617", "pdf_url": "https://arxiv.org/pdf/2605.28617", "primary_query": "planning-agent" }, { "id": "2605.28424", "title": "Skill0.5: Joint Skill Internalization and Utilization for Out-of-Distribution Generalization in Agentic Reinforcement Learning", "url": "https://arxiv.org/abs/2605.28424", "published": "2026-05-27", "updated": "2026-05-27", "authors": [ "Jiapeng Zhu", "Jianxiang Yu", "Yibo Zhao", "Chengcheng Han", "Qi Gu", "Xunliang Cai", "Xiang Li", "Weining Qian" ], "categories": [ "cs.CL" ], "topics": [ "agent-safety", "memory" ], "score": 13, "relevance": "high", "matched_queries": [ "autonomous-agent-llm" ], "arxiv_id": "2605.28424", "source": "arxiv", "source_id": "arxiv:2605.28424", "pdf_url": "https://arxiv.org/pdf/2605.28424", "primary_query": "autonomous-agent-llm" }, { "id": "2605.26497", "title": "Aligning Provenance with Authorization: A Dual-Graph Defense for LLM Agents", "url": "https://arxiv.org/abs/2605.26497", "published": "2026-05-26", "updated": "2026-05-26", "authors": [ "Peiran Wang", "Ying Li", "Yuan Tian" ], "categories": [ "cs.CR" ], "topics": [ "agent-safety", "reasoning", "tool-use" ], "score": 13, "relevance": "high", "matched_queries": [ "agent-safety" ], "arxiv_id": "2605.26497", "source": "arxiv", "source_id": "arxiv:2605.26497", "pdf_url": "https://arxiv.org/pdf/2605.26497", "primary_query": "agent-safety" }, { "id": "2605.27123", "title": "Rethinking Agentic RAG: Toward LLM-Driven Logical Retrieval Beyond Embeddings", "url": "https://arxiv.org/abs/2605.27123", "published": "2026-05-26", "updated": "2026-05-26", "authors": [ "Yuqi Zeng", "Qixiang Deng", "Yulei Wan", "Ruiquan Jiang", "Xiaoqing Zheng", "Xuanjing Huang" ], "categories": [ "cs.IR" ], "topics": [ "agent-evaluation", "computer-use", "rag" ], "score": 13, "relevance": "high", "matched_queries": [ "rag-agent" ], "arxiv_id": "2605.27123", "source": "arxiv", "source_id": "arxiv:2605.27123", "pdf_url": "https://arxiv.org/pdf/2605.27123", "primary_query": "rag-agent" }, { "id": "2605.26165", "title": "Tool-Schema Compression Enables Agentic RAG Under Constrained Context Budgets", "url": "https://arxiv.org/abs/2605.26165", "published": "2026-05-24", "updated": "2026-05-24", "authors": [ "Furkan Sakizli" ], "categories": [ "cs.SE", "cs.AI", "cs.CL" ], "topics": [ "agent-evaluation", "rag", "tool-use" ], "score": 13, "relevance": "high", "matched_queries": [ "rag-agent" ], "arxiv_id": "2605.26165", "source": "arxiv", "source_id": "arxiv:2605.26165", "pdf_url": "https://arxiv.org/pdf/2605.26165", "primary_query": "rag-agent" }, { "id": "2605.23899", "title": "From Raw Experience to Skill Consumption: A Systematic Study of Model-Generated Agent Skills", "url": "https://arxiv.org/abs/2605.23899", "published": "2026-05-22", "updated": "2026-05-22", "authors": [ "Zisu Huang", "Jingwen Xu", "Yifan Yang", "Ziyang Gong", "Qihao Yang", "Muzhao Tian", "Xiaohua Wang", "Changze Lv", "Xuemei Gao", "Qi Dai", "Bei Liu", "Kai Qiu", "Xue Yang", "Dongdong Chen", "Xiaoqing Zheng", "Chong Luo" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "coding-agent", "computer-use", "rag", "tool-use" ], "score": 13, "relevance": "high", "matched_queries": [ "language-agent" ], "arxiv_id": "2605.23899", "source": "arxiv", "source_id": "arxiv:2605.23899", "pdf_url": "https://arxiv.org/pdf/2605.23899", "primary_query": "language-agent" }, { "id": "2605.17075", "title": "A Red Teaming Framework for Evaluating Robustness of AI-enabled Security Orchestration, Automation, and Response Systems", "url": "https://arxiv.org/abs/2605.17075", "published": "2026-05-16", "updated": "2026-05-16", "authors": [ "Ayan Javeed Shaikh", "Nathaniel D. Bastian", "Ankit Shah" ], "categories": [ "cs.CR" ], "topics": [ "agent-evaluation", "agent-safety", "planning", "tool-use", "workflow-agent", "world-model" ], "score": 13, "relevance": "high", "matched_queries": [ "autonomous-agent-llm" ], "arxiv_id": "2605.17075", "source": "arxiv", "source_id": "arxiv:2605.17075", "pdf_url": "https://arxiv.org/pdf/2605.17075", "primary_query": "autonomous-agent-llm" }, { "id": "2605.14322", "title": "Are Agents Ready to Teach? A Multi-Stage Benchmark for Real-World Teaching Workflows", "url": "https://arxiv.org/abs/2605.14322", "published": "2026-05-14", "updated": "2026-05-20", "authors": [ "Zixin Chen", "Peng Liu", "Rui Sheng", "Haobo Li", "Jianhong Tu", "Xiaodong Deng", "Kashun Shum", "Dayiheng Liu", "Huamin Qu" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "tool-use", "workflow-agent" ], "score": 13, "relevance": "high", "matched_queries": [ "language-agent" ], "arxiv_id": "2605.14322", "source": "arxiv", "source_id": "arxiv:2605.14322", "pdf_url": "https://arxiv.org/pdf/2605.14322", "primary_query": "language-agent" }, { "id": "2605.11928", "title": "When Simulation Lies: A Sim-to-Real Benchmark and Domain-Randomized RL Recipe for Tool-Use Agents", "url": "https://arxiv.org/abs/2605.11928", "published": "2026-05-12", "updated": "2026-05-12", "authors": [ "Xiaolin Zhou", "Aojie Yuan", "Zheng Luo", "Zipeng Ling", "Xixiao Pan", "Yicheng Gao", "Haiyue Zhang", "Jiate Li", "Shuli Jiang", "Prince Zizhuang Wang", "Zixuan Zhu", "Jinbo Liu", "Ryan A. Rossi", "Hua Wei", "Xiyang Hu" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "tool-use", "world-model" ], "score": 13, "relevance": "high", "matched_queries": [ "function-calling", "language-agent" ], "arxiv_id": "2605.11928", "source": "arxiv", "source_id": "arxiv:2605.11928", "pdf_url": "https://arxiv.org/pdf/2605.11928", "primary_query": "function-calling" }, { "id": "2605.10870", "title": "Remember the Decision, Not the Description: A Rate-Distortion Framework for Agent Memory", "url": "https://arxiv.org/abs/2605.10870", "published": "2026-05-11", "updated": "2026-05-11", "authors": [ "Mingxi Zou", "Zhihan Guo", "Langzhang Liang", "Zhuo Wang", "Qifan Wang", "Qingsong Wen", "Irwin King", "Lizhen Qu", "Zenglin Xu" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "memory", "planning" ], "score": 13, "relevance": "high", "matched_queries": [ "language-agent" ], "arxiv_id": "2605.10870", "source": "arxiv", "source_id": "arxiv:2605.10870", "pdf_url": "https://arxiv.org/pdf/2605.10870", "primary_query": "language-agent" }, { "id": "2605.08876", "title": "OTora: A Unified Red Teaming Framework for Reasoning-Level Denial-of-Service in LLM Agents", "url": "https://arxiv.org/abs/2605.08876", "published": "2026-05-09", "updated": "2026-06-07", "authors": [ "Xinyu Li", "Ronghui Mu", "Lin Li", "Tianjin Huang", "Gaojie Jin" ], "categories": [ "cs.LG" ], "topics": [ "computer-use", "reasoning", "tool-use" ], "score": 13, "relevance": "high", "matched_queries": [ "autonomous-agent-llm" ], "arxiv_id": "2605.08876", "source": "arxiv", "source_id": "arxiv:2605.08876", "pdf_url": "https://arxiv.org/pdf/2605.08876", "primary_query": "autonomous-agent-llm" }, { "id": "2605.03328", "title": "LLM-ADAM: A Generalizable LLM Agent Framework for Pre-Print Anomaly Detection in Additive Manufacturing", "url": "https://arxiv.org/abs/2605.03328", "published": "2026-05-05", "updated": "2026-05-05", "authors": [ "Ahmadreza Eslaminia", "Chuhan Cai", "Cameron Smith", "Ruo-Syuan Mei", "Shichen Li", "Rajiv Malhotra", "Klara Nahrstedt", "Chenhui Shao" ], "categories": [ "cs.LG", "cs.AI" ], "topics": [ "agent-evaluation", "computer-use", "planning" ], "score": 13, "relevance": "high", "matched_queries": [ "planning-agent" ], "arxiv_id": "2605.03328", "source": "arxiv", "source_id": "arxiv:2605.03328", "pdf_url": "https://arxiv.org/pdf/2605.03328", "primary_query": "planning-agent" }, { "id": "2604.27092", "title": "End-to-end autonomous scientific discovery on a real optical platform", "url": "https://arxiv.org/abs/2604.27092", "published": "2026-04-29", "updated": "2026-04-29", "authors": [ "Shuxing Yang", "Fujia Chen", "Rui Zhao", "Junyao Wu", "Yize Wang", "Haiyao Luo", "Ning Han", "Qiaolu Chen", "Yuze Hu", "Wenhao Li", "Mingzhu Li", "Hongsheng Chen", "Yihao Yang" ], "categories": [ "cs.AI", "physics.optics" ], "topics": [ "memory", "planning", "reasoning", "tool-use", "workflow-agent" ], "score": 13, "relevance": "high", "matched_queries": [ "autonomous-agent-llm" ], "arxiv_id": "2604.27092", "source": "arxiv", "source_id": "arxiv:2604.27092", "pdf_url": "https://arxiv.org/pdf/2604.27092", "primary_query": "autonomous-agent-llm" }, { "id": "2604.20994", "title": "Breaking MCP with Function Hijacking Attacks: Novel Threats for Function Calling and Agentic Models", "url": "https://arxiv.org/abs/2604.20994", "published": "2026-04-22", "updated": "2026-04-22", "authors": [ "Yannis Belkhiter", "Giulio Zizzo", "Sergio Maffeis", "Seshu Tirupathi", "John D. Kelleher" ], "categories": [ "cs.CR", "cs.AI", "cs.CL" ], "topics": [ "agent-safety", "reasoning", "tool-use" ], "score": 13, "relevance": "high", "matched_queries": [ "function-calling" ], "arxiv_id": "2604.20994", "source": "arxiv", "source_id": "arxiv:2604.20994", "pdf_url": "https://arxiv.org/pdf/2604.20994", "primary_query": "function-calling" }, { "id": "2603.28900", "title": "Robust Multi-Agent Reinforcement Learning for Small UAS Separation Assurance under GPS Degradation and Spoofing", "url": "https://arxiv.org/abs/2603.28900", "published": "2026-03-30", "updated": "2026-03-30", "authors": [ "Alex Zongo", "Filippos Fotiadis", "Ufuk Topcu", "Peng Wei" ], "categories": [ "cs.RO", "cs.AI", "cs.LG", "eess.SY" ], "topics": [ "agent-evaluation", "agent-safety", "multi-agent", "world-model" ], "score": 13, "relevance": "high", "matched_queries": [ "agent-safety" ], "arxiv_id": "2603.28900", "source": "arxiv", "source_id": "arxiv:2603.28900", "pdf_url": "https://arxiv.org/pdf/2603.28900", "primary_query": "agent-safety" }, { "id": "2603.25353", "title": "SafeGuard ASF: SR Agentic Humanoid Robot System for Autonomous Industrial Safety", "url": "https://arxiv.org/abs/2603.25353", "published": "2026-03-26", "updated": "2026-03-26", "authors": [ "Thanh Nguyen Canh", "Thang Tran Viet", "Thanh Tuan Tran", "Ben Wei Lim" ], "categories": [ "cs.RO" ], "topics": [ "agent-safety", "embodied-agent", "reasoning", "tool-use", "world-model" ], "score": 13, "relevance": "high", "matched_queries": [ "agent-safety" ], "arxiv_id": "2603.25353", "source": "arxiv", "source_id": "arxiv:2603.25353", "pdf_url": "https://arxiv.org/pdf/2603.25353", "primary_query": "agent-safety" }, { "id": "2603.19684", "title": "TSegAgent: Zero-Shot Tooth Segmentation via Geometry-Aware Vision-Language Agents", "url": "https://arxiv.org/abs/2603.19684", "published": "2026-03-20", "updated": "2026-06-23", "authors": [ "Shaojie Zhuang", "Lu Yin", "Guangshun Wei", "Yunpeng Li", "Xilu Wang", "Yuanfeng Zhou" ], "categories": [ "cs.CV" ], "topics": [ "agent-evaluation", "coding-agent", "rag", "reasoning", "tool-use" ], "score": 13, "relevance": "high", "matched_queries": [ "language-agent" ], "arxiv_id": "2603.19684", "source": "arxiv", "source_id": "arxiv:2603.19684", "pdf_url": "https://arxiv.org/pdf/2603.19684", "primary_query": "language-agent" }, { "id": "2603.17392", "title": "Agentic Cognitive Profiling: Realigning Automated Alzheimer's Disease Detection with Clinical Construct Validity", "url": "https://arxiv.org/abs/2603.17392", "published": "2026-03-18", "updated": "2026-03-18", "authors": [ "Jiawen Kang", "Kun Li", "Dongrui Han", "Jinchao Li", "Junan Li", "Lingwei Meng", "Xixin Wu", "Helen Meng" ], "categories": [ "cs.MA", "cs.IR", "q-bio.NC" ], "topics": [ "agent-evaluation", "tool-use" ], "score": 13, "relevance": "high", "matched_queries": [ "function-calling" ], "arxiv_id": "2603.17392", "source": "arxiv", "source_id": "arxiv:2603.17392", "pdf_url": "https://arxiv.org/pdf/2603.17392", "primary_query": "function-calling" }, { "id": "2603.15666", "title": "Compiled Memory: Not More Information, but More Precise Instructions for Language Agents", "url": "https://arxiv.org/abs/2603.15666", "published": "2026-03-12", "updated": "2026-03-12", "authors": [ "James Rhodes", "George Kang" ], "categories": [ "cs.AI" ], "topics": [ "memory", "rag" ], "score": 13, "relevance": "high", "matched_queries": [ "language-agent" ], "arxiv_id": "2603.15666", "source": "arxiv", "source_id": "arxiv:2603.15666", "pdf_url": "https://arxiv.org/pdf/2603.15666", "primary_query": "language-agent" }, { "id": "2603.00801", "title": "The Synthetic Web: Adversarially-Curated Mini-Internets for Diagnosing Epistemic Weaknesses of Language Agents", "url": "https://arxiv.org/abs/2603.00801", "published": "2026-02-28", "updated": "2026-02-28", "authors": [ "Shrey Shah", "Levent Ozgur" ], "categories": [ "cs.AI", "cs.IR" ], "topics": [ "agent-evaluation", "embodied-agent", "rag", "tool-use" ], "score": 13, "relevance": "high", "matched_queries": [ "language-agent" ], "arxiv_id": "2603.00801", "source": "arxiv", "source_id": "arxiv:2603.00801", "pdf_url": "https://arxiv.org/pdf/2603.00801", "primary_query": "language-agent" }, { "id": "2602.21127", "title": "\"Are You Sure?\": An Empirical Study of Human Perception Vulnerability in LLM-Driven Agentic Systems", "url": "https://arxiv.org/abs/2602.21127", "published": "2026-02-24", "updated": "2026-02-24", "authors": [ "Xinfeng Li", "Shenyu Dai", "Kelong Zheng", "Yue Xiao", "Gelei Deng", "Wei Dong", "Xiaofeng Wang" ], "categories": [ "cs.HC", "cs.AI", "cs.CR", "cs.SI" ], "topics": [ "agent-safety", "tool-use", "workflow-agent" ], "score": 13, "relevance": "high", "matched_queries": [ "agent-safety" ], "arxiv_id": "2602.21127", "source": "arxiv", "source_id": "arxiv:2602.21127", "pdf_url": "https://arxiv.org/pdf/2602.21127", "primary_query": "agent-safety" }, { "id": "2603.00131", "title": "Thought Virus: Viral Misalignment via Subliminal Prompting in Multi-Agent Systems", "url": "https://arxiv.org/abs/2603.00131", "published": "2026-02-23", "updated": "2026-02-23", "authors": [ "Moritz Weckbecker", "Jonas Müller", "Ben Hagag", "Michael Mulet" ], "categories": [ "cs.MA", "cs.AI" ], "topics": [ "agent-safety", "multi-agent", "tool-use" ], "score": 13, "relevance": "high", "matched_queries": [ "agent-safety" ], "arxiv_id": "2603.00131", "source": "arxiv", "source_id": "arxiv:2603.00131", "pdf_url": "https://arxiv.org/pdf/2603.00131", "primary_query": "agent-safety" }, { "id": "2602.19008", "title": "Capable but Unreliable: Canonical Path Deviation as a Causal Mechanism of Agent Failure in Long-Horizon Tasks", "url": "https://arxiv.org/abs/2602.19008", "published": "2026-02-22", "updated": "2026-02-22", "authors": [ "Wilson Y. Lee" ], "categories": [ "cs.CL", "cs.LG" ], "topics": [ "agent-evaluation", "computer-use", "planning", "tool-use" ], "score": 13, "relevance": "high", "matched_queries": [ "language-agent" ], "arxiv_id": "2602.19008", "source": "arxiv", "source_id": "arxiv:2602.19008", "pdf_url": "https://arxiv.org/pdf/2602.19008", "primary_query": "language-agent" }, { "id": "2602.14234", "title": "REDSearcher: A Scalable and Cost-Efficient Framework for Long-Horizon Search Agents", "url": "https://arxiv.org/abs/2602.14234", "published": "2026-02-15", "updated": "2026-02-15", "authors": [ "Zheng Chu", "Xiao Wang", "Jack Hong", "Huiming Fan", "Yuqi Huang", "Yue Yang", "Guohai Xu", "Chenxiao Zhao", "Cheng Xiang", "Shengchao Hu", "Dongdong Kuang", "Ming Liu", "Bing Qin", "Xing Yu" ], "categories": [ "cs.AI", "cs.CL" ], "topics": [ "agent-evaluation", "planning", "rag", "tool-use" ], "score": 13, "relevance": "high", "matched_queries": [ "function-calling" ], "arxiv_id": "2602.14234", "source": "arxiv", "source_id": "arxiv:2602.14234", "pdf_url": "https://arxiv.org/pdf/2602.14234", "primary_query": "function-calling" }, { "id": "2602.14281", "title": "MCPShield: A Security Cognition Layer for Adaptive Trust Calibration in Model Context Protocol Agents", "url": "https://arxiv.org/abs/2602.14281", "published": "2026-02-15", "updated": "2026-02-24", "authors": [ "Zhenhong Zhou", "Yuanhe Zhang", "Hongwei Cai", "Moayad Aloqaily", "Ouns Bouachir", "Linsey Pang", "Prakhar Mehrotra", "Kun Wang", "Qingsong Wen" ], "categories": [ "cs.CR", "cs.CL" ], "topics": [ "agent-safety", "computer-use", "reasoning", "tool-use" ], "score": 13, "relevance": "high", "matched_queries": [ "agent-safety" ], "arxiv_id": "2602.14281", "source": "arxiv", "source_id": "arxiv:2602.14281", "pdf_url": "https://arxiv.org/pdf/2602.14281", "primary_query": "agent-safety" }, { "id": "2602.13665", "title": "HyFunc: Accelerating LLM-based Function Calls for Agentic AI through Hybrid-Model Cascade and Dynamic Templating", "url": "https://arxiv.org/abs/2602.13665", "published": "2026-02-14", "updated": "2026-02-14", "authors": [ "Weibin Liao", "Jian-guang Lou", "Haoyi Xiong" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "computer-use" ], "score": 13, "relevance": "high", "matched_queries": [ "function-calling" ], "arxiv_id": "2602.13665", "source": "arxiv", "source_id": "arxiv:2602.13665", "pdf_url": "https://arxiv.org/pdf/2602.13665", "primary_query": "function-calling" }, { "id": "2602.08082", "title": "Spectral Guardrails for Agents in the Wild: Detecting Tool Use Hallucinations via Attention Topology", "url": "https://arxiv.org/abs/2602.08082", "published": "2026-02-08", "updated": "2026-02-08", "authors": [ "Valentin Noël" ], "categories": [ "cs.LG", "cs.AI", "eess.SP" ], "topics": [ "agent-evaluation", "agent-safety", "tool-use" ], "score": 13, "relevance": "high", "matched_queries": [ "agent-safety" ], "arxiv_id": "2602.08082", "source": "arxiv", "source_id": "arxiv:2602.08082", "pdf_url": "https://arxiv.org/pdf/2602.08082", "primary_query": "agent-safety" }, { "id": "2602.10133", "title": "AgentTrace: A Structured Logging Framework for Agent System Observability", "url": "https://arxiv.org/abs/2602.10133", "published": "2026-02-07", "updated": "2026-02-07", "authors": [ "Adam AlSayyad", "Kelvin Yuxiang Huang", "Richik Pal" ], "categories": [ "cs.SE", "cs.AI" ], "topics": [ "agent-evaluation", "agent-safety", "reasoning", "tool-use" ], "score": 13, "relevance": "high", "matched_queries": [ "agent-safety" ], "arxiv_id": "2602.10133", "source": "arxiv", "source_id": "arxiv:2602.10133", "pdf_url": "https://arxiv.org/pdf/2602.10133", "primary_query": "agent-safety" }, { "id": "2602.05386", "title": "Spider-Sense: Intrinsic Risk Sensing for Efficient Agent Defense with Hierarchical Adaptive Screening", "url": "https://arxiv.org/abs/2602.05386", "published": "2026-02-05", "updated": "2026-02-06", "authors": [ "Zhenxiong Yu", "Zhi Yang", "Zhiheng Jin", "Shuhe Wang", "Heng Zhang", "Yanlin Fei", "Lingfeng Zeng", "Fangqi Lou", "Shuo Zhang", "Tu Hu", "Jingping Liu", "Rongze Chen", "Xingyu Zhu", "Kunyi Wang", "Chaofa Yuan", "Xin Guo", "Zhaowei Liu", "Feipeng Zhang", "Jie Huang", "Huacan Wang", "Ronghao Chen", "Liwen Zhang" ], "categories": [ "cs.CR", "cs.AI" ], "topics": [ "agent-evaluation", "agent-safety", "reasoning", "tool-use" ], "score": 13, "relevance": "high", "matched_queries": [ "agent-safety" ], "arxiv_id": "2602.05386", "source": "arxiv", "source_id": "arxiv:2602.05386", "pdf_url": "https://arxiv.org/pdf/2602.05386", "primary_query": "agent-safety" }, { "id": "2602.03117", "title": "AgentDyn: Are Your Agent Security Defenses Deployable in Real-World Dynamic Environments?", "url": "https://arxiv.org/abs/2602.03117", "published": "2026-02-03", "updated": "2026-05-07", "authors": [ "Hao Li", "Ruoyao Wen", "Shanghao Shi", "Ning Zhang", "Yevgeniy Vorobeychik", "Chaowei Xiao" ], "categories": [ "cs.CR" ], "topics": [ "agent-evaluation", "agent-safety", "planning", "tool-use" ], "score": 13, "relevance": "high", "matched_queries": [ "agent-safety" ], "arxiv_id": "2602.03117", "source": "arxiv", "source_id": "arxiv:2602.03117", "pdf_url": "https://arxiv.org/pdf/2602.03117", "primary_query": "agent-safety" }, { "id": "2601.12988", "title": "PaperGuide: Making Small Language-Model Paper-Reading Agents More Efficient", "url": "https://arxiv.org/abs/2601.12988", "published": "2026-01-19", "updated": "2026-01-19", "authors": [ "Zijian Wang", "Tiancheng Huang", "Hanqi Li", "Da Ma", "Lu Chen", "Kai Yu" ], "categories": [ "cs.LG" ], "topics": [ "agent-evaluation", "computer-use", "planning", "reasoning", "tool-use" ], "score": 13, "relevance": "high", "matched_queries": [ "function-calling" ], "arxiv_id": "2601.12988", "source": "arxiv", "source_id": "arxiv:2601.12988", "pdf_url": "https://arxiv.org/pdf/2601.12988", "primary_query": "function-calling" }, { "id": "2601.06606", "title": "CEDAR: Context Engineering for Agentic Data Science", "url": "https://arxiv.org/abs/2601.06606", "published": "2026-01-10", "updated": "2026-04-22", "authors": [ "Rishiraj Saha Roy", "Chris Hinze", "Luzian Hahn", "Fabian Kuech" ], "categories": [ "cs.LG", "cs.AI" ], "topics": [ "planning", "workflow-agent" ], "score": 13, "relevance": "high", "matched_queries": [ "function-calling" ], "arxiv_id": "2601.06606", "source": "arxiv", "source_id": "arxiv:2601.06606", "pdf_url": "https://arxiv.org/pdf/2601.06606", "primary_query": "function-calling" }, { "id": "2512.23747", "title": "State-of-the-art Small Language Coder Model: Mify-Coder", "url": "https://arxiv.org/abs/2512.23747", "published": "2025-12-26", "updated": "2025-12-26", "authors": [ "Abhinav Parmar", "Abhisek Panigrahi", "Abhishek Kumar Dwivedi", "Abhishek Bhattacharya", "Adarsh Ramachandra", "Aditya Choudhary", "Aditya Garg", "Aditya Raj", "Alankrit Bhatt", "Alpesh Yadav", "Anant Vishnu", "Ananthu Pillai", "Ankush Kumar", "Aryan Patnaik", "Aswatha Narayanan S", "Avanish Raj Singh", "Bhavya Shree Gadda", "Brijesh Pankajbhai Kachhadiya", "Buggala Jahnavi", "Chidurala Nithin Krishna", "Chintan Shah", "Chunduru Akshaya", "Debarshi Banerjee", "Debrup Dey", "Deepa R.", "Deepika B G", "Faiz ur Rahman", "Gagan Gayari", "Gudhi Jagadeesh Kumar Naidu", "Gursimar Singh", "Harshal Tyagi", "Harshini K", "James Mani Vathalloor", "Jayarama Nettar", "Jayashree Gajjam", "Joe Walter Sugil George", "Kamalakara Sri Krishna Tadepalli", "Kamalkumar Rathinasamy", "Karan Chaurasia", "Karthikeyan S", "Kashish Arora", "Kaushal Desai", "Khushboo Buwade", "Kiran Manjrekar", "Malikireddy Venkata Sai Likhitha", "Manjunath A", "Mitali Mahavir Bedmutha", "Mohammed Rafee Tarafdar", "Nikhil Tiwari", "Nikitha K Gigi", "Pavan Ravikumar", "Pendyala Swarnanjali", "Piyush Anand", "Prakash Chandrasekar", "Prasanna Bhalchandra Gawade", "Prasanth Sivan", "Preeti Khurana", "Priyanshi Babbar", "Rajab Ali Mondal", "Rajesh Kumar Vissapragada", "Rajeshwari Ganesan", "Rajeswari Koppisetti", "Ramjee R.", "Ramkumar Thiruppathisamy", "Rani G. S.", "S Reka", "Samarth Gupta", "Sandeep Reddy Kothakota", "Sarathy K", "Sathyanarayana Sampath Kumar", "Saurabh Kumar", "Shashank Khasare", "Shenbaga Devi Venkatesh Kumar", "Shiva Rama Krishna Parvatham", "Shoeb Shaikh", "Shrishanmathi A", "Shubham Pathak", "Sree Samhita Koppaka", "Sreenivasa Raghavan K S", "Sreeram Venkatasubramanian", "Suprabha Desai Bojja", "Swetha R", "Syed Ahmed", "Chinmai Harshitha Thota", "Tushar Yadav", "Veeravelly Kusumitha", "V V S S Prasanth Patnaik", "Vidya Sri Sesetti", "Vijayakeerthi K", "Vikram Raj Bakshi", "Vinay K K", "Vinoth Kumar Loganathan", "Vipin Tiwari", "Vivek Kumar Shrivastav", "V Venkata Sri Datta Charan", "Wasim Akhtar Khan" ], "categories": [ "cs.SE", "cs.AI", "cs.CL" ], "topics": [ "agent-evaluation", "agent-safety", "coding-agent", "computer-use", "workflow-agent" ], "score": 13, "relevance": "high", "matched_queries": [ "function-calling" ], "arxiv_id": "2512.23747", "source": "arxiv", "source_id": "arxiv:2512.23747", "pdf_url": "https://arxiv.org/pdf/2512.23747", "primary_query": "function-calling" }, { "id": "2510.24645", "title": "FunReason-MT Technical Report: Advanced Data Synthesis Solution for Real-world Multi-Turn Tool-use", "url": "https://arxiv.org/abs/2510.24645", "published": "2025-10-28", "updated": "2025-11-16", "authors": [ "Zengzhuang Xu", "Bingguang Hao", "Zechuan Wang", "Yuntao Wen", "Xinyi Xu", "Yang Liu", "Long Chen", "Dong Wang", "Maolin Wang", "Tong Zhao", "Yicheng Chen", "Cunyin Peng", "Jinjie Gu", "Leilei Gan", "Xiangyu Zhao", "Chenyi Zhuang", "Shi Gu" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "computer-use", "multi-agent", "tool-use" ], "score": 13, "relevance": "high", "matched_queries": [ "function-calling" ], "arxiv_id": "2510.24645", "source": "arxiv", "source_id": "arxiv:2510.24645", "pdf_url": "https://arxiv.org/pdf/2510.24645", "primary_query": "function-calling" }, { "id": "2510.04206", "title": "AgentRL: Scaling Agentic Reinforcement Learning with a Multi-Turn, Multi-Task Framework", "url": "https://arxiv.org/abs/2510.04206", "published": "2025-10-05", "updated": "2025-10-05", "authors": [ "Hanchen Zhang", "Xiao Liu", "Bowen Lv", "Xueqiao Sun", "Bohao Jing", "Iat Long Iong", "Zhenyu Hou", "Zehan Qi", "Hanyu Lai", "Yifan Xu", "Rui Lu", "Hongning Wang", "Jie Tang", "Yuxiao Dong" ], "categories": [ "cs.AI" ], "topics": [ "rag", "tool-use" ], "score": 13, "relevance": "high", "matched_queries": [ "function-calling" ], "arxiv_id": "2510.04206", "source": "arxiv", "source_id": "arxiv:2510.04206", "pdf_url": "https://arxiv.org/pdf/2510.04206", "primary_query": "function-calling" }, { "id": "2509.13311", "title": "Towards General Agentic Intelligence via Environment Scaling", "url": "https://arxiv.org/abs/2509.13311", "published": "2025-09-16", "updated": "2025-09-16", "authors": [ "Runnan Fang", "Shihao Cai", "Baixuan Li", "Jialong Wu", "Guangyu Li", "Wenbiao Yin", "Xinyu Wang", "Xiaobin Wang", "Liangcai Su", "Zhen Zhang", "Shibin Wu", "Zhengwei Tao", "Yong Jiang", "Pengjun Xie", "Fei Huang", "Jingren Zhou" ], "categories": [ "cs.CL" ], "topics": [ "agent-evaluation", "tool-use" ], "score": 13, "relevance": "high", "matched_queries": [ "function-calling" ], "arxiv_id": "2509.13311", "source": "arxiv", "source_id": "arxiv:2509.13311", "pdf_url": "https://arxiv.org/pdf/2509.13311", "primary_query": "function-calling" }, { "id": "2509.02494", "title": "GridMind: LLMs-Powered Agents for Power System Analysis and Operations", "url": "https://arxiv.org/abs/2509.02494", "published": "2025-09-02", "updated": "2025-09-02", "authors": [ "Hongwei Jin", "Kibaek Kim", "Jonghwan Kwon" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "multi-agent", "workflow-agent" ], "score": 13, "relevance": "high", "matched_queries": [ "function-calling" ], "arxiv_id": "2509.02494", "source": "arxiv", "source_id": "arxiv:2509.02494", "pdf_url": "https://arxiv.org/pdf/2509.02494", "primary_query": "function-calling" }, { "id": "2508.17094", "title": "PowerChain: A Verifiable Agentic AI System for Automating Distribution Grid Analyses", "url": "https://arxiv.org/abs/2508.17094", "published": "2025-08-23", "updated": "2025-10-21", "authors": [ "Emmanuel O. Badmus", "Peng Sang", "Dimitrios Stamoulis", "Amritanshu Pandey" ], "categories": [ "cs.AI", "eess.SY" ], "topics": [ "planning", "rag", "reasoning", "tool-use", "workflow-agent" ], "score": 13, "relevance": "high", "matched_queries": [ "function-calling" ], "arxiv_id": "2508.17094", "source": "arxiv", "source_id": "arxiv:2508.17094", "pdf_url": "https://arxiv.org/pdf/2508.17094", "primary_query": "function-calling" }, { "id": "2508.12685", "title": "ToolACE-MT: Non-Autoregressive Generation for Agentic Multi-Turn Interaction", "url": "https://arxiv.org/abs/2508.12685", "published": "2025-08-18", "updated": "2026-02-13", "authors": [ "Xingshan Zeng", "Weiwen Liu", "Lingzhi Wang", "Liangyou Li", "Fei Mi", "Yasheng Wang", "Lifeng Shang", "Xin Jiang", "Qun Liu" ], "categories": [ "cs.CL", "cs.AI", "cs.LG" ], "topics": [ "tool-use", "world-model" ], "score": 13, "relevance": "high", "matched_queries": [ "function-calling" ], "arxiv_id": "2508.12685", "source": "arxiv", "source_id": "arxiv:2508.12685", "pdf_url": "https://arxiv.org/pdf/2508.12685", "primary_query": "function-calling" }, { "id": "2507.20666", "title": "MIMII-Agent: Leveraging LLMs with Function Calling for Relative Evaluation of Anomalous Sound Detection", "url": "https://arxiv.org/abs/2507.20666", "published": "2025-07-28", "updated": "2025-07-28", "authors": [ "Harsh Purohit", "Tomoya Nishida", "Kota Dohi", "Takashi Endo", "Yohei Kawaguchi" ], "categories": [ "eess.AS", "cs.AI", "cs.LG", "cs.SD" ], "topics": [ "agent-evaluation", "rag", "tool-use" ], "score": 13, "relevance": "high", "matched_queries": [ "function-calling" ], "arxiv_id": "2507.20666", "source": "arxiv", "source_id": "arxiv:2507.20666", "pdf_url": "https://arxiv.org/pdf/2507.20666", "primary_query": "function-calling" }, { "id": "2507.20395", "title": "MazeEval: A Benchmark for Testing Sequential Decision-Making in Language Models", "url": "https://arxiv.org/abs/2507.20395", "published": "2025-07-27", "updated": "2025-07-27", "authors": [ "Hafsteinn Einarsson" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "computer-use", "embodied-agent", "reasoning" ], "score": 13, "relevance": "high", "matched_queries": [ "function-calling" ], "arxiv_id": "2507.20395", "source": "arxiv", "source_id": "arxiv:2507.20395", "pdf_url": "https://arxiv.org/pdf/2507.20395", "primary_query": "function-calling" }, { "id": "2607.06503", "title": "Doomed from the Start: Early Abort of LLM Agent Episodes via a Recall-Controlled Probe Cascade", "url": "https://arxiv.org/abs/2607.06503", "published": "2026-07-07", "updated": "2026-07-07", "authors": [ "Kai Ruan", "Zihe Huang", "Ziqi Zhou", "Qianshan Wei", "Xuan Wang", "Hao Sun" ], "categories": [ "cs.AI" ], "topics": [ "agent-safety", "tool-use" ], "score": 12, "relevance": "medium", "matched_queries": [ "llm-agent" ], "arxiv_id": "2607.06503", "source": "arxiv", "source_id": "arxiv:2607.06503", "pdf_url": "https://arxiv.org/pdf/2607.06503", "primary_query": "llm-agent" }, { "id": "2607.04729", "title": "RustMizan: A Compilable, Contamination-Aware Benchmarking Framework for Rust Vulnerabilities", "url": "https://arxiv.org/abs/2607.04729", "published": "2026-07-06", "updated": "2026-07-06", "authors": [ "Tarek Elsayed", "Shiping Yang", "Eunsong Koh", "Sanika Goyal", "Vincent Huang", "Paul Ngo", "Nathan Young", "Mohammad Omidvar Tehrani", "Alvyn Kang", "Arnell Kang", "Zeyu Chen", "Angélica Moreira", "Xuan Feng", "Angel X. Chang", "Nick Sumner", "Steven Y. Ko" ], "categories": [ "cs.CR", "cs.AI", "cs.SE" ], "topics": [ "agent-evaluation", "agent-safety" ], "score": 12, "relevance": "medium", "matched_queries": [ "llm-agent" ], "arxiv_id": "2607.04729", "source": "arxiv", "source_id": "arxiv:2607.04729", "pdf_url": "https://arxiv.org/pdf/2607.04729", "primary_query": "llm-agent" }, { "id": "2607.05577", "title": "Narrative World Model: Narratology-Grounded Writer Memory for Long-Form Fiction", "url": "https://arxiv.org/abs/2607.05577", "published": "2026-07-06", "updated": "2026-07-06", "authors": [ "Mohammad Saifullah", "Thomas Kornmaier", "Taaha Kazi", "Vasu Sharma", "Aditya Sanjiv Kanade", "Aanand Kumar Yadav" ], "categories": [ "cs.AI", "cs.CL", "cs.IR" ], "topics": [ "agent-evaluation", "memory", "rag", "tool-use", "world-model" ], "score": 12, "relevance": "medium", "matched_queries": [ "agent-memory" ], "arxiv_id": "2607.05577", "source": "arxiv", "source_id": "arxiv:2607.05577", "pdf_url": "https://arxiv.org/pdf/2607.05577", "primary_query": "agent-memory" }, { "id": "2607.05477", "title": "Decision Protocols in Multi-Agent Large Language Model Conversations", "url": "https://arxiv.org/abs/2607.05477", "published": "2026-07-06", "updated": "2026-07-06", "authors": [ "Lars Benedikt Kaesberg" ], "categories": [ "cs.MA", "cs.AI", "cs.CL" ], "topics": [ "agent-evaluation", "multi-agent" ], "score": 12, "relevance": "medium", "matched_queries": [ "multi-agent-llm" ], "arxiv_id": "2607.05477", "source": "arxiv", "source_id": "arxiv:2607.05477", "pdf_url": "https://arxiv.org/pdf/2607.05477", "primary_query": "multi-agent-llm" }, { "id": "2607.04235", "title": "Spinning Straw into Gold: Relabeling LLM Agent Trajectories in Hindsight for Successful Demonstrations", "url": "https://arxiv.org/abs/2607.04235", "published": "2026-07-05", "updated": "2026-07-05", "authors": [ "Zichao Li", "Gang Wu", "Zichao Wang", "Ruiyi Zhang", "Wanrong Zhu", "Ryan A. Rossi", "Vlad I Morariu", "Jihyung Kil" ], "categories": [ "cs.CL" ], "topics": [ "planning", "tool-use" ], "score": 12, "relevance": "medium", "matched_queries": [ "llm-agent" ], "arxiv_id": "2607.04235", "source": "arxiv", "source_id": "arxiv:2607.04235", "pdf_url": "https://arxiv.org/pdf/2607.04235", "primary_query": "llm-agent" }, { "id": "2607.01764", "title": "Mastermind: Strategy-grounded Learning for Repository-Scale Vulnerability Reproduction", "url": "https://arxiv.org/abs/2607.01764", "published": "2026-07-02", "updated": "2026-07-02", "authors": [ "Mingzhe Du", "Luu Anh Tuan", "Tianyi Wu", "Renyang Liu", "Zhijiang Guo", "Dong Huang", "See-Kiong Ng" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "coding-agent", "computer-use", "planning", "tool-use" ], "score": 12, "relevance": "medium", "matched_queries": [ "llm-agent" ], "arxiv_id": "2607.01764", "source": "arxiv", "source_id": "arxiv:2607.01764", "pdf_url": "https://arxiv.org/pdf/2607.01764", "primary_query": "llm-agent" }, { "id": "2607.02116", "title": "ContextNest: Verifiable Context Governance for Autonomous AI Agent", "url": "https://arxiv.org/abs/2607.02116", "published": "2026-07-02", "updated": "2026-07-06", "authors": [ "Misha Sulpovar", "Benn R. Konsynski", "Qaish Kanchwala", "Gabe Goodhart" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "rag", "tool-use" ], "score": 12, "relevance": "medium", "matched_queries": [ "ai-agent", "rag-agent" ], "arxiv_id": "2607.02116", "source": "arxiv", "source_id": "arxiv:2607.02116", "pdf_url": "https://arxiv.org/pdf/2607.02116", "primary_query": "ai-agent" }, { "id": "2607.02389", "title": "Steerability via constraints: a substrate for scalable oversight of coding agents", "url": "https://arxiv.org/abs/2607.02389", "published": "2026-07-02", "updated": "2026-07-02", "authors": [ "Thomas Winninger" ], "categories": [ "cs.AI", "cs.CR", "cs.SE" ], "topics": [ "agent-safety", "coding-agent", "tool-use" ], "score": 12, "relevance": "medium", "matched_queries": [ "coding-agent" ], "arxiv_id": "2607.02389", "source": "arxiv", "source_id": "arxiv:2607.02389", "pdf_url": "https://arxiv.org/pdf/2607.02389", "primary_query": "coding-agent" }, { "id": "2607.02802", "title": "Seduced by the Narrative: Assessing Rule Adherence in Semi-Open Textual Sandboxes", "url": "https://arxiv.org/abs/2607.02802", "published": "2026-07-02", "updated": "2026-07-02", "authors": [ "Weiying Chen", "Junlong Shen", "Zhanyuan Guo", "Xiaoou Zhou" ], "categories": [ "cs.CL", "cs.AI" ], "topics": [ "agent-evaluation", "multi-agent", "reasoning", "tool-use" ], "score": 12, "relevance": "medium", "matched_queries": [ "multi-agent-llm" ], "arxiv_id": "2607.02802", "source": "arxiv", "source_id": "arxiv:2607.02802", "pdf_url": "https://arxiv.org/pdf/2607.02802", "primary_query": "multi-agent-llm" }, { "id": "2607.01846", "title": "CLAP: Closed-Loop Training, Evaluation, and Release Control for Domain Agent Post-training", "url": "https://arxiv.org/abs/2607.01846", "published": "2026-07-02", "updated": "2026-07-02", "authors": [ "Fangfei Li", "Chenyang Zhao", "Long Wang", "Feng Tian", "Zhiyue Zheng", "Lv Guo" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "agent-safety", "rag", "tool-use" ], "score": 12, "relevance": "medium", "matched_queries": [ "rag-agent" ], "arxiv_id": "2607.01846", "source": "arxiv", "source_id": "arxiv:2607.01846", "pdf_url": "https://arxiv.org/pdf/2607.01846", "primary_query": "rag-agent" }, { "id": "2607.01136", "title": "Skills Are Not Islands: Measuring Dependency and Risk in Agent Skill Supply Chains", "url": "https://arxiv.org/abs/2607.01136", "published": "2026-07-01", "updated": "2026-07-01", "authors": [ "Changguo Jia", "Tianqi Zhao", "Runzhi He", "Minghui Zhou" ], "categories": [ "cs.SE", "cs.AI" ], "topics": [ "agent-evaluation", "agent-safety", "tool-use", "workflow-agent" ], "score": 12, "relevance": "medium", "matched_queries": [ "llm-agent" ], "arxiv_id": "2607.01136", "source": "arxiv", "source_id": "arxiv:2607.01136", "pdf_url": "https://arxiv.org/pdf/2607.01136", "primary_query": "llm-agent" }, { "id": "2607.00339", "title": "TRACE: State-Aware Query Processing over Temporal Evidence Graphs for Conversational Data", "url": "https://arxiv.org/abs/2607.00339", "published": "2026-07-01", "updated": "2026-07-01", "authors": [ "Maolin Wang", "Yu Wang", "Zichun Liu", "Baiyuan Qiu", "Chenbin Zhang", "Jiguang Shen", "Haoran Yang", "Hao Miao" ], "categories": [ "cs.CL" ], "topics": [ "agent-evaluation", "computer-use", "memory", "planning", "rag", "reasoning" ], "score": 12, "relevance": "medium", "matched_queries": [ "ai-agent" ], "arxiv_id": "2607.00339", "source": "arxiv", "source_id": "arxiv:2607.00339", "pdf_url": "https://arxiv.org/pdf/2607.00339", "primary_query": "ai-agent" }, { "id": "2607.02605", "title": "A Survey of LLM-Driven Penetration Testing: Taxonomy, Co-Evolution, and Open Challenges", "url": "https://arxiv.org/abs/2607.02605", "published": "2026-07-01", "updated": "2026-07-01", "authors": [ "Zheyuan He", "Jiaxun Dong", "Zihao Li", "Ting Chen", "Gelei Deng", "Feng Luo", "Jinkun Ji", "Yuanlong Cao", "Xiapu Luo" ], "categories": [ "cs.SE" ], "topics": [ "agent-evaluation", "agent-safety", "reasoning", "tool-use", "workflow-agent" ], "score": 12, "relevance": "medium", "matched_queries": [ "agent-evaluation" ], "arxiv_id": "2607.02605", "source": "arxiv", "source_id": "arxiv:2607.02605", "pdf_url": "https://arxiv.org/pdf/2607.02605", "primary_query": "agent-evaluation" }, { "id": "2606.31518", "title": "Design and Implementation of Agentic Orchestrations and Orchestration of Agents", "url": "https://arxiv.org/abs/2606.31518", "published": "2026-06-30", "updated": "2026-06-30", "authors": [ "Stefanie Rinderle-Ma", "Juergen Mangler", "Johannes Loebbecke", "Dominik Voigt", "Nataliia Klievtsova", "Matthias Ehrendorfer" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation" ], "score": 12, "relevance": "medium", "matched_queries": [ "ai-agent" ], "arxiv_id": "2606.31518", "source": "arxiv", "source_id": "arxiv:2606.31518", "pdf_url": "https://arxiv.org/pdf/2606.31518", "primary_query": "ai-agent" }, { "id": "2606.31023", "title": "Certified Speculative Execution for Untrusted AI Agents", "url": "https://arxiv.org/abs/2606.31023", "published": "2026-06-30", "updated": "2026-06-30", "authors": [ "Chenyu Zhou", "Qiliang Jiang", "Shuning Wu", "Xu Zhou" ], "categories": [ "cs.CR", "cs.LG" ], "topics": [ "agent-safety", "reasoning" ], "score": 12, "relevance": "medium", "matched_queries": [ "ai-agent" ], "arxiv_id": "2606.31023", "source": "arxiv", "source_id": "arxiv:2606.31023", "pdf_url": "https://arxiv.org/pdf/2606.31023", "primary_query": "ai-agent" }, { "id": "2606.31613", "title": "Robust Autonomous UAV Landing on Maritime Platforms via Multimodal Agentic AI and Active Wave Compensation", "url": "https://arxiv.org/abs/2606.31613", "published": "2026-06-30", "updated": "2026-06-30", "authors": [ "Francisco S. Neves", "Pedro N. Pereira", "Raul D. S. G. Campilho", "Andry M. Pinto" ], "categories": [ "cs.CV", "cs.RO" ], "topics": [ "agent-evaluation", "agent-safety", "planning", "world-model" ], "score": 12, "relevance": "medium", "matched_queries": [ "agentic-ai" ], "arxiv_id": "2606.31613", "source": "arxiv", "source_id": "arxiv:2606.31613", "pdf_url": "https://arxiv.org/pdf/2606.31613", "primary_query": "agentic-ai" }, { "id": "2606.31392", "title": "ReGRPO: Reflection-Augmented Policy Optimization for Tool-Using Agents", "url": "https://arxiv.org/abs/2606.31392", "published": "2026-06-30", "updated": "2026-06-30", "authors": [ "Binjie Zhang", "Mike Zheng Shou" ], "categories": [ "cs.AI" ], "topics": [ "computer-use", "planning", "rag", "reasoning", "tool-use" ], "score": 12, "relevance": "medium", "matched_queries": [ "tool-use" ], "arxiv_id": "2606.31392", "source": "arxiv", "source_id": "arxiv:2606.31392", "pdf_url": "https://arxiv.org/pdf/2606.31392", "primary_query": "tool-use" }, { "id": "2606.31461", "title": "CSTrader: A Testbed for Language-Grounded Trading in a Community-Driven Virtual Asset Market", "url": "https://arxiv.org/abs/2606.31461", "published": "2026-06-30", "updated": "2026-06-30", "authors": [ "Yao Shi", "Kingfung Luo", "Nan Tang", "Yuyu Luo" ], "categories": [ "cs.AI", "cs.CE" ], "topics": [ "agent-evaluation", "agent-safety", "multi-agent", "tool-use" ], "score": 12, "relevance": "medium", "matched_queries": [ "multi-agent-llm" ], "arxiv_id": "2606.31461", "source": "arxiv", "source_id": "arxiv:2606.31461", "pdf_url": "https://arxiv.org/pdf/2606.31461", "primary_query": "multi-agent-llm" }, { "id": "2606.31039", "title": "Truth or Sophistry? LoFa: A Benchmark for LLM Robustness Against Logical Fallacies", "url": "https://arxiv.org/abs/2606.31039", "published": "2026-06-30", "updated": "2026-06-30", "authors": [ "Xudong Shen", "Li Yuan", "Ye Chen", "Xin Wu", "Yi Cai", "Zhiyong Wu" ], "categories": [ "cs.CL" ], "topics": [ "agent-evaluation", "computer-use", "multi-agent" ], "score": 12, "relevance": "medium", "matched_queries": [ "multi-agent-llm" ], "arxiv_id": "2606.31039", "source": "arxiv", "source_id": "arxiv:2606.31039", "pdf_url": "https://arxiv.org/pdf/2606.31039", "primary_query": "multi-agent-llm" }, { "id": "2606.29722", "title": "Attraction, Not Adaptation: How AI Agent Communities Develop Distinct Linguistic Identities", "url": "https://arxiv.org/abs/2606.29722", "published": "2026-06-29", "updated": "2026-06-29", "authors": [ "Daming Li", "Simeng Han", "Can Meng", "Wanyu Lei", "Jialu Zhang" ], "categories": [ "cs.SI" ], "topics": [ "computer-use", "multi-agent", "tool-use" ], "score": 12, "relevance": "medium", "matched_queries": [ "ai-agent" ], "arxiv_id": "2606.29722", "source": "arxiv", "source_id": "arxiv:2606.29722", "pdf_url": "https://arxiv.org/pdf/2606.29722", "primary_query": "ai-agent" }, { "id": "2606.30531", "title": "Entity Binding Failures in Tool-Augmented Agents", "url": "https://arxiv.org/abs/2606.30531", "published": "2026-06-29", "updated": "2026-06-29", "authors": [ "Rahul Suresh Babu", "Shashank Indukuri" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "agent-safety", "computer-use", "tool-use", "workflow-agent" ], "score": 12, "relevance": "medium", "matched_queries": [ "tool-use" ], "arxiv_id": "2606.30531", "source": "arxiv", "source_id": "arxiv:2606.30531", "pdf_url": "https://arxiv.org/pdf/2606.30531", "primary_query": "tool-use" }, { "id": "2606.29871", "title": "AI Training Manager: Bounded Closed-Loop Control of Adaptive Training Recipes", "url": "https://arxiv.org/abs/2606.29871", "published": "2026-06-29", "updated": "2026-06-29", "authors": [ "Anjali Rao", "Nikhil Kamalkumar Advani" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "coding-agent", "embodied-agent", "memory", "tool-use" ], "score": 12, "relevance": "medium", "matched_queries": [ "coding-agent" ], "arxiv_id": "2606.29871", "source": "arxiv", "source_id": "arxiv:2606.29871", "pdf_url": "https://arxiv.org/pdf/2606.29871", "primary_query": "coding-agent" }, { "id": "2606.30479", "title": "COHORT: Collaborative Orchestration for Hardening via Offensive Replay on Emulated Topologies", "url": "https://arxiv.org/abs/2606.30479", "published": "2026-06-29", "updated": "2026-06-29", "authors": [ "Chen Frydman", "Aviram Zilberman", "Rubin Krief", "Abed Showgan", "Andres Murillo", "Sekiya Motoyoshi", "Asaf Shabtai", "Yuval Elovici", "Rami Puzis" ], "categories": [ "cs.NI", "cs.AI", "cs.CR", "cs.MA" ], "topics": [ "agent-evaluation", "multi-agent", "tool-use", "workflow-agent", "world-model" ], "score": 12, "relevance": "medium", "matched_queries": [ "multi-agent-llm" ], "arxiv_id": "2606.30479", "source": "arxiv", "source_id": "arxiv:2606.30479", "pdf_url": "https://arxiv.org/pdf/2606.30479", "primary_query": "multi-agent-llm" }, { "id": "2606.29033", "title": "Human-in-the-Loop Nugget Annotation for Accountable LLM-as-a-Judge Evaluations", "url": "https://arxiv.org/abs/2606.29033", "published": "2026-06-27", "updated": "2026-06-27", "authors": [ "Laura Dietz" ], "categories": [ "cs.IR" ], "topics": [ "agent-evaluation", "tool-use", "workflow-agent" ], "score": 12, "relevance": "medium", "matched_queries": [ "ai-agent" ], "arxiv_id": "2606.29033", "source": "arxiv", "source_id": "arxiv:2606.29033", "pdf_url": "https://arxiv.org/pdf/2606.29033", "primary_query": "ai-agent" }, { "id": "2606.27936", "title": "Agentic AI-Powered Re-Identification: An Emerging, Scalable Threat to Mobility Microdata Privacy", "url": "https://arxiv.org/abs/2606.27936", "published": "2026-06-26", "updated": "2026-06-26", "authors": [ "Oscar Thees", "Roman Müller", "Matthias Templ" ], "categories": [ "cs.CR", "cs.AI", "stat.AP" ], "topics": [ "agent-evaluation", "agent-safety" ], "score": 12, "relevance": "medium", "matched_queries": [ "agentic-ai" ], "arxiv_id": "2606.27936", "source": "arxiv", "source_id": "arxiv:2606.27936", "pdf_url": "https://arxiv.org/pdf/2606.27936", "primary_query": "agentic-ai" }, { "id": "2606.26790", "title": "OPID: On-Policy Skill Distillation for Agentic Reinforcement Learning", "url": "https://arxiv.org/abs/2606.26790", "published": "2026-06-25", "updated": "2026-06-25", "authors": [ "Shuo Yang", "Jinyang Wu", "Zhengxi Lu", "Yuhao Shen", "Fan Zhang", "Lang Feng", "Shuai Zhang", "Haoran Luo", "Zheng Lian", "Zhengqi Wen", "Jianhua Tao" ], "categories": [ "cs.CL" ], "topics": [ "computer-use", "tool-use", "workflow-agent" ], "score": 12, "relevance": "medium", "matched_queries": [ "language-agent" ], "arxiv_id": "2606.26790", "source": "arxiv", "source_id": "arxiv:2606.26790", "pdf_url": "https://arxiv.org/pdf/2606.26790", "primary_query": "language-agent" }, { "id": "2606.27443", "title": "When Does Personality Composition Matter for Multi-Agent LLM Teams?", "url": "https://arxiv.org/abs/2606.27443", "published": "2026-06-25", "updated": "2026-06-25", "authors": [ "Aryan Keluskar", "Amrita Bhattacharjee", "Huan Liu" ], "categories": [ "cs.AI", "cs.CL" ], "topics": [ "coding-agent", "multi-agent" ], "score": 12, "relevance": "medium", "matched_queries": [ "multi-agent-llm" ], "arxiv_id": "2606.27443", "source": "arxiv", "source_id": "arxiv:2606.27443", "pdf_url": "https://arxiv.org/pdf/2606.27443", "primary_query": "multi-agent-llm" }, { "id": "2606.27409", "title": "Delayed Verification Destabilizes Multi-Agent LLM Belief: Instability Thresholds and Optimal Corrector Placement", "url": "https://arxiv.org/abs/2606.27409", "published": "2026-06-25", "updated": "2026-06-25", "authors": [ "Igor Itkin" ], "categories": [ "cs.MA", "cs.CL", "cs.LG", "eess.SY" ], "topics": [ "multi-agent", "reasoning" ], "score": 12, "relevance": "medium", "matched_queries": [ "multi-agent-llm" ], "arxiv_id": "2606.27409", "source": "arxiv", "source_id": "arxiv:2606.27409", "pdf_url": "https://arxiv.org/pdf/2606.27409", "primary_query": "multi-agent-llm" }, { "id": "2606.26289", "title": "Augmentation with Dilution: A Large-Scale Empirical Study of Human Contributor Ecosystems After AI Coding Agent Adoption", "url": "https://arxiv.org/abs/2606.26289", "published": "2026-06-24", "updated": "2026-06-24", "authors": [ "Weixing Zhang", "Bowen Jiang", "Anne Koziolek" ], "categories": [ "cs.SE" ], "topics": [ "coding-agent", "tool-use" ], "score": 12, "relevance": "medium", "matched_queries": [ "ai-agent", "coding-agent" ], "arxiv_id": "2606.26289", "source": "arxiv", "source_id": "arxiv:2606.26289", "pdf_url": "https://arxiv.org/pdf/2606.26289", "primary_query": "ai-agent" }, { "id": "2606.25996", "title": "Autodata: An agentic data scientist to create high quality synthetic data", "url": "https://arxiv.org/abs/2606.25996", "published": "2026-06-24", "updated": "2026-07-04", "authors": [ "Ilia Kulikov", "Chenxi Whitehouse", "Tianhao Wu", "Yixin Nie", "Swarnadeep Saha", "Eryk Helenowski", "Weizhe Yuan", "Olga Golovneva", "Jack Lanchantin", "Yoram Bachrach", "Jakob Foerster", "Xian Li", "Han Fang", "Sainbayar Sukhbaatar", "Jason Weston" ], "categories": [ "cs.AI", "cs.CL", "cs.LG" ], "topics": [ "agent-evaluation", "reasoning" ], "score": 12, "relevance": "medium", "matched_queries": [ "ai-agent" ], "arxiv_id": "2606.25996", "source": "arxiv", "source_id": "arxiv:2606.25996", "pdf_url": "https://arxiv.org/pdf/2606.25996", "primary_query": "ai-agent" }, { "id": "2606.25836", "title": "AI Snitches Get Glitches: Towards Evading Agentic Surveillance", "url": "https://arxiv.org/abs/2606.25836", "published": "2026-06-24", "updated": "2026-06-26", "authors": [ "Hyejun Jeong", "Dzung Pham", "Amir Houmansadr", "Eugene Bagdasarian" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "agent-safety", "tool-use" ], "score": 12, "relevance": "medium", "matched_queries": [ "ai-agent" ], "arxiv_id": "2606.25836", "source": "arxiv", "source_id": "arxiv:2606.25836", "pdf_url": "https://arxiv.org/pdf/2606.25836", "primary_query": "ai-agent" }, { "id": "2606.25332", "title": "Decoupling Reconnaissance and Exploitation: Measuring the Capability Boundaries of LLM-Based Web Penetration Testing", "url": "https://arxiv.org/abs/2606.25332", "published": "2026-06-24", "updated": "2026-06-24", "authors": [ "Liwei Yu", "Shuo Li", "Ming Zhou", "Ge Chu", "Yan Guo" ], "categories": [ "cs.CR", "cs.AI" ], "topics": [ "agent-evaluation", "agent-safety", "multi-agent", "tool-use" ], "score": 12, "relevance": "medium", "matched_queries": [ "multi-agent-llm" ], "arxiv_id": "2606.25332", "source": "arxiv", "source_id": "arxiv:2606.25332", "pdf_url": "https://arxiv.org/pdf/2606.25332", "primary_query": "multi-agent-llm" }, { "id": "2606.24530", "title": "NatureBench: Can Coding Agents Match the Published SOTA of Nature-Family Papers?", "url": "https://arxiv.org/abs/2606.24530", "published": "2026-06-23", "updated": "2026-07-06", "authors": [ "Yuru Wang", "Lejun Cheng", "Yuxin Zuo", "Sihang Zeng", "Bingxiang He", "Che Jiang", "Junlin Yang", "Yuchong Wang", "Kaikai Zhao", "Weifeng Huang", "Kai Tian", "Zhenzhao Yuan", "Jincheng Zhong", "Weizhi Wang", "Ning Ding", "Bowen Zhou", "Kaiyan Zhang" ], "categories": [ "cs.CL" ], "topics": [ "agent-evaluation", "coding-agent", "rag" ], "score": 12, "relevance": "medium", "matched_queries": [ "coding-agent" ], "arxiv_id": "2606.24530", "source": "arxiv", "source_id": "arxiv:2606.24530", "pdf_url": "https://arxiv.org/pdf/2606.24530", "primary_query": "coding-agent" }, { "id": "2606.24370", "title": "When Helpfulness Overrides Causal Caution: Context-Dependent Suppression and Recovery in LLMs", "url": "https://arxiv.org/abs/2606.24370", "published": "2026-06-23", "updated": "2026-06-23", "authors": [ "Hiroshi Okumura" ], "categories": [ "cs.AI", "cs.CY" ], "topics": [ "agent-evaluation", "multi-agent", "planning", "reasoning" ], "score": 12, "relevance": "medium", "matched_queries": [ "multi-agent-llm" ], "arxiv_id": "2606.24370", "source": "arxiv", "source_id": "arxiv:2606.24370", "pdf_url": "https://arxiv.org/pdf/2606.24370", "primary_query": "multi-agent-llm" }, { "id": "2606.22916", "title": "Intent-Governed Tool Authorization for AI Agents", "url": "https://arxiv.org/abs/2606.22916", "published": "2026-06-22", "updated": "2026-06-22", "authors": [ "Genliang Zhu", "Chu Wang" ], "categories": [ "cs.AI" ], "topics": [ "tool-use", "workflow-agent" ], "score": 12, "relevance": "medium", "matched_queries": [ "ai-agent", "tool-use" ], "arxiv_id": "2606.22916", "source": "arxiv", "source_id": "arxiv:2606.22916", "pdf_url": "https://arxiv.org/pdf/2606.22916", "primary_query": "ai-agent" }, { "id": "2606.23654", "title": "EnterpriseClawBench: Benchmarking Agents from Real Workplace Sessions", "url": "https://arxiv.org/abs/2606.23654", "published": "2026-06-22", "updated": "2026-06-22", "authors": [ "Jincheng Zhong", "Weizhi Wang", "Che Jiang", "Kai Tian", "Zhenzhao Yuan", "Junlin Yang", "Dianqiao Lei", "Kaiyan Zhang" ], "categories": [ "cs.CL", "cs.SE" ], "topics": [ "agent-evaluation", "tool-use", "workflow-agent" ], "score": 12, "relevance": "medium", "matched_queries": [ "agent-evaluation" ], "arxiv_id": "2606.23654", "source": "arxiv", "source_id": "arxiv:2606.23654", "pdf_url": "https://arxiv.org/pdf/2606.23654", "primary_query": "agent-evaluation" }, { "id": "2606.23277", "title": "GIF: Locally Sound Geometric Information Flow Control for LLMs", "url": "https://arxiv.org/abs/2606.23277", "published": "2026-06-22", "updated": "2026-06-22", "authors": [ "Adam Storek", "Nikolaus Holzer", "Zhuo Zhang", "Suman Jana" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "agent-safety", "reasoning", "tool-use" ], "score": 12, "relevance": "medium", "matched_queries": [ "tool-use" ], "arxiv_id": "2606.23277", "source": "arxiv", "source_id": "arxiv:2606.23277", "pdf_url": "https://arxiv.org/pdf/2606.23277", "primary_query": "tool-use" }, { "id": "2606.23112", "title": "Self-Evolution for Multi-Turn Tool-Calling Agents via Divergence-Point Preference Learning", "url": "https://arxiv.org/abs/2606.23112", "published": "2026-06-22", "updated": "2026-06-22", "authors": [ "Jiaqiang Tang" ], "categories": [ "cs.LG", "cs.AI", "cs.CL" ], "topics": [ "agent-evaluation", "agent-safety", "planning", "rag", "reasoning", "tool-use" ], "score": 12, "relevance": "medium", "matched_queries": [ "tool-use" ], "arxiv_id": "2606.23112", "source": "arxiv", "source_id": "arxiv:2606.23112", "pdf_url": "https://arxiv.org/pdf/2606.23112", "primary_query": "tool-use" }, { "id": "2606.23797", "title": "From Task-Guided Conversational Graphs to Goal-Oriented Dialogue Runtimes", "url": "https://arxiv.org/abs/2606.23797", "published": "2026-06-22", "updated": "2026-06-22", "authors": [ "Mariano Garralda-Barrio" ], "categories": [ "cs.SE", "cs.AI", "cs.CL", "cs.MA" ], "topics": [ "agent-evaluation", "computer-use", "multi-agent", "tool-use", "workflow-agent" ], "score": 12, "relevance": "medium", "matched_queries": [ "multi-agent-llm" ], "arxiv_id": "2606.23797", "source": "arxiv", "source_id": "arxiv:2606.23797", "pdf_url": "https://arxiv.org/pdf/2606.23797", "primary_query": "multi-agent-llm" }, { "id": "2606.22329", "title": "BabelJudge: Measuring LLM-as-a-Judge Reliability Across Languages and Agent Trajectories", "url": "https://arxiv.org/abs/2606.22329", "published": "2026-06-21", "updated": "2026-06-21", "authors": [ "Shreyas KC" ], "categories": [ "cs.CL", "cs.AI" ], "topics": [ "agent-evaluation", "tool-use" ], "score": 12, "relevance": "medium", "matched_queries": [ "agent-evaluation" ], "arxiv_id": "2606.22329", "source": "arxiv", "source_id": "arxiv:2606.22329", "pdf_url": "https://arxiv.org/pdf/2606.22329", "primary_query": "agent-evaluation" }, { "id": "2606.22504", "title": "Lingering Authority: Revocable Resource-and-Effect Capabilities for Coding Agents", "url": "https://arxiv.org/abs/2606.22504", "published": "2026-06-21", "updated": "2026-06-21", "authors": [ "Igor Santos-Grueiro" ], "categories": [ "cs.CR", "cs.AI", "cs.SE" ], "topics": [ "agent-evaluation", "coding-agent", "planning", "tool-use" ], "score": 12, "relevance": "medium", "matched_queries": [ "coding-agent" ], "arxiv_id": "2606.22504", "source": "arxiv", "source_id": "arxiv:2606.22504", "pdf_url": "https://arxiv.org/pdf/2606.22504", "primary_query": "coding-agent" }, { "id": "2606.22337", "title": "Theorist Toolbox: Tools for Agent Based LLM-assisted economic theory Research", "url": "https://arxiv.org/abs/2606.22337", "published": "2026-06-21", "updated": "2026-06-23", "authors": [ "Moran Koren" ], "categories": [ "econ.TH", "cs.GT", "econ.GN" ], "topics": [ "computer-use", "multi-agent", "reasoning", "tool-use" ], "score": 12, "relevance": "medium", "matched_queries": [ "multi-agent-llm" ], "arxiv_id": "2606.22337", "source": "arxiv", "source_id": "arxiv:2606.22337", "pdf_url": "https://arxiv.org/pdf/2606.22337", "primary_query": "multi-agent-llm" }, { "id": "2606.22692", "title": "VISTA Architect: A graph database-oriented health AI system demonstrated in multidisciplinary tumor boards", "url": "https://arxiv.org/abs/2606.22692", "published": "2026-06-21", "updated": "2026-06-21", "authors": [ "Tuomo Kiiskinen", "Jason Fries", "Philip Adamson", "David Wu", "Timothy John Ellis-Caleo", "Aaron Fanous", "Balasubramanian Narasimhan", "Joel Neal", "Sylvia Plevritis", "Manuel A. Rivas" ], "categories": [ "cs.AI", "cs.CL", "cs.DB", "cs.IR" ], "topics": [ "agent-evaluation", "computer-use", "rag", "tool-use" ], "score": 12, "relevance": "medium", "matched_queries": [ "rag-agent" ], "arxiv_id": "2606.22692", "source": "arxiv", "source_id": "arxiv:2606.22692", "pdf_url": "https://arxiv.org/pdf/2606.22692", "primary_query": "rag-agent" }, { "id": "2606.21955", "title": "From RAN Control to Agentic Intelligence: Architecture and Vision for Energy Efficient AI-RAN", "url": "https://arxiv.org/abs/2606.21955", "published": "2026-06-20", "updated": "2026-06-20", "authors": [ "Sabrine Aroua", "Alexis I. Aravanis", "Ilias Chatzistefanidis", "Hamza Abbar", "Anh-Khoa Dang", "Anastasios Giovanidis", "Salah-Eddine El Ayoubi", "Stephane Senecal", "Martha Vlachou Konchylaki", "Navid Nikaein" ], "categories": [ "cs.NI", "cs.AI" ], "topics": [ "rag", "tool-use" ], "score": 12, "relevance": "medium", "matched_queries": [ "agentic-ai" ], "arxiv_id": "2606.21955", "source": "arxiv", "source_id": "arxiv:2606.21955", "pdf_url": "https://arxiv.org/pdf/2606.21955", "primary_query": "agentic-ai" }, { "id": "2606.21894", "title": "Skills for the future software profession: beyond agentic AI!", "url": "https://arxiv.org/abs/2606.21894", "published": "2026-06-20", "updated": "2026-06-23", "authors": [ "Sungmin Kang", "Baishakhi Ray", "Abhik Roychoudhury" ], "categories": [ "cs.SE", "cs.AI" ], "topics": [ "coding-agent", "tool-use" ], "score": 12, "relevance": "medium", "matched_queries": [ "agentic-ai", "coding-agent" ], "arxiv_id": "2606.21894", "source": "arxiv", "source_id": "arxiv:2606.21894", "pdf_url": "https://arxiv.org/pdf/2606.21894", "primary_query": "agentic-ai" }, { "id": "2606.20978", "title": "How Should Agents Read Demonstrations? Hierarchical Structure Beats Flat Action Logs", "url": "https://arxiv.org/abs/2606.20978", "published": "2026-06-18", "updated": "2026-06-18", "authors": [ "Honjar Xing", "Jefferson Lin", "Henry Lieberman" ], "categories": [ "cs.AI", "cs.HC" ], "topics": [ "agent-evaluation", "planning", "tool-use", "workflow-agent" ], "score": 12, "relevance": "medium", "matched_queries": [ "planning-agent" ], "arxiv_id": "2606.20978", "source": "arxiv", "source_id": "arxiv:2606.20978", "pdf_url": "https://arxiv.org/pdf/2606.20978", "primary_query": "planning-agent" }, { "id": "2606.20729", "title": "LLM-Guided Test-Time Discovery of Quantum-Chemical Approximation Algorithms", "url": "https://arxiv.org/abs/2606.20729", "published": "2026-06-17", "updated": "2026-06-17", "authors": [ "Masaya Hagai", "Yuta Suzuki", "Tomoya Murata", "Shuhei Kurita", "Masaki Adachi" ], "categories": [ "physics.chem-ph", "cond-mat.mtrl-sci", "cs.AI", "cs.LG" ], "topics": [ "agent-evaluation", "computer-use", "tool-use", "workflow-agent", "world-model" ], "score": 12, "relevance": "medium", "matched_queries": [ "agentic-ai" ], "arxiv_id": "2606.20729", "source": "arxiv", "source_id": "arxiv:2606.20729", "pdf_url": "https://arxiv.org/pdf/2606.20729", "primary_query": "agentic-ai" }, { "id": "2606.19416", "title": "MortarBench: Evaluating Mortgage Loan Origination Agents", "url": "https://arxiv.org/abs/2606.19416", "published": "2026-06-17", "updated": "2026-06-22", "authors": [ "Matthew Toles", "Yunan Lu", "Manav Munjal", "Bojun Liu", "Yuanhao Deng", "Stephanie Selig", "Derek Rindner", "Cheng Li", "Zhou Yu" ], "categories": [ "cs.LG" ], "topics": [ "agent-evaluation", "agent-safety", "rag", "workflow-agent" ], "score": 12, "relevance": "medium", "matched_queries": [ "agent-evaluation" ], "arxiv_id": "2606.19416", "source": "arxiv", "source_id": "arxiv:2606.19416", "pdf_url": "https://arxiv.org/pdf/2606.19416", "primary_query": "agent-evaluation" }, { "id": "2606.18550", "title": "The Gate Is Only as Honest as Its Contracts: ContractGuard for the Contract Layer of Risk-Aware Causal Gating", "url": "https://arxiv.org/abs/2606.18550", "published": "2026-06-17", "updated": "2026-06-17", "authors": [ "Laxmipriya Ganesh Iyer", "Rahul Suresh Babu" ], "categories": [ "cs.CR" ], "topics": [ "agent-evaluation", "agent-safety", "reasoning", "tool-use" ], "score": 12, "relevance": "medium", "matched_queries": [ "tool-use" ], "arxiv_id": "2606.18550", "source": "arxiv", "source_id": "arxiv:2606.18550", "pdf_url": "https://arxiv.org/pdf/2606.18550", "primary_query": "tool-use" }, { "id": "2606.19616", "title": "Before the Pull Request: Mining Multi-Agent Coordination", "url": "https://arxiv.org/abs/2606.19616", "published": "2026-06-17", "updated": "2026-06-17", "authors": [ "Dipankar Sarkar" ], "categories": [ "cs.SE", "cs.AI", "cs.MA" ], "topics": [ "coding-agent", "multi-agent", "tool-use" ], "score": 12, "relevance": "medium", "matched_queries": [ "coding-agent" ], "arxiv_id": "2606.19616", "source": "arxiv", "source_id": "arxiv:2606.19616", "pdf_url": "https://arxiv.org/pdf/2606.19616", "primary_query": "coding-agent" }, { "id": "2606.18890", "title": "Skill-Guided Continuation Distillation for GUI Agents", "url": "https://arxiv.org/abs/2606.18890", "published": "2026-06-17", "updated": "2026-06-17", "authors": [ "Zhimin Fan", "Hongwei Yu", "Yeqing Shen", "Haolong Yan", "Guozhen Peng", "Tianhao Peng", "Yudong Zhang", "Xiaowen Zhang", "Kaijun Tan", "Zheng Ge", "Xiangyu Zhang", "Daxin Jiang" ], "categories": [ "cs.AI" ], "topics": [ "computer-use", "planning", "reasoning", "tool-use" ], "score": 12, "relevance": "medium", "matched_queries": [ "web-gui-agent" ], "arxiv_id": "2606.18890", "source": "arxiv", "source_id": "arxiv:2606.18890", "pdf_url": "https://arxiv.org/pdf/2606.18890", "primary_query": "web-gui-agent" }, { "id": "2606.19602", "title": "Configurable Clinical Information Extraction with Agentic RAG: What Works, What Breaks, and Why", "url": "https://arxiv.org/abs/2606.19602", "published": "2026-06-17", "updated": "2026-06-17", "authors": [ "Osman Alperen Çinar-Koraş", "Marie Bauer", "Sameh Khattab", "Merlin Engelke", "Moon Kim", "Stephan Settelmeier", "Shigeyasu Sugawara", "Fabian Freisleben", "Felix Nensa", "Jens Kleesiek" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "rag", "reasoning", "tool-use" ], "score": 12, "relevance": "medium", "matched_queries": [ "rag-agent" ], "arxiv_id": "2606.19602", "source": "arxiv", "source_id": "arxiv:2606.19602", "pdf_url": "https://arxiv.org/pdf/2606.19602", "primary_query": "rag-agent" }, { "id": "2606.30658", "title": "Agentic AI Enhances Physician Trust in Clinical Decision Making", "url": "https://arxiv.org/abs/2606.30658", "published": "2026-06-16", "updated": "2026-06-16", "authors": [ "Zhiling Yan", "Zhe Fang", "David J King", "Ann Pongsakul", "Eashan Adhikarla", "Hui Ren", "Sunyang Fu", "Quanzheng Li", "Lifang He", "Xiang Li", "Hongfang Liu", "Yonghui Wu", "Lichao Sun" ], "categories": [ "cs.CY", "cs.AI" ], "topics": [ "agent-evaluation", "planning", "reasoning", "tool-use" ], "score": 12, "relevance": "medium", "matched_queries": [ "agentic-ai" ], "arxiv_id": "2606.30658", "source": "arxiv", "source_id": "arxiv:2606.30658", "pdf_url": "https://arxiv.org/pdf/2606.30658", "primary_query": "agentic-ai" }, { "id": "2606.17628", "title": "OPD-Evolver: Cultivating Holistic Agent Evolver via On-Policy Distillation", "url": "https://arxiv.org/abs/2606.17628", "published": "2026-06-16", "updated": "2026-06-16", "authors": [ "Guibin Zhang", "Xun Xu", "Yanwei Yue", "Zikun Su", "Wangchunshu Zhou", "Xiaobin Hu", "Shuicheng Yan" ], "categories": [ "cs.CL" ], "topics": [ "agent-evaluation", "coding-agent", "memory", "reasoning", "tool-use" ], "score": 12, "relevance": "medium", "matched_queries": [ "agent-memory" ], "arxiv_id": "2606.17628", "source": "arxiv", "source_id": "arxiv:2606.17628", "pdf_url": "https://arxiv.org/pdf/2606.17628", "primary_query": "agent-memory" }, { "id": "2606.20724", "title": "When Web Agents Finish but Still Fail: Reproducible Triggers and Trace Diagnostics for Parallel Web Exploration", "url": "https://arxiv.org/abs/2606.20724", "published": "2026-06-16", "updated": "2026-06-29", "authors": [ "Aagam Sogani", "Botao Rui", "Swetha Vaidyanathan", "Rishi Agarwal", "Minghao Yan", "Shivaram Venkataraman" ], "categories": [ "cs.AI", "cs.LG" ], "topics": [ "agent-evaluation", "computer-use", "planning", "rag", "tool-use" ], "score": 12, "relevance": "medium", "matched_queries": [ "web-gui-agent" ], "arxiv_id": "2606.20724", "source": "arxiv", "source_id": "arxiv:2606.20724", "pdf_url": "https://arxiv.org/pdf/2606.20724", "primary_query": "web-gui-agent" }, { "id": "2606.18448", "title": "VISUALSKILL: Multimodal Skills for Computer-Use Agents", "url": "https://arxiv.org/abs/2606.18448", "published": "2026-06-16", "updated": "2026-06-16", "authors": [ "Ziyan Jiang", "Li An", "Yujian Liu", "Jiabao Ji", "Qiucheng Wu", "Jacob Andreas", "Yang Zhang", "Shiyu Chang" ], "categories": [ "cs.CL" ], "topics": [ "agent-evaluation", "computer-use", "planning", "rag", "tool-use", "workflow-agent" ], "score": 12, "relevance": "medium", "matched_queries": [ "web-gui-agent" ], "arxiv_id": "2606.18448", "source": "arxiv", "source_id": "arxiv:2606.18448", "pdf_url": "https://arxiv.org/pdf/2606.18448", "primary_query": "web-gui-agent" }, { "id": "2606.15139", "title": "Self-Driving Negotiator: An interactive, verifiable benchmark for social negotiation and theory of mind under hidden intent", "url": "https://arxiv.org/abs/2606.15139", "published": "2026-06-13", "updated": "2026-06-13", "authors": [ "Ashutosh Kumar" ], "categories": [ "cs.GT", "cs.RO" ], "topics": [ "agent-evaluation", "planning", "rag", "tool-use" ], "score": 12, "relevance": "medium", "matched_queries": [ "language-agent" ], "arxiv_id": "2606.15139", "source": "arxiv", "source_id": "arxiv:2606.15139", "pdf_url": "https://arxiv.org/pdf/2606.15139", "primary_query": "language-agent" }, { "id": "2606.15385", "title": "Reward Hacking in Language Model Agents: Revisiting AI Safety Gridworlds", "url": "https://arxiv.org/abs/2606.15385", "published": "2026-06-13", "updated": "2026-06-13", "authors": [ "Ömer Veysel Çağatan", "Xuandong Zhao" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "agent-safety", "coding-agent" ], "score": 12, "relevance": "medium", "matched_queries": [ "agent-safety" ], "arxiv_id": "2606.15385", "source": "arxiv", "source_id": "arxiv:2606.15385", "pdf_url": "https://arxiv.org/pdf/2606.15385", "primary_query": "agent-safety" }, { "id": "2606.14989", "title": "Hierarchical Generative Agents for Simulating Sequential Human Behavior", "url": "https://arxiv.org/abs/2606.14989", "published": "2026-06-12", "updated": "2026-06-12", "authors": [ "Maria G. Mendoza", "Lucas Waldburger", "Jin Lee", "Shankar Sastry" ], "categories": [ "cs.MA" ], "topics": [ "embodied-agent", "planning", "reasoning", "world-model" ], "score": 12, "relevance": "medium", "matched_queries": [ "planning-agent" ], "arxiv_id": "2606.14989", "source": "arxiv", "source_id": "arxiv:2606.14989", "pdf_url": "https://arxiv.org/pdf/2606.14989", "primary_query": "planning-agent" }, { "id": "2606.13220", "title": "LLM-as-an-Investigator: Evidence-First Reasoning for Robust Interactive Problem Diagnosis", "url": "https://arxiv.org/abs/2606.13220", "published": "2026-06-11", "updated": "2026-06-11", "authors": [ "Fabrizio Marozzo", "Pietro Liò" ], "categories": [ "cs.AI", "cs.CE", "cs.ET", "cs.LG", "cs.MA" ], "topics": [ "agent-evaluation", "computer-use", "planning", "reasoning" ], "score": 12, "relevance": "medium", "matched_queries": [ "agent-evaluation" ], "arxiv_id": "2606.13220", "source": "arxiv", "source_id": "arxiv:2606.13220", "pdf_url": "https://arxiv.org/pdf/2606.13220", "primary_query": "agent-evaluation" }, { "id": "2606.12290", "title": "Selection Integrity for LLM Graph Memory: An Accumulability Criterion for Information-Flow-Blind Retrieval", "url": "https://arxiv.org/abs/2606.12290", "published": "2026-06-10", "updated": "2026-06-10", "authors": [ "Zeming Fei", "Hongming Fei", "Xiaoyang Wang", "Yang yang", "Prosanta Gope", "Biplab Sikdar", "Ying Zhang" ], "categories": [ "cs.CR" ], "topics": [ "agent-evaluation", "memory", "rag", "tool-use" ], "score": 12, "relevance": "medium", "matched_queries": [ "agent-memory" ], "arxiv_id": "2606.12290", "source": "arxiv", "source_id": "arxiv:2606.12290", "pdf_url": "https://arxiv.org/pdf/2606.12290", "primary_query": "agent-memory" }, { "id": "2606.12587", "title": "Strategic Decision Support for AI Agents", "url": "https://arxiv.org/abs/2606.12587", "published": "2026-06-10", "updated": "2026-06-10", "authors": [ "Shayan Kiyani", "Sima Noorani", "George Pappas", "Hamed Hassani" ], "categories": [ "cs.AI", "cs.HC" ], "topics": [ "multi-agent", "tool-use" ], "score": 12, "relevance": "medium", "matched_queries": [ "tool-use" ], "arxiv_id": "2606.12587", "source": "arxiv", "source_id": "arxiv:2606.12587", "pdf_url": "https://arxiv.org/pdf/2606.12587", "primary_query": "tool-use" }, { "id": "2606.10651", "title": "Kwai Keye-VL-2.0 Technical Report", "url": "https://arxiv.org/abs/2606.10651", "published": "2026-06-09", "updated": "2026-06-09", "authors": [ "Kwai Keye Team", "Bin Wen", "Changyi Liu", "Chengru Song", "Chongling Rao", "Guowang Zhang", "Han Li", "Haonan Fan", "Hengrui Ju", "Jiankang Chen", "Jiapeng Chen", "Jiawei Yuan", "Kaixuan Yang", "Kaiyu Jiang", "Kun Gai", "Lingzhi Zhou", "Na Nie", "Sen Na", "Tianke Zhang", "Tingting Gao", "Xuanyu Zheng", "Yulong Chen", "Fan Yang", "Haixuan Gao", "Lele Yang", "Mingqiao Liu", "Muxi Diao", "Qi Zhang", "Qile Su", "Wei Chen", "Wentao Hong", "Xingyu Lu", "Yancheng Long", "Yankai Yang", "Yingxin Li", "Yiyang Fan", "Yu Xia", "Yuzhe Chen", "Ziliang Lai", "Chuan Yi", "Haonan Jia", "Tianming Liang", "Weixin Xu", "Xiaoxiao Ma", "Yang Tian", "Yufei Han", "Feng Han", "Hang Li", "Jing Wang", "Jinghui Jia", "Junmin Chen", "Junyu Shi", "Ruilin Zhang" ], "categories": [ "cs.CV" ], "topics": [ "agent-evaluation", "agent-safety", "multi-agent", "rag", "reasoning", "tool-use" ], "score": 12, "relevance": "medium", "matched_queries": [ "agent-evaluation" ], "arxiv_id": "2606.10651", "source": "arxiv", "source_id": "arxiv:2606.10651", "pdf_url": "https://arxiv.org/pdf/2606.10651", "primary_query": "agent-evaluation" }, { "id": "2606.10956", "title": "Mind the Gap: Can Frontier LLMs Pass a Standardized Office Proficiency Exam?", "url": "https://arxiv.org/abs/2606.10956", "published": "2026-06-09", "updated": "2026-06-09", "authors": [ "Tengchao Lv", "Dongdong Zhang", "Jiayu Ding", "Yilin Jia", "Yuzhong Zhao", "Yupan Huang", "Wenshan Wu", "Xiangyang Zhou", "Shaohan Huang", "Nan Yang", "Li Dong", "Lei Cui", "Furu Wei" ], "categories": [ "cs.AI", "cs.CL" ], "topics": [ "agent-evaluation", "planning", "reasoning", "workflow-agent" ], "score": 12, "relevance": "medium", "matched_queries": [ "planning-agent" ], "arxiv_id": "2606.10956", "source": "arxiv", "source_id": "arxiv:2606.10956", "pdf_url": "https://arxiv.org/pdf/2606.10956", "primary_query": "planning-agent" }, { "id": "2606.08661", "title": "Data Agents Under Attack: Vulnerabilities in LLM-Driven Analytical Systems", "url": "https://arxiv.org/abs/2606.08661", "published": "2026-06-07", "updated": "2026-06-07", "authors": [ "Kuncan Wang", "Ziting Wang", "Peizhuo Lv", "Haoyang Li", "Guoliang Li", "Gao Cong", "Wei Dong" ], "categories": [ "cs.CR", "cs.AI", "cs.DB" ], "topics": [ "agent-evaluation", "agent-safety", "reasoning", "tool-use", "workflow-agent" ], "score": 12, "relevance": "medium", "matched_queries": [ "agent-safety" ], "arxiv_id": "2606.08661", "source": "arxiv", "source_id": "arxiv:2606.08661", "pdf_url": "https://arxiv.org/pdf/2606.08661", "primary_query": "agent-safety" }, { "id": "2606.08539", "title": "AgentTrust: A Self-Improving Trust Layer for AI-Agent Actions", "url": "https://arxiv.org/abs/2606.08539", "published": "2026-06-07", "updated": "2026-06-07", "authors": [ "Chenglin Yang" ], "categories": [ "cs.AI" ], "topics": [ "memory", "rag", "tool-use" ], "score": 12, "relevance": "medium", "matched_queries": [ "rag-agent" ], "arxiv_id": "2606.08539", "source": "arxiv", "source_id": "arxiv:2606.08539", "pdf_url": "https://arxiv.org/pdf/2606.08539", "primary_query": "rag-agent" }, { "id": "2606.24896", "title": "Why Memory Components Fail: Eight Years of License and Sustainability Events in Open-Source Data Infrastructure", "url": "https://arxiv.org/abs/2606.24896", "published": "2026-06-05", "updated": "2026-06-05", "authors": [ "Dmitrii Dmitrenko" ], "categories": [ "cs.DL", "cs.CY" ], "topics": [ "coding-agent", "memory", "tool-use" ], "score": 12, "relevance": "medium", "matched_queries": [ "agent-memory" ], "arxiv_id": "2606.24896", "source": "arxiv", "source_id": "arxiv:2606.24896", "pdf_url": "https://arxiv.org/pdf/2606.24896", "primary_query": "agent-memory" }, { "id": "2606.07074", "title": "SlimSearcher: Training Efficiency-Aware Web Agents via Adaptive Reward Gating", "url": "https://arxiv.org/abs/2606.07074", "published": "2026-06-05", "updated": "2026-06-05", "authors": [ "Zequn Xie", "Junjie Wang", "Dan Yang", "Jie Feng", "Yue Shen", "Jian Wang", "Jinjie Gu" ], "categories": [ "cs.LG", "cs.AI" ], "topics": [ "agent-evaluation", "computer-use", "planning", "rag", "reasoning", "tool-use" ], "score": 12, "relevance": "medium", "matched_queries": [ "web-gui-agent" ], "arxiv_id": "2606.07074", "source": "arxiv", "source_id": "arxiv:2606.07074", "pdf_url": "https://arxiv.org/pdf/2606.07074", "primary_query": "web-gui-agent" }, { "id": "2606.07027", "title": "StainFlow: Entity-Stain Tracking and Evidence Linking for Process Rewards in GUI Agents", "url": "https://arxiv.org/abs/2606.07027", "published": "2026-06-05", "updated": "2026-06-12", "authors": [ "Haojie Hao", "Longkun Hao", "Yihang Lou", "Yan Bai", "Zhenyang Li", "Zhichao Yang", "Dongshuo Huang", "Hongyu Lin", "Lanqing Hong", "Jiakai Wang", "Xianglong Liu" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "computer-use", "planning" ], "score": 12, "relevance": "medium", "matched_queries": [ "web-gui-agent" ], "arxiv_id": "2606.07027", "source": "arxiv", "source_id": "arxiv:2606.07027", "pdf_url": "https://arxiv.org/pdf/2606.07027", "primary_query": "web-gui-agent" }, { "id": "2606.07486", "title": "OPENPATH: A Supervisor--Specialist Agent System for Personalized, Accessible, and Multi-stop Urban Trip Planning", "url": "https://arxiv.org/abs/2606.07486", "published": "2026-06-05", "updated": "2026-06-05", "authors": [ "Ziyang Xiong", "He Zong", "Zhiyuan Xue", "Manxi Wu" ], "categories": [ "eess.SY" ], "topics": [ "multi-agent", "planning" ], "score": 12, "relevance": "medium", "matched_queries": [ "planning-agent" ], "arxiv_id": "2606.07486", "source": "arxiv", "source_id": "arxiv:2606.07486", "pdf_url": "https://arxiv.org/pdf/2606.07486", "primary_query": "planning-agent" }, { "id": "2606.05553", "title": "ArcANE: Do Role-Playing Language Agents Stay in Character at the Right Time?", "url": "https://arxiv.org/abs/2606.05553", "published": "2026-06-04", "updated": "2026-06-04", "authors": [ "Woojung Song", "Nalim Kim", "Sangjun Song", "Chaewon Heo", "Jongwon Lim", "Yohan Jo" ], "categories": [ "cs.CL", "cs.AI" ], "topics": [ "agent-evaluation", "rag" ], "score": 12, "relevance": "medium", "matched_queries": [ "language-agent" ], "arxiv_id": "2606.05553", "source": "arxiv", "source_id": "arxiv:2606.05553", "pdf_url": "https://arxiv.org/pdf/2606.05553", "primary_query": "language-agent" }, { "id": "2606.05901", "title": "Reducing Hallucinations in Complex Question Answering using Simple Graph-based Retrieval-Augmented Generation (long version)", "url": "https://arxiv.org/abs/2606.05901", "published": "2026-06-04", "updated": "2026-06-04", "authors": [ "Christopher J. Wedge", "Joshua Stutter", "Danny Dixon", "Jacek Cała" ], "categories": [ "cs.CL", "cs.AI" ], "topics": [ "agent-evaluation", "agent-safety", "rag", "reasoning", "tool-use" ], "score": 12, "relevance": "medium", "matched_queries": [ "rag-agent" ], "arxiv_id": "2606.05901", "source": "arxiv", "source_id": "arxiv:2606.05901", "pdf_url": "https://arxiv.org/pdf/2606.05901", "primary_query": "rag-agent" }, { "id": "2606.05233", "title": "Domain-Conditioned Safety in Frontier Computer-Using Agents: A 793-Episode Browser Benchmark, a Coding-Domain Cross-Reference, and a Reproducibility Audit of Recent Red-Teaming", "url": "https://arxiv.org/abs/2606.05233", "published": "2026-06-03", "updated": "2026-06-03", "authors": [ "Nicholas Saban" ], "categories": [ "cs.CR", "cs.AI", "cs.CL" ], "topics": [ "agent-evaluation", "agent-safety", "coding-agent", "computer-use" ], "score": 12, "relevance": "medium", "matched_queries": [ "agent-evaluation" ], "arxiv_id": "2606.05233", "source": "arxiv", "source_id": "arxiv:2606.05233", "pdf_url": "https://arxiv.org/pdf/2606.05233", "primary_query": "agent-evaluation" }, { "id": "2606.09890", "title": "PreAct-Bench: Benchmarking Predictive Monitoring in LLMs", "url": "https://arxiv.org/abs/2606.09890", "published": "2026-06-03", "updated": "2026-06-03", "authors": [ "Hainiu Xu", "Italo Luis da Silva", "Jiangnan Ye", "Yuhao Wang", "Wei Liu", "Linyi Yang", "Jonathan Richard Schwarz", "Nicola Paoletti", "Yulan He", "Hanqi Yan" ], "categories": [ "cs.LG", "cs.AI", "cs.CL" ], "topics": [ "agent-evaluation", "agent-safety", "reasoning", "tool-use" ], "score": 12, "relevance": "medium", "matched_queries": [ "autonomous-agent-llm" ], "arxiv_id": "2606.09890", "source": "arxiv", "source_id": "arxiv:2606.09890", "pdf_url": "https://arxiv.org/pdf/2606.09890", "primary_query": "autonomous-agent-llm" }, { "id": "2606.03054", "title": "ToolGate: Token-Efficient Pre-Call Control for Tool-Augmented Vision-Language Agents", "url": "https://arxiv.org/abs/2606.03054", "published": "2026-06-02", "updated": "2026-06-02", "authors": [ "Anjie Liu", "Yan Song", "Zhixun Chen", "Ziqin Gong", "Zhongwei Yu", "Jun Wang" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "rag", "tool-use" ], "score": 12, "relevance": "medium", "matched_queries": [ "language-agent" ], "arxiv_id": "2606.03054", "source": "arxiv", "source_id": "arxiv:2606.03054", "pdf_url": "https://arxiv.org/pdf/2606.03054", "primary_query": "language-agent" }, { "id": "2606.02994", "title": "Inducing Reasoning Primitives from Agent Traces", "url": "https://arxiv.org/abs/2606.02994", "published": "2026-06-02", "updated": "2026-06-02", "authors": [ "Zhihan Lei", "Jiarui Yan", "Joshua Momo", "William W. Cohen" ], "categories": [ "cs.AI", "cs.CL" ], "topics": [ "planning", "rag", "reasoning", "tool-use" ], "score": 12, "relevance": "medium", "matched_queries": [ "planning-agent" ], "arxiv_id": "2606.02994", "source": "arxiv", "source_id": "arxiv:2606.02994", "pdf_url": "https://arxiv.org/pdf/2606.02994", "primary_query": "planning-agent" }, { "id": "2606.02754", "title": "$Ψ$-Bench: Evaluating Persona-Sensitive Influencing in Persuasive Dialogues", "url": "https://arxiv.org/abs/2606.02754", "published": "2026-06-01", "updated": "2026-06-01", "authors": [ "Peixuan Han", "Hongyi Du", "Jiayu Liu", "Yihang Sun", "Yutong Liu", "Jiaxuan You" ], "categories": [ "cs.LG" ], "topics": [ "agent-evaluation", "computer-use", "rag", "tool-use" ], "score": 12, "relevance": "medium", "matched_queries": [ "language-agent" ], "arxiv_id": "2606.02754", "source": "arxiv", "source_id": "arxiv:2606.02754", "pdf_url": "https://arxiv.org/pdf/2606.02754", "primary_query": "language-agent" }, { "id": "2606.02875", "title": "Handoff Debt: The Rediscovery Cost When Coding Agents Take Over Interrupted Tasks", "url": "https://arxiv.org/abs/2606.02875", "published": "2026-06-01", "updated": "2026-06-01", "authors": [ "Dipesh KC", "Anjila Budathoki" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "coding-agent" ], "score": 12, "relevance": "medium", "matched_queries": [ "agent-evaluation" ], "arxiv_id": "2606.02875", "source": "arxiv", "source_id": "arxiv:2606.02875", "pdf_url": "https://arxiv.org/pdf/2606.02875", "primary_query": "agent-evaluation" }, { "id": "2605.29486", "title": "PhoneWorld: Scaling Phone-Use Agent Environments", "url": "https://arxiv.org/abs/2605.29486", "published": "2026-05-28", "updated": "2026-05-28", "authors": [ "Zhengyang Tang", "Yuxuan Liu", "Xin Lai", "Junyi Li", "Pengyuan Lyu", "Jason", "Yiduo Guo", "Zhengyao Fang", "Yang Ding", "Yi Zhang", "Weinong Wang", "Huawen Shen", "Xingran Zhou", "Liang Wu", "Fei Tang", "Sunqi Fan", "Shangpin Peng", "Zheng Ruan", "Anran Zhang", "Benyou Wang", "Rui Yan", "Ji-Rong Wen", "Chengquan Zhang", "Han Hu" ], "categories": [ "cs.CL", "cs.AI", "cs.LG" ], "topics": [ "agent-evaluation", "computer-use", "rag", "reasoning", "tool-use" ], "score": 12, "relevance": "medium", "matched_queries": [ "agent-evaluation" ], "arxiv_id": "2605.29486", "source": "arxiv", "source_id": "arxiv:2605.29486", "pdf_url": "https://arxiv.org/pdf/2605.29486", "primary_query": "agent-evaluation" }, { "id": "2605.28629", "title": "Mobile-Aptus: Confidence-Driven Proactive and Robust Interaction in MLLM-based Mobile-Using Agents", "url": "https://arxiv.org/abs/2605.28629", "published": "2026-05-27", "updated": "2026-05-27", "authors": [ "Zheng Wu", "Pengzhou Cheng", "Zongru Wu", "Yuan Guo", "Tianjie Ju", "Aston Zhang", "Gongshen Liu", "Zhuosheng Zhang" ], "categories": [ "cs.CL" ], "topics": [ "agent-evaluation", "computer-use", "rag", "tool-use" ], "score": 12, "relevance": "medium", "matched_queries": [ "agent-evaluation" ], "arxiv_id": "2605.28629", "source": "arxiv", "source_id": "arxiv:2605.28629", "pdf_url": "https://arxiv.org/pdf/2605.28629", "primary_query": "agent-evaluation" }, { "id": "2605.27331", "title": "Maat: The Agentic Legal Research Assistant for Competition Protection", "url": "https://arxiv.org/abs/2605.27331", "published": "2026-05-26", "updated": "2026-05-26", "authors": [ "Basant Mounir", "Farida Madkour", "Amira Abdelaziz", "Asmaa Sami" ], "categories": [ "cs.AI" ], "topics": [ "rag", "tool-use" ], "score": 12, "relevance": "medium", "matched_queries": [ "rag-agent" ], "arxiv_id": "2605.27331", "source": "arxiv", "source_id": "arxiv:2605.27331", "pdf_url": "https://arxiv.org/pdf/2605.27331", "primary_query": "rag-agent" }, { "id": "2605.25641", "title": "Iterate Until Retrieved: Factual Nugget Optimization for Discoverable Continual Corrections in Agentic RAG", "url": "https://arxiv.org/abs/2605.25641", "published": "2026-05-25", "updated": "2026-05-25", "authors": [ "Moshe Hazoom", "Gal Patel", "Alon Talmor", "Tom Hope" ], "categories": [ "cs.CL" ], "topics": [ "agent-evaluation", "rag", "tool-use" ], "score": 12, "relevance": "medium", "matched_queries": [ "rag-agent" ], "arxiv_id": "2605.25641", "source": "arxiv", "source_id": "arxiv:2605.25641", "pdf_url": "https://arxiv.org/pdf/2605.25641", "primary_query": "rag-agent" }, { "id": "2605.25480", "title": "Retrieval as Reasoning: Self-Evolving Agent-Native Retrieval via LLM-Wiki", "url": "https://arxiv.org/abs/2605.25480", "published": "2026-05-25", "updated": "2026-05-26", "authors": [ "Haoliang Ming", "Feifei Li", "Xiaoqing Wu", "Wenhui Que" ], "categories": [ "cs.CL" ], "topics": [ "agent-evaluation", "rag", "reasoning", "tool-use" ], "score": 12, "relevance": "medium", "matched_queries": [ "rag-agent" ], "arxiv_id": "2605.25480", "source": "arxiv", "source_id": "arxiv:2605.25480", "pdf_url": "https://arxiv.org/pdf/2605.25480", "primary_query": "rag-agent" }, { "id": "2605.25002", "title": "MemMark: State-Evolution Attribution Watermarking for Agent Long-Term Memory Systems", "url": "https://arxiv.org/abs/2605.25002", "published": "2026-05-24", "updated": "2026-05-26", "authors": [ "Haobo Zhang", "Xutao Mao", "Guangyuan Dong", "Ziwei Li", "Xuanbo Su", "Kaijie Chen", "Jing Yang", "Zheng Lin" ], "categories": [ "cs.CR" ], "topics": [ "computer-use", "memory" ], "score": 12, "relevance": "medium", "matched_queries": [ "agent-memory" ], "arxiv_id": "2605.25002", "source": "arxiv", "source_id": "arxiv:2605.25002", "pdf_url": "https://arxiv.org/pdf/2605.25002", "primary_query": "agent-memory" }, { "id": "2605.20485", "title": "ZEBRA: Zero-shot Budgeted Resource Allocation for LLM Orchestration", "url": "https://arxiv.org/abs/2605.20485", "published": "2026-05-19", "updated": "2026-05-19", "authors": [ "May Hamri", "Inbal Talgam-Cohen" ], "categories": [ "cs.LG" ], "topics": [ "agent-evaluation", "coding-agent", "computer-use", "multi-agent", "reasoning" ], "score": 12, "relevance": "medium", "matched_queries": [ "autonomous-agent-llm" ], "arxiv_id": "2605.20485", "source": "arxiv", "source_id": "arxiv:2605.20485", "pdf_url": "https://arxiv.org/pdf/2605.20485", "primary_query": "autonomous-agent-llm" }, { "id": "2605.18597", "title": "Latent Action Reparameterization for Efficient Agent Inference", "url": "https://arxiv.org/abs/2605.18597", "published": "2026-05-18", "updated": "2026-05-19", "authors": [ "Wenhao Huang", "Qingwen Zeng", "Qiyue Chen", "Zijie Guo", "Yu Sun", "Cheng Yang", "Siru Ouyang", "Jiri Gesi", "Fang Wu", "Jiayi Zhang", "Huaming Chen", "Bang Liu", "Xiangru Tang", "Chenglin Wu" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "planning", "tool-use" ], "score": 12, "relevance": "medium", "matched_queries": [ "planning-agent" ], "arxiv_id": "2605.18597", "source": "arxiv", "source_id": "arxiv:2605.18597", "pdf_url": "https://arxiv.org/pdf/2605.18597", "primary_query": "planning-agent" }, { "id": "2605.15665", "title": "PRISM: Prompt Reliability via Iterative Simulation and Monitoring for Enterprise Conversational AI", "url": "https://arxiv.org/abs/2605.15665", "published": "2026-05-15", "updated": "2026-05-15", "authors": [ "Keshava Chaitanya", "Jahnavi Gundakaram" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "memory", "tool-use", "workflow-agent", "world-model" ], "score": 12, "relevance": "medium", "matched_queries": [ "language-agent" ], "arxiv_id": "2605.15665", "source": "arxiv", "source_id": "arxiv:2605.15665", "pdf_url": "https://arxiv.org/pdf/2605.15665", "primary_query": "language-agent" }, { "id": "2605.14051", "title": "SPIN: Structural LLM Planning via Iterative Navigation for Industrial Tasks", "url": "https://arxiv.org/abs/2605.14051", "published": "2026-05-13", "updated": "2026-05-13", "authors": [ "Yusuke Ozaki", "Dhaval Patel" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "embodied-agent", "planning", "rag", "tool-use", "workflow-agent" ], "score": 12, "relevance": "medium", "matched_queries": [ "planning-agent" ], "arxiv_id": "2605.14051", "source": "arxiv", "source_id": "arxiv:2605.14051", "pdf_url": "https://arxiv.org/pdf/2605.14051", "primary_query": "planning-agent" }, { "id": "2605.13037", "title": "MAP: A Map-then-Act Paradigm for Long-Horizon Interactive Agent Reasoning", "url": "https://arxiv.org/abs/2605.13037", "published": "2026-05-13", "updated": "2026-05-13", "authors": [ "Yuxin Liu", "Ziang Ye", "Yueqing Sun", "Mingye Zhu", "Jinwei Xiao", "Zhuowen Han", "Qi GU", "Xunliang Cai", "Lei Zhang" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "planning", "reasoning" ], "score": 12, "relevance": "medium", "matched_queries": [ "planning-agent" ], "arxiv_id": "2605.13037", "source": "arxiv", "source_id": "arxiv:2605.13037", "pdf_url": "https://arxiv.org/pdf/2605.13037", "primary_query": "planning-agent" }, { "id": "2605.11224", "title": "ABRA: Agent Benchmark for Radiology Applications", "url": "https://arxiv.org/abs/2605.11224", "published": "2026-05-11", "updated": "2026-05-11", "authors": [ "Bulat Maksudov", "Vladislav Kurenkov", "Kathleen M. Curran", "Alessandra Mileo" ], "categories": [ "cs.CV", "cs.AI" ], "topics": [ "agent-evaluation", "embodied-agent", "planning", "tool-use" ], "score": 12, "relevance": "medium", "matched_queries": [ "function-calling" ], "arxiv_id": "2605.11224", "source": "arxiv", "source_id": "arxiv:2605.11224", "pdf_url": "https://arxiv.org/pdf/2605.11224", "primary_query": "function-calling" }, { "id": "2605.11003", "title": "The Authorization-Execution Gap Is a Major Safety and Security Problem in Open-World Agents", "url": "https://arxiv.org/abs/2605.11003", "published": "2026-05-10", "updated": "2026-05-10", "authors": [ "Baoyuan Wu", "Qingshan Liu", "Adel Bibi", "Irwin King", "Siwei Lyu" ], "categories": [ "cs.CR", "cs.AI" ], "topics": [ "agent-evaluation", "agent-safety", "multi-agent", "rag", "tool-use" ], "score": 12, "relevance": "medium", "matched_queries": [ "agent-safety" ], "arxiv_id": "2605.11003", "source": "arxiv", "source_id": "arxiv:2605.11003", "pdf_url": "https://arxiv.org/pdf/2605.11003", "primary_query": "agent-safety" }, { "id": "2605.03855", "title": "Evaluating Generative Models as Interactive Emergent Representations of Human-Like Collaborative Behavior", "url": "https://arxiv.org/abs/2605.03855", "published": "2026-05-05", "updated": "2026-05-06", "authors": [ "Shinas Shaji", "Teena Chakkalayil Hassan", "Sebastian Houben", "Alex Mitrevski" ], "categories": [ "cs.RO" ], "topics": [ "agent-evaluation", "embodied-agent", "multi-agent", "planning", "tool-use" ], "score": 12, "relevance": "medium", "matched_queries": [ "planning-agent" ], "arxiv_id": "2605.03855", "source": "arxiv", "source_id": "arxiv:2605.03855", "pdf_url": "https://arxiv.org/pdf/2605.03855", "primary_query": "planning-agent" }, { "id": "2605.00380", "title": "ResRL: Boosting LLM Reasoning via Negative Sample Projection Residual Reinforcement Learning", "url": "https://arxiv.org/abs/2605.00380", "published": "2026-05-01", "updated": "2026-05-08", "authors": [ "Zihan Lin", "Xiaohan Wang", "Jie Cao", "Jiajun Chai", "Li Wang", "Xiaodong Lu", "Wei Lin", "Ran He", "Guojun Yin" ], "categories": [ "cs.LG", "cs.CL" ], "topics": [ "agent-evaluation", "agent-safety", "computer-use", "rag", "reasoning", "tool-use" ], "score": 12, "relevance": "medium", "matched_queries": [ "function-calling" ], "arxiv_id": "2605.00380", "source": "arxiv", "source_id": "arxiv:2605.00380", "pdf_url": "https://arxiv.org/pdf/2605.00380", "primary_query": "function-calling" }, { "id": "2605.00060", "title": "TADI: Tool-Augmented Drilling Intelligence via Agentic LLM Orchestration over Heterogeneous Wellsite Data", "url": "https://arxiv.org/abs/2605.00060", "published": "2026-04-30", "updated": "2026-04-30", "authors": [ "Rong Lu" ], "categories": [ "cs.AI", "eess.SY" ], "topics": [ "rag", "tool-use" ], "score": 12, "relevance": "medium", "matched_queries": [ "function-calling" ], "arxiv_id": "2605.00060", "source": "arxiv", "source_id": "arxiv:2605.00060", "pdf_url": "https://arxiv.org/pdf/2605.00060", "primary_query": "function-calling" }, { "id": "2604.27132", "title": "TRUST: A Framework for Decentralized AI Service v.0.1", "url": "https://arxiv.org/abs/2604.27132", "published": "2026-04-29", "updated": "2026-04-29", "authors": [ "Yu-Chao Huang", "Zhen Tan", "Mohan Zhang", "Pingzhi Li", "Zhuo Zhang", "Tianlong Chen" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "agent-safety", "multi-agent", "reasoning", "tool-use" ], "score": 12, "relevance": "medium", "matched_queries": [ "autonomous-agent-llm" ], "arxiv_id": "2604.27132", "source": "arxiv", "source_id": "arxiv:2604.27132", "pdf_url": "https://arxiv.org/pdf/2604.27132", "primary_query": "autonomous-agent-llm" }, { "id": "2604.18982", "title": "SAVOIR: Learning Social Savoir-Faire via Shapley-based Reward Attribution", "url": "https://arxiv.org/abs/2604.18982", "published": "2026-04-21", "updated": "2026-04-21", "authors": [ "Xiachong Feng", "Yi Jiang", "Xiaocheng Feng", "Deyi Yin", "Libo Qin", "Yangfan Ye", "Lei Huang", "Weitao Ma", "Yuxuan Gu", "Chonghan Qin", "Bing Qin", "Lingpeng Kong" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "rag", "reasoning", "tool-use" ], "score": 12, "relevance": "medium", "matched_queries": [ "language-agent" ], "arxiv_id": "2604.18982", "source": "arxiv", "source_id": "arxiv:2604.18982", "pdf_url": "https://arxiv.org/pdf/2604.18982", "primary_query": "language-agent" }, { "id": "2604.03070", "title": "How Your Credentials Are Leaked by LLM Agent Skills: An Empirical Study", "url": "https://arxiv.org/abs/2604.03070", "published": "2026-04-03", "updated": "2026-06-19", "authors": [ "Zhihao Chen", "Ying Zhang", "Yi Liu", "Gelei Deng", "Yuekang Li", "Yanjun Zhang", "Jianting Ning", "Leo Yu Zhang", "Lei Ma", "Zhiqiang Li" ], "categories": [ "cs.CR", "cs.AI" ], "topics": [ "agent-safety", "tool-use" ], "score": 12, "relevance": "medium", "matched_queries": [ "agent-safety" ], "arxiv_id": "2604.03070", "source": "arxiv", "source_id": "arxiv:2604.03070", "pdf_url": "https://arxiv.org/pdf/2604.03070", "primary_query": "agent-safety" }, { "id": "2604.00830", "title": "Learning to Learn-at-Test-Time: Language Agents with Learnable Adaptation Policies", "url": "https://arxiv.org/abs/2604.00830", "published": "2026-04-01", "updated": "2026-04-02", "authors": [ "Zhanzhi Lou", "Hui Chen", "Yibo Li", "Qian Wang", "Bryan Hooi" ], "categories": [ "cs.LG", "cs.AI" ], "topics": [ "agent-evaluation", "computer-use", "tool-use" ], "score": 12, "relevance": "medium", "matched_queries": [ "language-agent" ], "arxiv_id": "2604.00830", "source": "arxiv", "source_id": "arxiv:2604.00830", "pdf_url": "https://arxiv.org/pdf/2604.00830", "primary_query": "language-agent" }, { "id": "2604.00992", "title": "Tube-Based Safety for Anticipative Tracking in Multi-Agent Systems", "url": "https://arxiv.org/abs/2604.00992", "published": "2026-04-01", "updated": "2026-04-01", "authors": [ "Armel Koulong", "Ali Pakniyat" ], "categories": [ "eess.SY" ], "topics": [ "agent-safety", "multi-agent", "world-model" ], "score": 12, "relevance": "medium", "matched_queries": [ "agent-safety" ], "arxiv_id": "2604.00992", "source": "arxiv", "source_id": "arxiv:2604.00992", "pdf_url": "https://arxiv.org/pdf/2604.00992", "primary_query": "agent-safety" }, { "id": "2603.29560", "title": "Distributed Predictive Control Barrier Functions: Towards Scalable Safety Certification in Modular Multi-Agent Systems", "url": "https://arxiv.org/abs/2603.29560", "published": "2026-03-31", "updated": "2026-03-31", "authors": [ "Jonas Ohnemus", "Alexandre Didier", "Ahmed Aboudonia", "Andrea Carron", "Melanie N. Zeilinger" ], "categories": [ "eess.SY", "cs.RO", "math.OC" ], "topics": [ "agent-safety", "multi-agent", "world-model" ], "score": 12, "relevance": "medium", "matched_queries": [ "agent-safety" ], "arxiv_id": "2603.29560", "source": "arxiv", "source_id": "arxiv:2603.29560", "pdf_url": "https://arxiv.org/pdf/2603.29560", "primary_query": "agent-safety" }, { "id": "2603.12644", "title": "Uncovering Security Threats and Architecting Defenses in Autonomous Agents: A Case Study of OpenClaw", "url": "https://arxiv.org/abs/2603.12644", "published": "2026-03-13", "updated": "2026-03-13", "authors": [ "Zonghao Ying", "Xiao Yang", "Siyang Wu", "Yumeng Song", "Yang Qu", "Hainan Li", "Tianlin Li", "Jiakai Wang", "Aishan Liu", "Xianglong Liu" ], "categories": [ "cs.CR" ], "topics": [ "agent-safety", "reasoning", "tool-use", "workflow-agent" ], "score": 12, "relevance": "medium", "matched_queries": [ "agent-safety" ], "arxiv_id": "2603.12644", "source": "arxiv", "source_id": "arxiv:2603.12644", "pdf_url": "https://arxiv.org/pdf/2603.12644", "primary_query": "agent-safety" }, { "id": "2603.10213", "title": "Sabiá-4 Technical Report", "url": "https://arxiv.org/abs/2603.10213", "published": "2026-03-10", "updated": "2026-03-10", "authors": [ "Thiago Laitz", "Thales Sales Almeida", "Hugo Abonizio", "Roseval Malaquias Junior", "Giovana Kerche Bonás", "Marcos Piau", "Celio Larcher", "Ramon Pires", "Rodrigo Nogueira" ], "categories": [ "cs.CL" ], "topics": [ "agent-evaluation", "agent-safety", "embodied-agent", "tool-use" ], "score": 12, "relevance": "medium", "matched_queries": [ "function-calling" ], "arxiv_id": "2603.10213", "source": "arxiv", "source_id": "arxiv:2603.10213", "pdf_url": "https://arxiv.org/pdf/2603.10213", "primary_query": "function-calling" }, { "id": "2603.08316", "title": "SlowBA: An efficiency backdoor attack towards VLM-based GUI agents", "url": "https://arxiv.org/abs/2603.08316", "published": "2026-03-09", "updated": "2026-07-01", "authors": [ "Junxian Li", "Tu Lan", "Haozhen Tan", "Yan Meng", "Haojin Zhu" ], "categories": [ "cs.CR", "cs.CL", "cs.CV" ], "topics": [ "agent-safety", "computer-use", "reasoning", "tool-use" ], "score": 12, "relevance": "medium", "matched_queries": [ "agent-safety" ], "arxiv_id": "2603.08316", "source": "arxiv", "source_id": "arxiv:2603.08316", "pdf_url": "https://arxiv.org/pdf/2603.08316", "primary_query": "agent-safety" }, { "id": "2602.23876", "title": "RF-Agent: Automated Reward Function Design via Language Agent Tree Search", "url": "https://arxiv.org/abs/2602.23876", "published": "2026-02-27", "updated": "2026-02-27", "authors": [ "Ning Gao", "Xiuhui Zhang", "Xingyu Jiang", "Mukang You", "Mohan Zhang", "Yue Deng" ], "categories": [ "cs.AI", "cs.LG" ], "topics": [ "rag", "reasoning" ], "score": 12, "relevance": "medium", "matched_queries": [ "language-agent" ], "arxiv_id": "2602.23876", "source": "arxiv", "source_id": "arxiv:2602.23876", "pdf_url": "https://arxiv.org/pdf/2602.23876", "primary_query": "language-agent" }, { "id": "2602.20156", "title": "Skill-Inject: Measuring Agent Vulnerability to Skill File Attacks", "url": "https://arxiv.org/abs/2602.20156", "published": "2026-02-23", "updated": "2026-02-25", "authors": [ "David Schmotz", "Luca Beurer-Kellner", "Sahar Abdelnabi", "Maksym Andriushchenko" ], "categories": [ "cs.CR", "cs.LG" ], "topics": [ "agent-evaluation", "agent-safety", "tool-use" ], "score": 12, "relevance": "medium", "matched_queries": [ "agent-safety" ], "arxiv_id": "2602.20156", "source": "arxiv", "source_id": "arxiv:2602.20156", "pdf_url": "https://arxiv.org/pdf/2602.20156", "primary_query": "agent-safety" }, { "id": "2602.17875", "title": "MultiVer: Zero-Shot Multi-Agent Vulnerability Detection", "url": "https://arxiv.org/abs/2602.17875", "published": "2026-02-19", "updated": "2026-02-19", "authors": [ "Shreshth Rajan" ], "categories": [ "cs.MA", "cs.AI" ], "topics": [ "agent-evaluation", "agent-safety", "multi-agent" ], "score": 12, "relevance": "medium", "matched_queries": [ "agent-safety" ], "arxiv_id": "2602.17875", "source": "arxiv", "source_id": "arxiv:2602.17875", "pdf_url": "https://arxiv.org/pdf/2602.17875", "primary_query": "agent-safety" }, { "id": "2512.00332", "title": "Assertion-Conditioned Compliance: A Provenance-Aware Vulnerability in Multi-Turn Tool-Calling Agents", "url": "https://arxiv.org/abs/2512.00332", "published": "2025-11-29", "updated": "2026-01-21", "authors": [ "Daud Waqas", "Aaryamaan Golthi", "Erika Hayashida", "Huanzhi Mao" ], "categories": [ "cs.CL", "cs.AI" ], "topics": [ "agent-evaluation", "agent-safety", "tool-use" ], "score": 12, "relevance": "medium", "matched_queries": [ "function-calling" ], "arxiv_id": "2512.00332", "source": "arxiv", "source_id": "arxiv:2512.00332", "pdf_url": "https://arxiv.org/pdf/2512.00332", "primary_query": "function-calling" }, { "id": "2510.24284", "title": "MCP-Flow: Facilitating LLM Agents to Master Real-World, Diverse and Scaling MCP Tools", "url": "https://arxiv.org/abs/2510.24284", "published": "2025-10-28", "updated": "2026-04-16", "authors": [ "Wenhao Wang", "Peizhi Niu", "Zhao Xu", "Zhaoyu Chen", "Jian Du", "Yaxin Du", "Xianghe Pang", "Keduan Huang", "Yanfeng Wang", "Qiang Yan", "Siheng Chen" ], "categories": [ "cs.AI" ], "topics": [ "tool-use" ], "score": 12, "relevance": "medium", "matched_queries": [ "function-calling" ], "arxiv_id": "2510.24284", "source": "arxiv", "source_id": "arxiv:2510.24284", "pdf_url": "https://arxiv.org/pdf/2510.24284", "primary_query": "function-calling" }, { "id": "2509.18420", "title": "Instruction-Following Evaluation in Function Calling for Large Language Models", "url": "https://arxiv.org/abs/2509.18420", "published": "2025-09-22", "updated": "2025-09-22", "authors": [ "Nikolai Skripko" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "tool-use" ], "score": 12, "relevance": "medium", "matched_queries": [ "function-calling" ], "arxiv_id": "2509.18420", "source": "arxiv", "source_id": "arxiv:2509.18420", "pdf_url": "https://arxiv.org/pdf/2509.18420", "primary_query": "function-calling" }, { "id": "2509.18169", "title": "PiERN: Token-Level Routing for Integrating High-Precision Computation and Reasoning", "url": "https://arxiv.org/abs/2509.18169", "published": "2025-09-17", "updated": "2026-04-20", "authors": [ "Hengbo Xiao", "Jingyuan Fan", "Xin Tong", "Jingzhao Zhang", "Chao Lu", "Guannan He" ], "categories": [ "cs.LG", "cs.CE", "cs.CL" ], "topics": [ "agent-evaluation", "multi-agent", "rag", "reasoning", "tool-use", "workflow-agent" ], "score": 12, "relevance": "medium", "matched_queries": [ "function-calling" ], "arxiv_id": "2509.18169", "source": "arxiv", "source_id": "arxiv:2509.18169", "pdf_url": "https://arxiv.org/pdf/2509.18169", "primary_query": "function-calling" }, { "id": "2508.20637", "title": "GDS Agent for Graph Algorithmic Reasoning", "url": "https://arxiv.org/abs/2508.20637", "published": "2025-08-28", "updated": "2025-11-05", "authors": [ "Borun Shi", "Ioannis Panagiotas" ], "categories": [ "cs.LG", "cs.AI", "cs.CL" ], "topics": [ "agent-evaluation", "rag", "reasoning", "tool-use" ], "score": 12, "relevance": "medium", "matched_queries": [ "function-calling" ], "arxiv_id": "2508.20637", "source": "arxiv", "source_id": "arxiv:2508.20637", "pdf_url": "https://arxiv.org/pdf/2508.20637", "primary_query": "function-calling" }, { "id": "2607.06184", "title": "What Resolve Rate Hides: Trajectory Structure Diagnostics for Coding Agents", "url": "https://arxiv.org/abs/2607.06184", "published": "2026-07-07", "updated": "2026-07-07", "authors": [ "Rui Shu", "Chun Yong Chong", "Xin Zhou", "Yun Peng", "Zihan Wu", "Xu Han", "Zeyang Zhuang", "Guowen Yuan", "Yuan Wang" ], "categories": [ "cs.SE" ], "topics": [ "coding-agent", "tool-use" ], "score": 11, "relevance": "medium", "matched_queries": [ "coding-agent" ], "arxiv_id": "2607.06184", "source": "arxiv", "source_id": "arxiv:2607.06184", "pdf_url": "https://arxiv.org/pdf/2607.06184", "primary_query": "coding-agent" }, { "id": "2607.06065", "title": "SWE-Review: Closing the Loop on Issue Resolution with Agentic Code Review", "url": "https://arxiv.org/abs/2607.06065", "published": "2026-07-07", "updated": "2026-07-07", "authors": [ "Ruoyu Wang", "Jierun Chen", "Shaowei Wang", "Chaofan Tao", "Sidi Yang", "Yuxin Jiang", "Kim-Hui Yap", "Lifeng Shang", "Xiaohui Li", "Haoli Bai" ], "categories": [ "cs.SE" ], "topics": [ "agent-evaluation", "coding-agent" ], "score": 11, "relevance": "medium", "matched_queries": [ "coding-agent" ], "arxiv_id": "2607.06065", "source": "arxiv", "source_id": "arxiv:2607.06065", "pdf_url": "https://arxiv.org/pdf/2607.06065", "primary_query": "coding-agent" }, { "id": "2607.05785", "title": "Can Large Language Models Generate Observability-Aware Code?", "url": "https://arxiv.org/abs/2607.05785", "published": "2026-07-07", "updated": "2026-07-07", "authors": [ "Yongliang Tao", "Hongyu Zhang", "Pengfei Gao", "Minghua Ma", "Zhiyu Fan", "Yu Kang", "Jue Zhang", "Si Qin", "Liqun Li", "Qingwei Lin", "Saravan Rajmohan" ], "categories": [ "cs.SE" ], "topics": [ "agent-evaluation", "coding-agent", "computer-use", "tool-use" ], "score": 11, "relevance": "medium", "matched_queries": [ "coding-agent" ], "arxiv_id": "2607.05785", "source": "arxiv", "source_id": "arxiv:2607.05785", "pdf_url": "https://arxiv.org/pdf/2607.05785", "primary_query": "coding-agent" }, { "id": "2607.05682", "title": "FirstResearch: Auditable Question Formation for LLM Scientific Discovery Agents", "url": "https://arxiv.org/abs/2607.05682", "published": "2026-07-06", "updated": "2026-07-06", "authors": [ "Yufeng Wang" ], "categories": [ "cs.AI" ], "topics": [ "planning", "rag" ], "score": 11, "relevance": "medium", "matched_queries": [ "llm-agent", "planning-agent" ], "arxiv_id": "2607.05682", "source": "arxiv", "source_id": "arxiv:2607.05682", "pdf_url": "https://arxiv.org/pdf/2607.05682", "primary_query": "llm-agent" }, { "id": "2607.04576", "title": "Progressive Disclosure for LLM-Maintained Wiki Knowledge Bases: a Preregistered Ablation", "url": "https://arxiv.org/abs/2607.04576", "published": "2026-07-06", "updated": "2026-07-06", "authors": [ "Theodore O. Cochran" ], "categories": [ "cs.CL", "cs.CY", "cs.IR" ], "topics": [ "agent-evaluation", "rag", "tool-use" ], "score": 11, "relevance": "medium", "matched_queries": [ "llm-agent", "tool-use" ], "arxiv_id": "2607.04576", "source": "arxiv", "source_id": "arxiv:2607.04576", "pdf_url": "https://arxiv.org/pdf/2607.04576", "primary_query": "llm-agent" }, { "id": "2607.04558", "title": "EEG-SpikeAgent: Agentic Closed-Loop Program Synthesis for Automated EEG Spike Detection", "url": "https://arxiv.org/abs/2607.04558", "published": "2026-07-06", "updated": "2026-07-06", "authors": [ "Sonali Santhosh", "Kelly Shuhong Yu", "Eugene Chang", "Jonathan Kim", "Kie Shidara", "Danilo Bernardo" ], "categories": [ "cs.CL", "cs.AI", "cs.LG" ], "topics": [ "agent-evaluation" ], "score": 11, "relevance": "medium", "matched_queries": [ "llm-agent" ], "arxiv_id": "2607.04558", "source": "arxiv", "source_id": "arxiv:2607.04558", "pdf_url": "https://arxiv.org/pdf/2607.04558", "primary_query": "llm-agent" }, { "id": "2607.05593", "title": "Collective Cognition in Hybrid Groups: A Network Science Synthesis", "url": "https://arxiv.org/abs/2607.05593", "published": "2026-07-06", "updated": "2026-07-06", "authors": [ "Babak Hemmatian", "Razan Baltaji", "Lav R. Varshney" ], "categories": [ "cs.HC" ], "topics": [ "memory", "multi-agent", "reasoning", "tool-use" ], "score": 11, "relevance": "medium", "matched_queries": [ "agentic-ai", "ai-agent" ], "arxiv_id": "2607.05593", "source": "arxiv", "source_id": "arxiv:2607.05593", "pdf_url": "https://arxiv.org/pdf/2607.05593", "primary_query": "agentic-ai" }, { "id": "2607.04948", "title": "Using Process Mining to Generate AI Agents from Software Engineering Process Records", "url": "https://arxiv.org/abs/2607.04948", "published": "2026-07-06", "updated": "2026-07-06", "authors": [ "Saimir Bala", "Fabiana Fournier", "Lior Limonad", "Andreas Metzger" ], "categories": [ "cs.SE" ], "topics": [ "coding-agent", "multi-agent" ], "score": 11, "relevance": "medium", "matched_queries": [ "ai-agent" ], "arxiv_id": "2607.04948", "source": "arxiv", "source_id": "arxiv:2607.04948", "pdf_url": "https://arxiv.org/pdf/2607.04948", "primary_query": "ai-agent" }, { "id": "2607.05462", "title": "Evaluating calibrated refusal and safe usefulness in dual-use biology settings", "url": "https://arxiv.org/abs/2607.05462", "published": "2026-07-06", "updated": "2026-07-06", "authors": [ "Edwin H. Wintermute", "Harmon Bhasin", "Christina M. Agapakis", "Dianzhuo Wang", "Evan Seeyave", "Arjun Banerjee", "Daniel Fulop", "Matthew C. Watson", "Adam J. Meyer", "Sandrine Boissel", "Jens H. Kuhn", "Rishi Jain", "Noah D. Taylor", "Helena Shomar", "Patrick M. Boyle", "Kenny Workman" ], "categories": [ "cs.CR", "cs.AI" ], "topics": [ "agent-evaluation", "agent-safety", "reasoning", "tool-use", "workflow-agent" ], "score": 11, "relevance": "medium", "matched_queries": [ "ai-agent" ], "arxiv_id": "2607.05462", "source": "arxiv", "source_id": "arxiv:2607.05462", "pdf_url": "https://arxiv.org/pdf/2607.05462", "primary_query": "ai-agent" }, { "id": "2607.05483", "title": "PatchOptic for Shared-State LLM Workflows with Projected Views and Verified Structured Updates", "url": "https://arxiv.org/abs/2607.05483", "published": "2026-07-06", "updated": "2026-07-06", "authors": [ "Zhaoyu Bai", "Jiaqi Cai" ], "categories": [ "cs.LG", "cs.AI", "cs.LO", "cs.MA", "cs.PL" ], "topics": [ "agent-evaluation", "rag", "workflow-agent" ], "score": 11, "relevance": "medium", "matched_queries": [ "agentic-ai", "rag-agent" ], "arxiv_id": "2607.05483", "source": "arxiv", "source_id": "arxiv:2607.05483", "pdf_url": "https://arxiv.org/pdf/2607.05483", "primary_query": "agentic-ai" }, { "id": "2607.05139", "title": "On the risk of coding before testing: An empirical study on LLM-based test generation workflow", "url": "https://arxiv.org/abs/2607.05139", "published": "2026-07-06", "updated": "2026-07-06", "authors": [ "Michael Konstantinou", "Florian Tambon", "Mike Papadakis" ], "categories": [ "cs.SE" ], "topics": [ "agent-evaluation", "agent-safety", "coding-agent", "reasoning", "workflow-agent" ], "score": 11, "relevance": "medium", "matched_queries": [ "agentic-ai" ], "arxiv_id": "2607.05139", "source": "arxiv", "source_id": "arxiv:2607.05139", "pdf_url": "https://arxiv.org/pdf/2607.05139", "primary_query": "agentic-ai" }, { "id": "2607.04096", "title": "Forethought: Verifiable Reasoning from Neurosymbolic Primitive Programming", "url": "https://arxiv.org/abs/2607.04096", "published": "2026-07-05", "updated": "2026-07-05", "authors": [ "Vishvesh Bhat", "Jay Vaghasiya", "Emmanuel Anaya Gonzalez" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "reasoning", "tool-use", "workflow-agent" ], "score": 11, "relevance": "medium", "matched_queries": [ "agentic-ai" ], "arxiv_id": "2607.04096", "source": "arxiv", "source_id": "arxiv:2607.04096", "pdf_url": "https://arxiv.org/pdf/2607.04096", "primary_query": "agentic-ai" }, { "id": "2607.04425", "title": "UI-MOPD: Multi-Platform On-Policy Distillation for Continual GUI Agent Learning", "url": "https://arxiv.org/abs/2607.04425", "published": "2026-07-05", "updated": "2026-07-05", "authors": [ "Niu Lian", "Alan Chen", "Zhehao Yu", "Chengzhen Duan", "Fazhan Liu", "Hui Liu", "Pei Fu", "Jian Luan", "Yaowei Wang", "Shu-Tao Xia", "Jinpeng Wang" ], "categories": [ "cs.CL", "cs.AI", "cs.CV", "cs.LG", "cs.MM" ], "topics": [ "computer-use", "rag", "tool-use" ], "score": 11, "relevance": "medium", "matched_queries": [ "web-gui-agent" ], "arxiv_id": "2607.04425", "source": "arxiv", "source_id": "arxiv:2607.04425", "pdf_url": "https://arxiv.org/pdf/2607.04425", "primary_query": "web-gui-agent" }, { "id": "2607.03651", "title": "LLM-Guided Transportation Hub Capacity Planning with Textual Business Inputs", "url": "https://arxiv.org/abs/2607.03651", "published": "2026-07-04", "updated": "2026-07-04", "authors": [ "Xiaoyue Liu", "Zheng Dong" ], "categories": [ "cs.LG", "math.OC" ], "topics": [ "agent-evaluation", "computer-use", "planning", "reasoning", "workflow-agent" ], "score": 11, "relevance": "medium", "matched_queries": [ "llm-agent", "planning-agent" ], "arxiv_id": "2607.03651", "source": "arxiv", "source_id": "arxiv:2607.03651", "pdf_url": "https://arxiv.org/pdf/2607.03651", "primary_query": "llm-agent" }, { "id": "2607.02975", "title": "Evaluating Generative Agents with Actions Grounded in Socially Distributed Task Environments using Incognita", "url": "https://arxiv.org/abs/2607.02975", "published": "2026-07-03", "updated": "2026-07-03", "authors": [ "Dan C. Hsu", "Luke Lu" ], "categories": [ "cs.AI", "cs.CL" ], "topics": [ "agent-evaluation", "tool-use", "world-model" ], "score": 11, "relevance": "medium", "matched_queries": [ "language-agent" ], "arxiv_id": "2607.02975", "source": "arxiv", "source_id": "arxiv:2607.02975", "pdf_url": "https://arxiv.org/pdf/2607.02975", "primary_query": "language-agent" }, { "id": "2607.03025", "title": "Human-Centric Reflective Architecture for Human-AI Collaborative Decision-Making", "url": "https://arxiv.org/abs/2607.03025", "published": "2026-07-03", "updated": "2026-07-03", "authors": [ "Andreas Kouridakis", "Dimitrios Patiniotis Spyropoulos", "George Vouros" ], "categories": [ "cs.AI", "cs.MA" ], "topics": [ "agent-evaluation", "agent-safety", "computer-use", "rag" ], "score": 11, "relevance": "medium", "matched_queries": [ "ai-agent" ], "arxiv_id": "2607.03025", "source": "arxiv", "source_id": "arxiv:2607.03025", "pdf_url": "https://arxiv.org/pdf/2607.03025", "primary_query": "ai-agent" }, { "id": "2607.03386", "title": "When Aggregate Alignment Misleads: Auditing Policy Repair Without Per-State Expert Actions", "url": "https://arxiv.org/abs/2607.03386", "published": "2026-07-03", "updated": "2026-07-03", "authors": [ "Peiying Zhu", "Sidi Chang" ], "categories": [ "cs.AI", "cs.LG" ], "topics": [ "agent-evaluation", "agent-safety", "tool-use" ], "score": 11, "relevance": "medium", "matched_queries": [ "agentic-ai" ], "arxiv_id": "2607.03386", "source": "arxiv", "source_id": "arxiv:2607.03386", "pdf_url": "https://arxiv.org/pdf/2607.03386", "primary_query": "agentic-ai" }, { "id": "2607.03332", "title": "AtomicCommitBench: Can Coding Agents Reconstruct Commit Histories from Squashed Patches?", "url": "https://arxiv.org/abs/2607.03332", "published": "2026-07-03", "updated": "2026-07-03", "authors": [ "Zhihao Lin", "Mingyi Zhou", "Li Li" ], "categories": [ "cs.SE" ], "topics": [ "agent-evaluation", "coding-agent" ], "score": 11, "relevance": "medium", "matched_queries": [ "coding-agent" ], "arxiv_id": "2607.03332", "source": "arxiv", "source_id": "arxiv:2607.03332", "pdf_url": "https://arxiv.org/pdf/2607.03332", "primary_query": "coding-agent" }, { "id": "2607.01810", "title": "Decoupling Code Complexity from Newcomer Participation: A Causal Study of AI Coding Agent Adoption in OSS", "url": "https://arxiv.org/abs/2607.01810", "published": "2026-07-02", "updated": "2026-07-02", "authors": [ "Weiwei Xu", "Xuanning Cui", "Hengzhi Ye", "Minghui Zhou" ], "categories": [ "cs.SE", "cs.AI" ], "topics": [ "agent-evaluation", "coding-agent", "tool-use" ], "score": 11, "relevance": "medium", "matched_queries": [ "coding-agent" ], "arxiv_id": "2607.01810", "source": "arxiv", "source_id": "arxiv:2607.01810", "pdf_url": "https://arxiv.org/pdf/2607.01810", "primary_query": "coding-agent" }, { "id": "2607.01760", "title": "Refploit: Facilitating Exploit Construction via Code-Agent Trajectory Repair", "url": "https://arxiv.org/abs/2607.01760", "published": "2026-07-02", "updated": "2026-07-02", "authors": [ "Zirui Chen", "Zhipeng Xue", "Jiayuan Zhou", "Xing Hu", "Xin Xia", "Xiaohu Yang" ], "categories": [ "cs.SE" ], "topics": [ "agent-evaluation", "coding-agent", "computer-use", "tool-use" ], "score": 11, "relevance": "medium", "matched_queries": [ "coding-agent" ], "arxiv_id": "2607.01760", "source": "arxiv", "source_id": "arxiv:2607.01760", "pdf_url": "https://arxiv.org/pdf/2607.01760", "primary_query": "coding-agent" }, { "id": "2607.01418", "title": "Adoption and Impact of Command-Line AI Coding Agents: A Study of Microsoft's Early 2026 Rollout of Claude Code and GitHub Copilot CLI", "url": "https://arxiv.org/abs/2607.01418", "published": "2026-07-01", "updated": "2026-07-01", "authors": [ "Emerson Murphy-Hill", "Jenna Butler", "Alexandra Savelieva" ], "categories": [ "cs.SE", "cs.AI", "cs.HC" ], "topics": [ "coding-agent", "tool-use" ], "score": 11, "relevance": "medium", "matched_queries": [ "coding-agent" ], "arxiv_id": "2607.01418", "source": "arxiv", "source_id": "arxiv:2607.01418", "pdf_url": "https://arxiv.org/pdf/2607.01418", "primary_query": "coding-agent" }, { "id": "2607.01087", "title": "Cheap Code, Costly Judgment: A Case Study on Governable Agentic Software Engineering", "url": "https://arxiv.org/abs/2607.01087", "published": "2026-07-01", "updated": "2026-07-04", "authors": [ "James C. Davis", "Paschal C. Amusuo", "Tanmay Singla", "Berk Çakar", "Kirsten A. Davis" ], "categories": [ "cs.SE", "cs.AI" ], "topics": [ "coding-agent", "tool-use" ], "score": 11, "relevance": "medium", "matched_queries": [ "coding-agent" ], "arxiv_id": "2607.01087", "source": "arxiv", "source_id": "arxiv:2607.01087", "pdf_url": "https://arxiv.org/pdf/2607.01087", "primary_query": "coding-agent" }, { "id": "2607.00895", "title": "Beyond Document Grounding: Span-Level Hallucination Detection over Code, Tool Output, and Documents", "url": "https://arxiv.org/abs/2607.00895", "published": "2026-07-01", "updated": "2026-07-01", "authors": [ "Ádám Kovács", "Bowei He", "Xue Liu", "István Boros", "Szilveszter Tóth", "Gábor Recski" ], "categories": [ "cs.CL" ], "topics": [ "agent-evaluation", "coding-agent", "rag", "tool-use" ], "score": 11, "relevance": "medium", "matched_queries": [ "coding-agent", "rag-agent" ], "arxiv_id": "2607.00895", "source": "arxiv", "source_id": "arxiv:2607.00895", "pdf_url": "https://arxiv.org/pdf/2607.00895", "primary_query": "coding-agent" }, { "id": "2607.01063", "title": "AutoRestTest at the SBFT 2026 Tool Competition", "url": "https://arxiv.org/abs/2607.01063", "published": "2026-07-01", "updated": "2026-07-01", "authors": [ "Tyler Stennett", "Myeongsoo Kim", "Saurabh Sinha", "Alessandro Orso" ], "categories": [ "cs.SE" ], "topics": [ "agent-evaluation", "multi-agent", "rag", "tool-use" ], "score": 11, "relevance": "medium", "matched_queries": [ "multi-agent-llm" ], "arxiv_id": "2607.01063", "source": "arxiv", "source_id": "arxiv:2607.01063", "pdf_url": "https://arxiv.org/pdf/2607.01063", "primary_query": "multi-agent-llm" }, { "id": "2606.31371", "title": "Calibrating the Evaluator: Does Probability Calibration Mitigate Preference Coupling in LLM Agent Feedback Loops?", "url": "https://arxiv.org/abs/2606.31371", "published": "2026-06-30", "updated": "2026-06-30", "authors": [ "Zewen Liu" ], "categories": [ "cs.LG", "cs.AI", "cs.CL" ], "topics": [ "agent-evaluation" ], "score": 11, "relevance": "medium", "matched_queries": [ "llm-agent" ], "arxiv_id": "2606.31371", "source": "arxiv", "source_id": "arxiv:2606.31371", "pdf_url": "https://arxiv.org/pdf/2606.31371", "primary_query": "llm-agent" }, { "id": "2606.31182", "title": "AI-Assisted Discovery of Convex Relaxations via Dual Agents", "url": "https://arxiv.org/abs/2606.31182", "published": "2026-06-30", "updated": "2026-06-30", "authors": [ "Sungyoon Kim", "Mert Pilanci" ], "categories": [ "cs.AI" ], "topics": [ "coding-agent" ], "score": 11, "relevance": "medium", "matched_queries": [ "coding-agent", "llm-agent" ], "arxiv_id": "2606.31182", "source": "arxiv", "source_id": "arxiv:2606.31182", "pdf_url": "https://arxiv.org/pdf/2606.31182", "primary_query": "coding-agent" }, { "id": "2606.31935", "title": "Delegation Rights: Property, Agency, and Investment Incentives in the Age of AI Agents", "url": "https://arxiv.org/abs/2606.31935", "published": "2026-06-30", "updated": "2026-06-30", "authors": [ "Yukun Zhang", "Kemu Xu" ], "categories": [ "econ.EM" ], "topics": [ "agent-safety", "tool-use", "world-model" ], "score": 11, "relevance": "medium", "matched_queries": [ "ai-agent" ], "arxiv_id": "2606.31935", "source": "arxiv", "source_id": "arxiv:2606.31935", "pdf_url": "https://arxiv.org/pdf/2606.31935", "primary_query": "ai-agent" }, { "id": "2606.31041", "title": "A Semantic-Layer-Mediated Agent for Natural Language to SQL over Heterogeneous Enterprise Databases", "url": "https://arxiv.org/abs/2606.31041", "published": "2026-06-30", "updated": "2026-06-30", "authors": [ "Ha Jeong Kim", "Saksonita Khoeurn", "Ye Ji Yoon" ], "categories": [ "cs.CL" ], "topics": [ "agent-evaluation", "rag", "reasoning", "workflow-agent" ], "score": 11, "relevance": "medium", "matched_queries": [ "agentic-ai" ], "arxiv_id": "2606.31041", "source": "arxiv", "source_id": "arxiv:2606.31041", "pdf_url": "https://arxiv.org/pdf/2606.31041", "primary_query": "agentic-ai" }, { "id": "2606.31270", "title": "Learning from Failure: Inference-Time Self-Improvement for Computer-Use Agents", "url": "https://arxiv.org/abs/2606.31270", "published": "2026-06-30", "updated": "2026-06-30", "authors": [ "Xueqiao Sun", "Xiaohan Wang", "Ludwig Schmidt", "Serena Yeung-Levy", "Yuhui Zhang" ], "categories": [ "cs.CV", "cs.AI", "cs.CL", "cs.CY", "cs.LG" ], "topics": [ "agent-evaluation", "rag", "reasoning" ], "score": 11, "relevance": "medium", "matched_queries": [ "web-gui-agent" ], "arxiv_id": "2606.31270", "source": "arxiv", "source_id": "arxiv:2606.31270", "pdf_url": "https://arxiv.org/pdf/2606.31270", "primary_query": "web-gui-agent" }, { "id": "2606.31154", "title": "PPT-Eval: A Benchmark for Computer-Use Agents on PowerPoint Tasks", "url": "https://arxiv.org/abs/2606.31154", "published": "2026-06-30", "updated": "2026-06-30", "authors": [ "Apurva Gandhi", "Vishwas Suryanarayanan", "Raja Hasnain Anwar", "Firoz Shaik", "Shubhang Desai", "Thong Q. Nguyen", "Muhammad Taqi Raza", "Vishal Chowdhary", "Graham Neubig" ], "categories": [ "cs.LG", "cs.AI" ], "topics": [ "agent-evaluation", "rag" ], "score": 11, "relevance": "medium", "matched_queries": [ "web-gui-agent" ], "arxiv_id": "2606.31154", "source": "arxiv", "source_id": "arxiv:2606.31154", "pdf_url": "https://arxiv.org/pdf/2606.31154", "primary_query": "web-gui-agent" }, { "id": "2606.31273", "title": "The Calibration Turn in AI-Assisted Research: A Conceptual and Methodological Framework for Evidence-Licensed Claims", "url": "https://arxiv.org/abs/2606.31273", "published": "2026-06-30", "updated": "2026-06-30", "authors": [ "Hongmin Li" ], "categories": [ "cs.LG" ], "topics": [ "agent-evaluation", "computer-use", "multi-agent", "workflow-agent" ], "score": 11, "relevance": "medium", "matched_queries": [ "multi-agent-llm" ], "arxiv_id": "2606.31273", "source": "arxiv", "source_id": "arxiv:2606.31273", "pdf_url": "https://arxiv.org/pdf/2606.31273", "primary_query": "multi-agent-llm" }, { "id": "2606.29717", "title": "Optimizing Expert-Designed Crystal Graph Networks for Band-Gap Prediction with an Autonomous LLM Research Loop", "url": "https://arxiv.org/abs/2606.29717", "published": "2026-06-29", "updated": "2026-06-29", "authors": [ "Chenmu Zhang", "Boris I. Yakobson" ], "categories": [ "cond-mat.mtrl-sci", "cs.AI", "cs.LG" ], "topics": [ "agent-evaluation", "coding-agent" ], "score": 11, "relevance": "medium", "matched_queries": [ "autonomous-agent-llm", "coding-agent", "llm-agent" ], "arxiv_id": "2606.29717", "source": "arxiv", "source_id": "arxiv:2606.29717", "pdf_url": "https://arxiv.org/pdf/2606.29717", "primary_query": "autonomous-agent-llm" }, { "id": "2606.29705", "title": "GUICrafter: Weakly-Supervised GUI Agent Leveraging Massive Unannotated Screenshots", "url": "https://arxiv.org/abs/2606.29705", "published": "2026-06-29", "updated": "2026-06-29", "authors": [ "Sunqi Fan", "Lingshan Chen", "Runqi Yin", "Qingle Liu", "Yongming Rao", "Meng-Hao Guo", "Shi-Min Hu" ], "categories": [ "cs.AI", "cs.CL", "cs.CV" ], "topics": [ "computer-use", "rag", "tool-use" ], "score": 11, "relevance": "medium", "matched_queries": [ "web-gui-agent" ], "arxiv_id": "2606.29705", "source": "arxiv", "source_id": "arxiv:2606.29705", "pdf_url": "https://arxiv.org/pdf/2606.29705", "primary_query": "web-gui-agent" }, { "id": "2606.30571", "title": "Attractor States Emerge in Multi-Turn LLM Conversations", "url": "https://arxiv.org/abs/2606.30571", "published": "2026-06-29", "updated": "2026-06-29", "authors": [ "Ting-Wen Ko", "Jonas Geiping" ], "categories": [ "cs.LG", "cs.CL" ], "topics": [ "agent-evaluation", "multi-agent", "tool-use" ], "score": 11, "relevance": "medium", "matched_queries": [ "autonomous-agent-llm", "multi-agent-llm" ], "arxiv_id": "2606.30571", "source": "arxiv", "source_id": "arxiv:2606.30571", "pdf_url": "https://arxiv.org/pdf/2606.30571", "primary_query": "autonomous-agent-llm" }, { "id": "2606.29981", "title": "Hephaestus: Toward a Cybersecurity AI Scientist", "url": "https://arxiv.org/abs/2606.29981", "published": "2026-06-29", "updated": "2026-06-29", "authors": [ "Jiaqi Li", "Yang Zhao", "Wen Lu", "Lvyang Zhang", "Lidong Zhai" ], "categories": [ "cs.CR" ], "topics": [ "agent-evaluation", "agent-safety", "multi-agent", "tool-use", "workflow-agent" ], "score": 11, "relevance": "medium", "matched_queries": [ "agent-safety" ], "arxiv_id": "2606.29981", "source": "arxiv", "source_id": "arxiv:2606.29981", "pdf_url": "https://arxiv.org/pdf/2606.29981", "primary_query": "agent-safety" }, { "id": "2606.29406", "title": "Adaptive AI Delegation under Uncertainty: A Bayesian Governance Policy for Sequential Decision Authority", "url": "https://arxiv.org/abs/2606.29406", "published": "2026-06-28", "updated": "2026-06-28", "authors": [ "Matthew Francis Dixon" ], "categories": [ "q-fin.RM", "math.OC" ], "topics": [ "agent-evaluation", "computer-use", "rag", "tool-use" ], "score": 11, "relevance": "medium", "matched_queries": [ "agentic-ai" ], "arxiv_id": "2606.29406", "source": "arxiv", "source_id": "arxiv:2606.29406", "pdf_url": "https://arxiv.org/pdf/2606.29406", "primary_query": "agentic-ai" }, { "id": "2606.29648", "title": "Hybrid Retriever Evolution for Multimodal Document Reasoning Agents", "url": "https://arxiv.org/abs/2606.29648", "published": "2026-06-28", "updated": "2026-06-28", "authors": [ "Bohan Yao", "Shruthan Radhakrishna", "Vikas Yadav" ], "categories": [ "cs.CL", "cs.AI", "cs.LG", "cs.MA" ], "topics": [ "agent-evaluation", "multi-agent", "rag", "reasoning", "tool-use" ], "score": 11, "relevance": "medium", "matched_queries": [ "tool-use" ], "arxiv_id": "2606.29648", "source": "arxiv", "source_id": "arxiv:2606.29648", "pdf_url": "https://arxiv.org/pdf/2606.29648", "primary_query": "tool-use" }, { "id": "2606.29472", "title": "Agent-Computer Observation Interfaces Enable Dynamic Computer Use", "url": "https://arxiv.org/abs/2606.29472", "published": "2026-06-28", "updated": "2026-06-28", "authors": [ "Bojie Li", "Noah Shi" ], "categories": [ "cs.AI" ], "topics": [ "computer-use", "tool-use" ], "score": 11, "relevance": "medium", "matched_queries": [ "coding-agent", "web-gui-agent" ], "arxiv_id": "2606.29472", "source": "arxiv", "source_id": "arxiv:2606.29472", "pdf_url": "https://arxiv.org/pdf/2606.29472", "primary_query": "coding-agent" }, { "id": "2606.29605", "title": "How much of an LLM-generated clinical corpus is actually new? A production-scale measurement of content redundancy for provenance classification", "url": "https://arxiv.org/abs/2606.29605", "published": "2026-06-28", "updated": "2026-06-28", "authors": [ "Ali H. Lazem", "William J. Teahan" ], "categories": [ "cs.CL" ], "topics": [ "agent-evaluation", "multi-agent", "rag", "tool-use" ], "score": 11, "relevance": "medium", "matched_queries": [ "multi-agent-llm" ], "arxiv_id": "2606.29605", "source": "arxiv", "source_id": "arxiv:2606.29605", "pdf_url": "https://arxiv.org/pdf/2606.29605", "primary_query": "multi-agent-llm" }, { "id": "2606.27650", "title": "GenWorld: Empirically Grounded Urban Simulation Infrastructure for Scalable LLM-Agent Studies", "url": "https://arxiv.org/abs/2606.27650", "published": "2026-06-26", "updated": "2026-06-26", "authors": [ "Gen Li", "Jieyuan Lan", "Pengcheng Xu", "Zongyuan Wu", "Masaki Ogura", "Tao Feng" ], "categories": [ "cs.MA" ], "topics": [ "computer-use", "planning", "rag", "world-model" ], "score": 11, "relevance": "medium", "matched_queries": [ "llm-agent" ], "arxiv_id": "2606.27650", "source": "arxiv", "source_id": "arxiv:2606.27650", "pdf_url": "https://arxiv.org/pdf/2606.27650", "primary_query": "llm-agent" }, { "id": "2606.28235", "title": "Govern the Repository, Not the Agent: Measuring Ecosystem-Level Risk in AI-Native Software", "url": "https://arxiv.org/abs/2606.28235", "published": "2026-06-26", "updated": "2026-06-26", "authors": [ "Daniel Russo" ], "categories": [ "cs.SE", "cs.AI" ], "topics": [ "agent-evaluation", "agent-safety", "coding-agent" ], "score": 11, "relevance": "medium", "matched_queries": [ "agentic-ai", "coding-agent" ], "arxiv_id": "2606.28235", "source": "arxiv", "source_id": "arxiv:2606.28235", "pdf_url": "https://arxiv.org/pdf/2606.28235", "primary_query": "agentic-ai" }, { "id": "2606.27909", "title": "Triadic Werewolf: A Jester Role for Multi-Hop Theory of Mind in LLMs", "url": "https://arxiv.org/abs/2606.27909", "published": "2026-06-26", "updated": "2026-06-26", "authors": [ "Avni Mittal" ], "categories": [ "cs.CL", "cs.AI", "cs.GT", "cs.MA" ], "topics": [ "agent-evaluation", "multi-agent", "reasoning", "tool-use" ], "score": 11, "relevance": "medium", "matched_queries": [ "multi-agent-llm" ], "arxiv_id": "2606.27909", "source": "arxiv", "source_id": "arxiv:2606.27909", "pdf_url": "https://arxiv.org/pdf/2606.27909", "primary_query": "multi-agent-llm" }, { "id": "2606.28002", "title": "Dialogue to Detection: A Multimodal Hybrid NLP Pipeline for Insurance Fraud Detection", "url": "https://arxiv.org/abs/2606.28002", "published": "2026-06-26", "updated": "2026-06-26", "authors": [ "Muhammad Shakeel Akram", "Amal Htait", "Abdul Hamid Sadka", "Emma Meisingseth", "Karishma Jaitly" ], "categories": [ "cs.CL", "cs.AI", "eess.AS" ], "topics": [ "agent-evaluation", "agent-safety", "computer-use", "rag", "tool-use" ], "score": 11, "relevance": "medium", "matched_queries": [ "rag-agent" ], "arxiv_id": "2606.28002", "source": "arxiv", "source_id": "arxiv:2606.28002", "pdf_url": "https://arxiv.org/pdf/2606.28002", "primary_query": "rag-agent" }, { "id": "2606.26722", "title": "Socratic agents for autonomous scientific discovery in high-dimensional physical systems", "url": "https://arxiv.org/abs/2606.26722", "published": "2026-06-25", "updated": "2026-06-25", "authors": [ "Xianrui Zeng", "Pengfei Liu", "Yirui Zang", "Yang Shen", "Fei Yu", "Chunlei Yu", "Minghao Liu", "Yang Du" ], "categories": [ "cs.AI", "physics.optics" ], "topics": [ "agent-evaluation", "coding-agent", "multi-agent", "planning", "workflow-agent" ], "score": 11, "relevance": "medium", "matched_queries": [ "agentic-ai" ], "arxiv_id": "2606.26722", "source": "arxiv", "source_id": "arxiv:2606.26722", "pdf_url": "https://arxiv.org/pdf/2606.26722", "primary_query": "agentic-ai" }, { "id": "2606.27595", "title": "Ko-WideSearch: A Korean Breadth-Search Benchmark for Exhaustive Set Enumeration by Web Agents", "url": "https://arxiv.org/abs/2606.27595", "published": "2026-06-25", "updated": "2026-06-25", "authors": [ "Minbyul Jeong" ], "categories": [ "cs.CL" ], "topics": [ "agent-evaluation", "computer-use" ], "score": 11, "relevance": "medium", "matched_queries": [ "agent-evaluation", "web-gui-agent" ], "arxiv_id": "2606.27595", "source": "arxiv", "source_id": "arxiv:2606.27595", "pdf_url": "https://arxiv.org/pdf/2606.27595", "primary_query": "agent-evaluation" }, { "id": "2606.26474", "title": "Localizing RL-Induced Tool Use to a Single Crosscoder Feature", "url": "https://arxiv.org/abs/2606.26474", "published": "2026-06-25", "updated": "2026-06-25", "authors": [ "Andrii Shportko", "Shubham Bhokare", "Ahmed Zeyad A Alzahrani", "Bowen Cheng", "Gustavo Mercier", "Jessica Hullman" ], "categories": [ "cs.LG", "cs.AI" ], "topics": [ "rag", "tool-use" ], "score": 11, "relevance": "medium", "matched_queries": [ "tool-use" ], "arxiv_id": "2606.26474", "source": "arxiv", "source_id": "arxiv:2606.26474", "pdf_url": "https://arxiv.org/pdf/2606.26474", "primary_query": "tool-use" }, { "id": "2606.27122", "title": "Mostly Automatic Translation of Language Interpreters from C to Safe Rust", "url": "https://arxiv.org/abs/2606.27122", "published": "2026-06-25", "updated": "2026-06-25", "authors": [ "Bo Wang", "Brandon Paulsen", "Joey Dodds", "Daniel Kroening", "Umang Mathur", "Prateek Saxena" ], "categories": [ "cs.PL", "cs.MA", "cs.SE" ], "topics": [ "agent-safety", "coding-agent", "memory", "multi-agent", "workflow-agent" ], "score": 11, "relevance": "medium", "matched_queries": [ "coding-agent" ], "arxiv_id": "2606.27122", "source": "arxiv", "source_id": "arxiv:2606.27122", "pdf_url": "https://arxiv.org/pdf/2606.27122", "primary_query": "coding-agent" }, { "id": "2606.26979", "title": "How Much Static Structure Do Code Agents Need? A Study of Deterministic Anchoring", "url": "https://arxiv.org/abs/2606.26979", "published": "2026-06-25", "updated": "2026-07-02", "authors": [ "Zhihao Lin", "Mingyi Zhou", "Yizhuo Yang", "Li Li" ], "categories": [ "cs.SE" ], "topics": [ "coding-agent", "computer-use", "embodied-agent", "tool-use" ], "score": 11, "relevance": "medium", "matched_queries": [ "coding-agent" ], "arxiv_id": "2606.26979", "source": "arxiv", "source_id": "arxiv:2606.26979", "pdf_url": "https://arxiv.org/pdf/2606.26979", "primary_query": "coding-agent" }, { "id": "2606.26298", "title": "Governing Actions, Not Agents: Institutional Attestation as a Governance Model for Autonomous AI Systems", "url": "https://arxiv.org/abs/2606.26298", "published": "2026-06-24", "updated": "2026-06-24", "authors": [ "Jakob Salfeld-Nebgen" ], "categories": [ "cs.AI", "cs.CR" ], "topics": [ "agent-evaluation", "agent-safety", "planning", "reasoning", "tool-use" ], "score": 11, "relevance": "medium", "matched_queries": [ "ai-agent" ], "arxiv_id": "2606.26298", "source": "arxiv", "source_id": "arxiv:2606.26298", "pdf_url": "https://arxiv.org/pdf/2606.26298", "primary_query": "ai-agent" }, { "id": "2606.26027", "title": "Why Multi-Step Tool-Use Reinforcement Learning Collapses and How Supervisory Signals Fix It", "url": "https://arxiv.org/abs/2606.26027", "published": "2026-06-24", "updated": "2026-06-24", "authors": [ "Yupu Hao", "Zhuoran Jin", "Huanxuan Liao", "Kang Liu", "Jun Zhao" ], "categories": [ "cs.CL", "cs.LG" ], "topics": [ "agent-evaluation", "computer-use", "tool-use" ], "score": 11, "relevance": "medium", "matched_queries": [ "tool-use" ], "arxiv_id": "2606.26027", "source": "arxiv", "source_id": "arxiv:2606.26027", "pdf_url": "https://arxiv.org/pdf/2606.26027", "primary_query": "tool-use" }, { "id": "2606.25257", "title": "How Do Developers Maintain and Evolve Their Agents' Instructions? An Empirical Study", "url": "https://arxiv.org/abs/2606.25257", "published": "2026-06-24", "updated": "2026-06-24", "authors": [ "Gianmario Voria", "Alfonso Cannavale", "Andrea De Lucia", "Yutaro Kashiwa", "Gemma Catolino", "Fabio Palomba" ], "categories": [ "cs.SE" ], "topics": [ "agent-evaluation", "coding-agent", "computer-use", "planning", "workflow-agent" ], "score": 11, "relevance": "medium", "matched_queries": [ "coding-agent" ], "arxiv_id": "2606.25257", "source": "arxiv", "source_id": "arxiv:2606.25257", "pdf_url": "https://arxiv.org/pdf/2606.25257", "primary_query": "coding-agent" }, { "id": "2606.24429", "title": "Detecting AI Coding Agents in Open Source: A Validated Multi-Method Census of 180 Million Repositories", "url": "https://arxiv.org/abs/2606.24429", "published": "2026-06-23", "updated": "2026-06-23", "authors": [ "Arsham Khosravani", "Audris Mockus" ], "categories": [ "cs.SE", "cs.AI" ], "topics": [ "agent-evaluation", "coding-agent", "tool-use" ], "score": 11, "relevance": "medium", "matched_queries": [ "coding-agent" ], "arxiv_id": "2606.24429", "source": "arxiv", "source_id": "arxiv:2606.24429", "pdf_url": "https://arxiv.org/pdf/2606.24429", "primary_query": "coding-agent" }, { "id": "2606.24965", "title": "Project Auto-World: Towards Automated Benchmarking of Neural Relational Reasoners", "url": "https://arxiv.org/abs/2606.24965", "published": "2026-06-23", "updated": "2026-06-23", "authors": [ "Anirban Das", "Joanne Boisson", "Irtaza Khalid", "Sumita Garai", "Steven Schockaert" ], "categories": [ "cs.AI", "cs.LG" ], "topics": [ "agent-evaluation", "reasoning" ], "score": 11, "relevance": "medium", "matched_queries": [ "autonomous-agent-llm" ], "arxiv_id": "2606.24965", "source": "arxiv", "source_id": "arxiv:2606.24965", "pdf_url": "https://arxiv.org/pdf/2606.24965", "primary_query": "autonomous-agent-llm" }, { "id": "2606.23449", "title": "AOHP: An Open-Source OS-Level Agent Harness for Personalized, Efficient and Secure Interaction", "url": "https://arxiv.org/abs/2606.23449", "published": "2026-06-22", "updated": "2026-06-22", "authors": [ "Shanhui Zhao", "Jiacheng Liu", "Guohong Liu", "Jichao Yan", "Jialei Ye", "Yuhao Yang", "Hao Wen", "Shizuo Tian", "Yizhen Yuan", "Yuxuan Chen", "Yunxin Liu", "Ju Ren", "Ya-Qin Zhang", "Chao Huang", "Yao Guo", "Yuanchun Li" ], "categories": [ "cs.AI", "cs.OS" ], "topics": [ "agent-safety", "memory", "tool-use", "workflow-agent" ], "score": 11, "relevance": "medium", "matched_queries": [ "ai-agent" ], "arxiv_id": "2606.23449", "source": "arxiv", "source_id": "arxiv:2606.23449", "pdf_url": "https://arxiv.org/pdf/2606.23449", "primary_query": "ai-agent" }, { "id": "2606.23937", "title": "When Retrieval Metrics Mislead: Measuring Policy Signal in Long-Horizon Tool-Use Agents", "url": "https://arxiv.org/abs/2606.23937", "published": "2026-06-22", "updated": "2026-06-22", "authors": [ "Tianyu Ding", "Juan Pablo De la Cruz Weinstein" ], "categories": [ "cs.CL", "cs.AI", "cs.LG" ], "topics": [ "agent-evaluation", "planning", "rag", "tool-use" ], "score": 11, "relevance": "medium", "matched_queries": [ "tool-use" ], "arxiv_id": "2606.23937", "source": "arxiv", "source_id": "arxiv:2606.23937", "pdf_url": "https://arxiv.org/pdf/2606.23937", "primary_query": "tool-use" }, { "id": "2606.23997", "title": "ChartWalker: Benchmarking the Cross-Chart RAG Task with Hierarchical Knowledge Graphs", "url": "https://arxiv.org/abs/2606.23997", "published": "2026-06-22", "updated": "2026-06-22", "authors": [ "Ning Tang", "Chenghan Xie", "Hanyang Yuan", "Yi Li", "Renhong Huang", "Qian Kou", "Xiaofeng Shi", "Hua Zhou", "Jiarong Xu" ], "categories": [ "cs.IR" ], "topics": [ "agent-evaluation", "rag", "reasoning" ], "score": 11, "relevance": "medium", "matched_queries": [ "rag-agent" ], "arxiv_id": "2606.23997", "source": "arxiv", "source_id": "arxiv:2606.23997", "pdf_url": "https://arxiv.org/pdf/2606.23997", "primary_query": "rag-agent" }, { "id": "2606.22376", "title": "Vibe Calibration: Autonomous Bring-up of a 112-Qubit Superconducting Quantum Processor by a Skill-Orchestrating Language Agent", "url": "https://arxiv.org/abs/2606.22376", "published": "2026-06-21", "updated": "2026-06-21", "authors": [ "Huikai Xu", "Jiaxiu Han", "Shigang Ou", "Cheng Ye", "Zisong Shen", "Jing Gao", "Yijia Wang", "Tianrui Che", "Yu Song", "Weiyang Liu", "Lei Wang", "Lin-Feng Zhang", "Pan Zhang", "Hai-Feng Yu" ], "categories": [ "quant-ph" ], "topics": [ "workflow-agent" ], "score": 11, "relevance": "medium", "matched_queries": [ "language-agent" ], "arxiv_id": "2606.22376", "source": "arxiv", "source_id": "arxiv:2606.22376", "pdf_url": "https://arxiv.org/pdf/2606.22376", "primary_query": "language-agent" }, { "id": "2606.22721", "title": "Habituation at the Gate: Rising Approval and Declining Scrutiny in Human Review of AI Agent Code", "url": "https://arxiv.org/abs/2606.22721", "published": "2026-06-21", "updated": "2026-06-21", "authors": [ "Haoran Yu", "Lifei Liu", "Xiaochong Jiang", "Yuwen Jia", "Su Wang", "Pin Qian", "Yihang Chen" ], "categories": [ "cs.SE" ], "topics": [ "coding-agent" ], "score": 11, "relevance": "medium", "matched_queries": [ "ai-agent", "coding-agent" ], "arxiv_id": "2606.22721", "source": "arxiv", "source_id": "arxiv:2606.22721", "pdf_url": "https://arxiv.org/pdf/2606.22721", "primary_query": "ai-agent" }, { "id": "2606.22711", "title": "Beyond Simpson's Paradox: A Cascade of Confounders in AI Agent Pull-Request Co-Authorship", "url": "https://arxiv.org/abs/2606.22711", "published": "2026-06-21", "updated": "2026-06-21", "authors": [ "Haoran Yu", "Xiaochong Jiang", "Lifei Liu", "Su Wang", "Pin Qian", "Yihang Chen" ], "categories": [ "cs.SE", "cs.AI" ], "topics": [ "coding-agent" ], "score": 11, "relevance": "medium", "matched_queries": [ "ai-agent", "coding-agent" ], "arxiv_id": "2606.22711", "source": "arxiv", "source_id": "arxiv:2606.22711", "pdf_url": "https://arxiv.org/pdf/2606.22711", "primary_query": "ai-agent" }, { "id": "2606.21843", "title": "Measuring What Persists: Conditioning Mechanisms and a Geometric Framework for AI Agent Identity", "url": "https://arxiv.org/abs/2606.21843", "published": "2026-06-20", "updated": "2026-06-20", "authors": [ "Andrew Tanner" ], "categories": [ "cs.AI", "cs.CL" ], "topics": [ "agent-evaluation", "agent-safety", "tool-use" ], "score": 11, "relevance": "medium", "matched_queries": [ "ai-agent" ], "arxiv_id": "2606.21843", "source": "arxiv", "source_id": "arxiv:2606.21843", "pdf_url": "https://arxiv.org/pdf/2606.21843", "primary_query": "ai-agent" }, { "id": "2606.21562", "title": "Compressing Observation History into Agent Memory: Distilling Transformers into Recurrent Transformers", "url": "https://arxiv.org/abs/2606.21562", "published": "2026-06-19", "updated": "2026-06-19", "authors": [ "Philippe Weinzaepfel", "Christian Wolf", "Bülent Mert Sariyildiz", "Guillaume Bono", "Gianluca Monaci" ], "categories": [ "cs.CV", "cs.LG" ], "topics": [ "embodied-agent", "memory", "planning" ], "score": 11, "relevance": "medium", "matched_queries": [ "agent-memory" ], "arxiv_id": "2606.21562", "source": "arxiv", "source_id": "arxiv:2606.21562", "pdf_url": "https://arxiv.org/pdf/2606.21562", "primary_query": "agent-memory" }, { "id": "2606.19857", "title": "Large Language Models Do Not Always Need Readable Language", "url": "https://arxiv.org/abs/2606.19857", "published": "2026-06-18", "updated": "2026-06-18", "authors": [ "Jiayi Zhu", "Haoxuan Peng", "Junxi Wang", "Liang Ke", "Chen Zhang", "Linfeng Zhang" ], "categories": [ "cs.CL", "cs.AI" ], "topics": [ "agent-evaluation", "memory", "multi-agent" ], "score": 11, "relevance": "medium", "matched_queries": [ "agent-memory" ], "arxiv_id": "2606.19857", "source": "arxiv", "source_id": "arxiv:2606.19857", "pdf_url": "https://arxiv.org/pdf/2606.19857", "primary_query": "agent-memory" }, { "id": "2606.20910", "title": "Whose Agent Are You? Multi-Layer Fingerprinting and Attribution of Autonomous Web Agents", "url": "https://arxiv.org/abs/2606.20910", "published": "2026-06-18", "updated": "2026-06-18", "authors": [ "Dayeon Kang", "Hyejun Jeong", "Jade Sheffey", "Pubali Datta", "Amir Houmansadr" ], "categories": [ "cs.CR", "cs.AI" ], "topics": [ "agent-safety", "computer-use", "embodied-agent", "tool-use" ], "score": 11, "relevance": "medium", "matched_queries": [ "web-gui-agent" ], "arxiv_id": "2606.20910", "source": "arxiv", "source_id": "arxiv:2606.20910", "pdf_url": "https://arxiv.org/pdf/2606.20910", "primary_query": "web-gui-agent" }, { "id": "2606.20487", "title": "Beyond Global Replanning: Hierarchical Recovery for Cross-Device Agent Systems", "url": "https://arxiv.org/abs/2606.20487", "published": "2026-06-18", "updated": "2026-06-18", "authors": [ "Shu Yao", "Yuhua Luo", "Qian Long", "Jingru Fan", "Zhuoyuan Yu", "Yuheng Wang", "Lin Wu", "Yufan Dang", "Huatao Li", "Chen Qian" ], "categories": [ "cs.CL" ], "topics": [ "agent-evaluation", "computer-use", "planning", "tool-use", "workflow-agent" ], "score": 11, "relevance": "medium", "matched_queries": [ "web-gui-agent" ], "arxiv_id": "2606.20487", "source": "arxiv", "source_id": "arxiv:2606.20487", "pdf_url": "https://arxiv.org/pdf/2606.20487", "primary_query": "web-gui-agent" }, { "id": "2606.20746", "title": "Amplify, Don't Create: Temporal Accumulation for Slow-Burn Prompt Injection", "url": "https://arxiv.org/abs/2606.20746", "published": "2026-06-17", "updated": "2026-06-17", "authors": [ "J Alex Corll" ], "categories": [ "cs.CR", "cs.LO" ], "topics": [ "agent-evaluation", "agent-safety", "computer-use", "planning", "tool-use" ], "score": 11, "relevance": "medium", "matched_queries": [ "agent-evaluation", "tool-use" ], "arxiv_id": "2606.20746", "source": "arxiv", "source_id": "arxiv:2606.20746", "pdf_url": "https://arxiv.org/pdf/2606.20746", "primary_query": "agent-evaluation" }, { "id": "2606.18613", "title": "Are LLMs Ready to Assist Physicians? PhysAssistBench for Interactive Doctor-Patient-EHR Assistance", "url": "https://arxiv.org/abs/2606.18613", "published": "2026-06-17", "updated": "2026-06-18", "authors": [ "Tianming Du", "Peijie Yu", "Sihan Shang", "Danli Shi", "My Linh Nguyen", "Shengbo Gao", "Guangyuan Li", "Yinghong Yu", "Yan Jiang", "Qianlong Zhao", "Behzad Bozorgtabar", "Shaoxiong Ji", "Jiazhen Pan", "Daniel Rueckert", "Jiancheng Yang" ], "categories": [ "cs.CL", "cs.AI" ], "topics": [ "agent-evaluation", "tool-use" ], "score": 11, "relevance": "medium", "matched_queries": [ "tool-use" ], "arxiv_id": "2606.18613", "source": "arxiv", "source_id": "arxiv:2606.18613", "pdf_url": "https://arxiv.org/pdf/2606.18613", "primary_query": "tool-use" }, { "id": "2606.19390", "title": "Execution-bound advisory automation for agentic AI: a reproducible AIBOM-driven CSAF-VEX framework", "url": "https://arxiv.org/abs/2606.19390", "published": "2026-06-16", "updated": "2026-06-16", "authors": [ "Petar Radanliev", "Omar Santos", "Carsten Maple", "Kay Atefi" ], "categories": [ "cs.SE", "cs.AI" ], "topics": [ "agent-evaluation", "workflow-agent" ], "score": 11, "relevance": "medium", "matched_queries": [ "agentic-ai" ], "arxiv_id": "2606.19390", "source": "arxiv", "source_id": "arxiv:2606.19390", "pdf_url": "https://arxiv.org/pdf/2606.19390", "primary_query": "agentic-ai" }, { "id": "2606.17454", "title": "Dissecting model behavior through agent trajectories", "url": "https://arxiv.org/abs/2606.17454", "published": "2026-06-16", "updated": "2026-06-17", "authors": [ "Gaurav Gupta", "Vatshank Chaturvedi", "Jun Huan", "Anoop Deoras" ], "categories": [ "cs.AI", "cs.LG" ], "topics": [ "agent-evaluation", "agent-safety", "tool-use" ], "score": 11, "relevance": "medium", "matched_queries": [ "agent-evaluation" ], "arxiv_id": "2606.17454", "source": "arxiv", "source_id": "arxiv:2606.17454", "pdf_url": "https://arxiv.org/pdf/2606.17454", "primary_query": "agent-evaluation" }, { "id": "2606.17929", "title": "PreAct: Computer-Using Agents that Get Faster on Repeated Tasks", "url": "https://arxiv.org/abs/2606.17929", "published": "2026-06-16", "updated": "2026-06-16", "authors": [ "Bojie Li" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "agent-safety", "computer-use" ], "score": 11, "relevance": "medium", "matched_queries": [ "web-gui-agent" ], "arxiv_id": "2606.17929", "source": "arxiv", "source_id": "arxiv:2606.17929", "pdf_url": "https://arxiv.org/pdf/2606.17929", "primary_query": "web-gui-agent" }, { "id": "2606.16465", "title": "When Agent Automation Becomes Profitable: Quantifying and Insuring Autonomous AI Risk through Trace-Economic Underwriting", "url": "https://arxiv.org/abs/2606.16465", "published": "2026-06-15", "updated": "2026-06-15", "authors": [ "Binyan Xu", "Xilin Dai", "Fan Yang", "Kehuan Zhang" ], "categories": [ "cs.AI", "cs.CE" ], "topics": [ "agent-safety", "tool-use", "workflow-agent" ], "score": 11, "relevance": "medium", "matched_queries": [ "tool-use" ], "arxiv_id": "2606.16465", "source": "arxiv", "source_id": "arxiv:2606.16465", "pdf_url": "https://arxiv.org/pdf/2606.16465", "primary_query": "tool-use" }, { "id": "2606.28365", "title": "CAMI: Cost-Aware Agent-Guided Multi-Indexing for Semantic Retrieval", "url": "https://arxiv.org/abs/2606.28365", "published": "2026-06-14", "updated": "2026-06-14", "authors": [ "Adnan Qidwai", "Anand Eswaran", "Sonam Mishra", "Jaydeep Sen", "Sachindra Joshi" ], "categories": [ "cs.IR", "cs.AI" ], "topics": [ "agent-evaluation", "agent-safety", "computer-use", "rag" ], "score": 11, "relevance": "medium", "matched_queries": [ "rag-agent" ], "arxiv_id": "2606.28365", "source": "arxiv", "source_id": "arxiv:2606.28365", "pdf_url": "https://arxiv.org/pdf/2606.28365", "primary_query": "rag-agent" }, { "id": "2606.15485", "title": "The Perils of Agency: How Developers Perceive, Prioritize, and Address Risks in Agentic AI Products", "url": "https://arxiv.org/abs/2606.15485", "published": "2026-06-13", "updated": "2026-06-13", "authors": [ "Hao-Ping Lee", "Jessica He", "David Piorkowski", "Thomas Serban von Davier", "Jodi Forlizzi", "Sauvik Das" ], "categories": [ "cs.CY", "cs.AI", "cs.HC", "cs.LG", "cs.SE" ], "topics": [ "agent-safety", "tool-use" ], "score": 11, "relevance": "medium", "matched_queries": [ "tool-use" ], "arxiv_id": "2606.15485", "source": "arxiv", "source_id": "arxiv:2606.15485", "pdf_url": "https://arxiv.org/pdf/2606.15485", "primary_query": "tool-use" }, { "id": "2606.15335", "title": "Privacy-Preserving Text Sanitization for Distributed Agents Collaboration via Disentangled Representations", "url": "https://arxiv.org/abs/2606.15335", "published": "2026-06-13", "updated": "2026-06-13", "authors": [ "Xuan Liu", "Hefeng Zhou", "Sicheng Chen", "Chao Yang", "Xingcheng Xu", "Jingjing Qu", "Jiong Lou", "Jie LI", "Xia Hu" ], "categories": [ "cs.CL", "cs.AI" ], "topics": [ "agent-evaluation", "agent-safety", "multi-agent", "rag" ], "score": 11, "relevance": "medium", "matched_queries": [ "rag-agent" ], "arxiv_id": "2606.15335", "source": "arxiv", "source_id": "arxiv:2606.15335", "pdf_url": "https://arxiv.org/pdf/2606.15335", "primary_query": "rag-agent" }, { "id": "2606.15007", "title": "Nemotron 3 Ultra: Open, Efficient Mixture-of-Experts Hybrid Mamba-Transformer Model for Agentic Reasoning", "url": "https://arxiv.org/abs/2606.15007", "published": "2026-06-12", "updated": "2026-06-12", "authors": [ "NVIDIA", ":", "Aaron Blakeman", "Aaron Thomas", "Aastha Jhunjhunwala", "Abhibha Gupta", "Abhinav Khattar", "Adam Rajfer", "Adi Renduchintala", "Adil Asif", "Aditya Vavre", "Adriana Flores Miranda", "Ahmad Bilal", "Aileen Zaman", "Ajay Hotchandani", "Akanksha Shukla", "Akhiad Bercovich", "Aleksander Ficek", "Alex Gronskiy", "Alex Kondratenko", "Alex Steiner", "Alex Ye", "Alexander Bukharin", "Alexandre Milesi", "Ali Taghibakhshi", "Alice Gatti", "Alisa Liu", "Alok Kumar", "Amar Phanishayee", "Ameya Sunil Mahabaleshwarkar", "Amir Klein", "Amit Zuker", "Amnon Geifman", "Anahita Bhiwandiwalla", "Ananth Subramaniam", "Andrea Santilli", "Andrew Fulks", "Andrew McHarg", "Andrew Tao", "Andrii Skliar", "Anjulie Agrusa", "Ankur Srivastava", "Ankur Verma", "Anna Shors", "Anna Warno", "Antoni-Joan Solergibert I Llaquet", "Arham Mehta", "Arkadiusz Nowaczynski", "Arti Jain", "Ashwath Aithal", "Ashwin Poojary", "Asif Ahamed", "Asit Mishra", "Asma Kuriparambil Thekkumpate", "Atefeh Sohrabizadeh", "Avinash Kaur", "Avinash Vem", "Ayush Dattagupta", "Barath Subramaniam Anandan", "Bardiya Sadeghi", "Ben Lanir", "Benedikt Schifferer", "Besmira Nushi", "Bilal Kartal", "Bill Thiede", "Bita Darvish Rouhani", "Bo Deng", "Bob Schatz", "Boris Ginsburg", "Boxin Wang", "Brad Nemire", "Brandon Norick", "Brian Dang", "Brian Westphal", "Brian Yu", "Brucek Khailany", "Bryan Catanzaro", "Carlo del Mundo", "Caryln Aarish", "Chankyu Lee", "Chantal Hwang", "Charbel Sakr", "Charles Wang", "Charlie Truong", "Chen Cui", "Cheng Cheng", "Cheng-Ping Hsieh", "Chenghao Zhang", "Chenhui Deng", "Chintan Patel", "Chris Alexiuk", "Christian Cosgrove", "Christian Munley", "Christine Harvey", "Christopher Parisien", "Chunyang Shen", "Coco Li", "Collin Neale", "Cynthia Gao", "Cyril Meurillon", "Dan Gil", "Dan Su", "Dan Zhao", "Dane Corneil", "Daniel Afrimi", "Daniel Egert", "Daniel Korzekwa", "Daniel Lo", "Daniel Machlab", "Daniel Serebrenik", "Daniil Sorokin", "Daria Gitman", "Daria Levy", "Darko Stosic", "David Mosallanezhad", "David Yu", "Davit Karamyan", "Deena Donia", "Deep Debroy", "Deepak Narayanan", "Devin O'Kelly", "Dheeraj Peri", "Dhruv Nathawani", "Di", "Wu", "Dima Rekesh", "Divyanshu Kakwani", "Donald Plummer", "Dong Anh", "Dongfeng Yu", "Dongfu Jiang", "Donnie Kim", "Dorrin Poorkay", "Duncan Riach", "Dusan Stosic", "Dustin VanStee", "Eavan Meng", "Edgar Minasyan", "Edward Lin", "Eileen Margaret Peters Long", "Elad Sarafin", "Elad Segal", "Elena Lantz", "Ellie Evans", "Elliott Ning", "Eric Chung", "Eric Harper", "Eric Pham-Hung", "Eric Tramel", "Eric Yang", "Erick Galinkin", "Erik Pounds", "Erika Goncalves Goncalves", "Evan Briones", "Evan Wu", "Evelina Bakhturina", "Evgeny Tsykunov", "Ewa Dobrowolska", "Faisal Ladhak", "Farzan Memarian", "Fay Wang", "Fei Jia", "Felipe Soares", "Felipe Vieira Frujeri", "Feng Chen", "Fengguang Lin", "Ferenc Galko", "Frank Sun", "Frankie Siino", "Frida Hou", "Gal Hubara Agam", "Gal Kaplun", "Gantavya Bhatt", "Gargi Prasad", "Garvit Kulshreshtha", "George Armstrong", "Gerald Shen", "Giulio Borghesi", "Gordana Neskovic", "Gorkem Batmaz", "Grace Lam", "Greg Mason", "Greg Pauloski", "Grigor Nalbandyan", "Grzegorz Chlebus", "Grzegorz Karch", "Guan-Ting Liu", "Guoming Zhang", "Guyue Huang", "Haggai Maron", "Haifeng Qian", "Haim Elisha", "Haoxing Ren", "Haran Kumar Shiv Kumar", "Haribhau Hud", "Harris Nover", "Harrison Saturley Hall", "Hayate Iso", "Helen Ngo", "Herbert Hum", "Herman Sahota", "Hexin Wang", "Himanshu Soni", "Hovhannes Tamoyan", "Hua Li", "Huanhuan Chen", "Hui Li", "Hui Wang", "Huy Nguyen", "Ian Chiles", "Ido Galil", "Ido Shahaf", "Igor Gitman", "Igor Shovkun", "Ilya Loshchilov", "Ingo Guehring", "Itamar Schen", "Itay Levy", "Itay Neeman", "Ivan Moshkov", "Izik Golan", "Izzy Putterman", "Jaemin Choi", "Jakub Slowikowski", "Jan Kautz", "Jane Polak Scowcroft", "Jared Casper", "Jatin Mitra", "Jeffrey Glick", "Jenny Chen", "Jesse Oliver", "Jiacheng Xu", "Jiafan Zhu", "Jialin Song", "Jian Zhang", "Jiantao Jiao", "Jiaqi Zeng", "Jie Lou", "Jim King", "Jimmy Zhang", "Jingquan Wang", "Jinhang Choi", "Jinju Chu", "Joey Conway", "Joey Guman", "Johan Jatko", "Johannes Rausch", "John Kamalu", "John Roberts", "Johnny Greco", "Johnny Mensel", "Jonah Alben", "Jonas Yang", "Jonathan Cohen", "Jonathan Raiman", "Joseph Jennings", "Joshua Mabry", "Joshua Pierce", "Joyjit Daw", "Julien Veron Vialard", "Junkeun Yi", "Jupinder Parmar", "Kajal Jain", "Kan Zhu", "Kari Briski", "Katherine Cheung", "Katherine Luna", "Keith Willowhawk", "Keith Wyss", "Keshav Santhanam", "Kevin Shih", "Kezhi Kong", "Khanh Nguyen", "Khushi Bhardwaj", "Kirthi Shankar Sivamani", "Konstantinos Krommydas", "Krishna C. Puvvada", "Krzysztof Pawelec", "Kumar Anik", "Kyle Keprios", "Kylie Day", "Lawrence McAfee", "Leo Du", "Leon Derczynski", "Li Ding", "Linda Liu", "Lingjie Wu", "Lior Kadoch", "Lizzie Wei", "Luis Vega", "Luke Robison", "Lun Su", "Maarten Van Segbroeck", "Maciej Jakub Mikulski", "Maer Rodrigues de Melo", "Magda Sypula", "Mahan Fathi", "Makesh Narsimhan Sreedhar", "Makesh Tarun Chandran", "Manoj Kilaru", "Maor Ashkenazi", "Marc Cuevas", "Marc Romeijn", "Marcin Chochowski", "Mark Cai", "Mark Mozolewski", "Markus Kliegl", "Marta Stepniewska-Dziubinska", "Martyna Patelka", "Mattei Machczynski", "Matvei Novikov", "Mauricio Ferrato", "Maximilian Golub", "Mehrzad Samadi", "Melissa Corpuz", "Mengru Wang", "Mengxi Wu", "Meredith Price", "Meriem Boubdir", "Micah Schaffer", "Michael Andersch", "Michael Boone", "Michael Gschwind", "Michael Lightstone", "Michael Loh", "Michal Bien", "Michal Zawalski", "Michelle Gill", "Miguel Martinez", "Mikail Khona", "Mike Chrzanowski", "Mike Houston", "Mingyuan Ma", "Minseok Lee", "Mohamed Fawzy", "Mohammad Dabbah", "Mohammad Shoeybi", "Mostofa Patwary", "Nabin Mulepati", "Najeeb Nabwani", "Namit Dhameja", "Narimane Hennouni", "Natalie Hereth", "Nathaniel Pinckney", "Nave Algarici", "Nave Assaf", "Netanel Haber", "Nicholas Knight", "Nick Reamaroon", "Nickson Quak", "Nidhi Bhatia", "Nikhil Desai", "Nikolai Ludwig", "Nima Tajbakhsh", "Ning Xu", "Nir Ailon", "Nirmal Juluru", "Nitin Nitin", "Ofri Masad", "Oleg Rybakov", "Oleksii Hrinchuk", "Oleksii Kuchaiev", "Olivia Viessmann", "Olivier Delalleau", "Oluwatobi Olabiyi", "Omer Ullman Argov", "Omri Puny", "Oren Tropp", "Pablo Ribalta", "Pallab Bhattacharya", "Panos Lampropoulos", "Parth Mannan", "Pasha Shamis", "Patrick Legresley", "Paul Gibbons", "Pavlo Molchanov", "Pawel Morkisz", "Peter Dykas", "Peter Jin", "Pierre-Yves Aquilanti", "Pinky Xu", "Piotr Januszewski", "Piotr Laskiewicz", "Pooya Jannaty", "Prakash Gurumurthy", "Pranav Prashant Thombre", "Prasoon Varshney", "Pritam Gundecha", "Przemek Tredak", "Puhui Meng", "Qiyu Wan", "Rabeeh Karimi Mahabadi", "Rachel Oberman", "Rachit Garg", "Radha Sri-Tharan", "Rahul Kandu", "Rakshit Sanadhya", "Ran El-Yaniv", "Ran Zilberstein", "Rasoul Shafipour", "Ray Macalisang", "Rayen Tian", "Reka Kovacs", "Renjie Pi", "Rick Izzo", "Rima Shahbazyan", "Rishabh Garg", "Rishi Puri", "Rita Fernandes Neves", "Ritchie Zhao", "Ritika Borkar", "Ritu Gala", "Riyad Islam", "Robert Clark", "Robert Hesse", "Robert Kirby", "Roger Waleffe", "Rohit Watve", "Roi Koren", "Ron Banner", "Ruoxi Zhang", "Russell J. Hewett", "Ryan Prenger", "Ryan Stewart", "Ryota Egashira", "Sadegh Mahdavi", "Saee Paliwal", "Sagar Singh", "Sahil Modi", "Salika Dave", "Samantha Shinagawa", "Samuel Kriman", "Sandip Bhaskar", "Sangkug Lym", "Sanjay Kariyappa", "Sanjeev Satheesh", "Saran Vikas Murari", "Satish Pasumarthi", "Saurabh Mishra", "Saurav Muralidharan", "Scott Hara", "Sean Narentharen", "Selvaraj Anandaraj", "Seonjin Na", "Seonmeyong Bak", "Seonmyeong Bak", "Sepehr Sameni", "Seph Mard", "Serge Panev", "Seth Henneman", "Seth Poulos", "Shahar Mor", "Shantanu Acharya", "Shaona Ghosh", "Sharath Turuvekere Sreenivas", "Sharon Mendelson", "Shaun Kotek", "Shawn Wang", "Shay Aharon", "Shaya Gharghabi", "Sheng-Chieh Lin", "Shi Chen", "Shiqing Fan", "Shirish Baskaran", "Shreya Gopa", "Shrimai Prabhumoye", "Shubham Pachori", "Shubham Toshniwal", "Shuoyang Ding", "Shwetha Krishnamurthy", "Siddharth Singh", "Simeng Sun", "Sirshak Das", "Sivakumar Arayandi Thottakara", "Smita Ithape", "Somshubra Majumdar", "Soumye Singhal", "Sri Harsha Singudasu", "Sridhar Bhuvanapalli", "Srimukh Veccham", "Stas Sergienko", "Stefania Alborghetti", "Stephen Ge", "Su Rong", "Sugam Dipak Devare", "Sukrit Rao", "Sumeet Kumar Barua", "Sungsoo Ha", "Sunny Gai", "Suriya Gunasekar", "Suseella Panguluri", "Suyog Gupta", "Sviataslau Hinzburh", "Sweta Priyadarshi", "Syeda Nahida Akter", "Talor Abramovich", "Tan Bui", "Tanay Varshney", "Tatevik Ter-Hovhannisyan", "Teodor-Dumitru Ene", "Terry Kong", "Thanh Do", "Tianhe Zhang", "Tiffany Moore", "Tijmen Blankevoort", "Tim Moon", "Tiyasa Mitra", "Tom Balough", "Tomasz Grzegorzek", "Tomasz Hliwiak", "Tomer Asida", "Tomer Bar Natan", "Tomer Keren", "Tomer Ronen", "Tony Salim", "Tony Wang", "Traian Rebedea", "Tugrul Konuk", "Twinkle Vashishth", "Udi Karpas", "Ushnish De", "Vahid Noorozi", "Venkat Srinivasan", "Venmugil Elango", "Vibhor Agrawal", "Victor Cui", "Vijay Korthikanti", "Vikas Mehta", "Vinay Rao", "Virginia Wu", "Vitaly Kurin", "Vitaly Lavrukhin", "Vladimir Anisimov", "Vu Pham", "Wanli Jiang", "Wasi Uddin Ahmad", "Wataru Ishihara", "Wei Du", "Wei Ping", "Weiheng Chai", "Wenliang Dai", "Wesley Helmholz", "Will Jennings", "Will Zhu", "Wojciech Prazuch", "Xiaowei Ren", "Xiwen Yu", "Yan Breek", "Yang Chen", "Yang Yu", "Yangyi Chen", "Yaniv Galron", "Yashaswi Karnati", "Yejin Choi", "Yev Meyer", "Yi-Fu Wu", "Yian Zhang", "Ying Lin", "Yonatan Geifman", "Yonggan Fu", "Youngeun Kwon", "Yu Yao", "Yugi Guvvla", "Yuki Huang", "Yunsheng Liu", "Zach Moshe", "Zachary Newell", "Zhilin Wang", "Zhiyu Li", "Zhongbo Zhu", "Zhuolin Yang", "Zihan Liu", "Zijie Yan", "Zsolt-Alon Wertheimer" ], "categories": [ "cs.CL", "cs.AI", "cs.LG" ], "topics": [ "reasoning" ], "score": 11, "relevance": "medium", "matched_queries": [ "autonomous-agent-llm" ], "arxiv_id": "2606.15007", "source": "arxiv", "source_id": "arxiv:2606.15007", "pdf_url": "https://arxiv.org/pdf/2606.15007", "primary_query": "autonomous-agent-llm" }, { "id": "2606.14266", "title": "Large Language Model Based Agent for Automated Discovery in Computational Physics", "url": "https://arxiv.org/abs/2606.14266", "published": "2026-06-12", "updated": "2026-06-12", "authors": [ "Hang Lin", "Chongwen Liu", "Gang Yan" ], "categories": [ "physics.comp-ph" ], "topics": [ "agent-evaluation", "computer-use", "rag", "tool-use" ], "score": 11, "relevance": "medium", "matched_queries": [ "autonomous-agent-llm" ], "arxiv_id": "2606.14266", "source": "arxiv", "source_id": "arxiv:2606.14266", "pdf_url": "https://arxiv.org/pdf/2606.14266", "primary_query": "autonomous-agent-llm" }, { "id": "2606.12830", "title": "Perceive, Interact, Reason: Building Tool-Augmented Visual Agents for Spatial Reasoning", "url": "https://arxiv.org/abs/2606.12830", "published": "2026-06-11", "updated": "2026-06-11", "authors": [ "Changye Li", "Meng Lu", "Yi Wu", "Ligeng Zhu" ], "categories": [ "cs.CV", "cs.AI" ], "topics": [ "agent-evaluation", "reasoning", "tool-use" ], "score": 11, "relevance": "medium", "matched_queries": [ "tool-use" ], "arxiv_id": "2606.12830", "source": "arxiv", "source_id": "arxiv:2606.12830", "pdf_url": "https://arxiv.org/pdf/2606.12830", "primary_query": "tool-use" }, { "id": "2606.13380", "title": "An LLM System for Autonomous Variational Quantum Circuit Design", "url": "https://arxiv.org/abs/2606.13380", "published": "2026-06-11", "updated": "2026-06-11", "authors": [ "Kenya Sakka", "Wataru Mizukami", "Kosuke Mitarai" ], "categories": [ "quant-ph", "cs.AI" ], "topics": [ "agent-evaluation", "rag", "workflow-agent" ], "score": 11, "relevance": "medium", "matched_queries": [ "autonomous-agent-llm" ], "arxiv_id": "2606.13380", "source": "arxiv", "source_id": "arxiv:2606.13380", "pdf_url": "https://arxiv.org/pdf/2606.13380", "primary_query": "autonomous-agent-llm" }, { "id": "2606.12666", "title": "CAPED: Context-Aware Privacy Exposure Defense for Mobile GUI Agents", "url": "https://arxiv.org/abs/2606.12666", "published": "2026-06-10", "updated": "2026-06-16", "authors": [ "Siyu Shen", "Fenghao Xu", "Wenrui Diao", "Kehuan Zhang" ], "categories": [ "cs.CR", "cs.AI" ], "topics": [ "agent-evaluation", "computer-use" ], "score": 11, "relevance": "medium", "matched_queries": [ "web-gui-agent" ], "arxiv_id": "2606.12666", "source": "arxiv", "source_id": "arxiv:2606.12666", "pdf_url": "https://arxiv.org/pdf/2606.12666", "primary_query": "web-gui-agent" }, { "id": "2606.10522", "title": "GUI-AC: Enhancing Continual Learning in GUI Agents", "url": "https://arxiv.org/abs/2606.10522", "published": "2026-06-09", "updated": "2026-07-06", "authors": [ "Can Lin", "Tao Feng", "Hangjie Yuan", "Dan Zhang", "Yifan Zhu", "Zhonghong Ou" ], "categories": [ "cs.CV" ], "topics": [ "computer-use", "rag", "tool-use" ], "score": 11, "relevance": "medium", "matched_queries": [ "web-gui-agent" ], "arxiv_id": "2606.10522", "source": "arxiv", "source_id": "arxiv:2606.10522", "pdf_url": "https://arxiv.org/pdf/2606.10522", "primary_query": "web-gui-agent" }, { "id": "2606.10062", "title": "Deployment-Time Memorization in Foundation-Model Agents", "url": "https://arxiv.org/abs/2606.10062", "published": "2026-06-08", "updated": "2026-06-08", "authors": [ "Lei", "Chen", "Guilin Zhang", "Kai Zhao", "Dalmo Cirne", "Andy Olsen", "Xu Chu", "Zeke Miller", "Alet Blanken", "Amine Anoun", "Jerry Ting" ], "categories": [ "cs.AI", "cs.MA" ], "topics": [ "agent-evaluation", "agent-safety", "memory", "rag", "tool-use" ], "score": 11, "relevance": "medium", "matched_queries": [ "agent-memory" ], "arxiv_id": "2606.10062", "source": "arxiv", "source_id": "arxiv:2606.10062", "pdf_url": "https://arxiv.org/pdf/2606.10062", "primary_query": "agent-memory" }, { "id": "2606.06708", "title": "Signal-Driven Observation for Long-Horizon Web Agents", "url": "https://arxiv.org/abs/2606.06708", "published": "2026-06-04", "updated": "2026-06-04", "authors": [ "Shubham Gaur", "Ian Lane" ], "categories": [ "cs.CL" ], "topics": [ "computer-use", "planning", "reasoning", "tool-use" ], "score": 11, "relevance": "medium", "matched_queries": [ "web-gui-agent" ], "arxiv_id": "2606.06708", "source": "arxiv", "source_id": "arxiv:2606.06708", "pdf_url": "https://arxiv.org/pdf/2606.06708", "primary_query": "web-gui-agent" }, { "id": "2606.04391", "title": "Online Skill Learning for Web Agents via State-Grounded Dynamic Retrieval", "url": "https://arxiv.org/abs/2606.04391", "published": "2026-06-03", "updated": "2026-06-03", "authors": [ "Jiaxi Li", "Ke Deng", "Yun Wang", "Jingyuan Huang", "Yucheng Shi", "Qiaoyu Tan", "Jin Lu", "Ninghao Liu" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "computer-use", "rag", "tool-use", "workflow-agent" ], "score": 11, "relevance": "medium", "matched_queries": [ "language-agent" ], "arxiv_id": "2606.04391", "source": "arxiv", "source_id": "arxiv:2606.04391", "pdf_url": "https://arxiv.org/pdf/2606.04391", "primary_query": "language-agent" }, { "id": "2606.05112", "title": "Evaluating Large Language Models in Dynamic Clinical Decision-Making with Standardized Patient Cases", "url": "https://arxiv.org/abs/2606.05112", "published": "2026-06-03", "updated": "2026-06-03", "authors": [ "Cheng Liang", "Pengcheng Qiu", "Ya Zhang", "Yanfeng Wang", "Chaoyi Wu", "Weidi Xie" ], "categories": [ "cs.CL" ], "topics": [ "agent-evaluation", "planning", "world-model" ], "score": 11, "relevance": "medium", "matched_queries": [ "agent-evaluation" ], "arxiv_id": "2606.05112", "source": "arxiv", "source_id": "arxiv:2606.05112", "pdf_url": "https://arxiv.org/pdf/2606.05112", "primary_query": "agent-evaluation" }, { "id": "2606.04435", "title": "Cascading Hallucination in Agentic RAG: The CHARM Framework for Detection and Mitigation", "url": "https://arxiv.org/abs/2606.04435", "published": "2026-06-03", "updated": "2026-06-03", "authors": [ "Saroj Mishra" ], "categories": [ "cs.AI", "cs.CL", "cs.CR", "cs.IR" ], "topics": [ "agent-evaluation", "rag", "reasoning" ], "score": 11, "relevance": "medium", "matched_queries": [ "rag-agent" ], "arxiv_id": "2606.04435", "source": "arxiv", "source_id": "arxiv:2606.04435", "pdf_url": "https://arxiv.org/pdf/2606.04435", "primary_query": "rag-agent" }, { "id": "2606.28347", "title": "Agentic Safety is an Epistemic Property, Not a Behavioral One", "url": "https://arxiv.org/abs/2606.28347", "published": "2026-06-02", "updated": "2026-06-02", "authors": [ "Charles L. Wang", "Keir Dorchen", "Peter Jin" ], "categories": [ "cs.CY", "cs.AI", "cs.LG" ], "topics": [ "agent-safety", "embodied-agent", "rag" ], "score": 11, "relevance": "medium", "matched_queries": [ "agent-safety" ], "arxiv_id": "2606.28347", "source": "arxiv", "source_id": "arxiv:2606.28347", "pdf_url": "https://arxiv.org/pdf/2606.28347", "primary_query": "agent-safety" }, { "id": "2606.28344", "title": "PIXELRAG: Web Screenshots Beat Text for Retrieval-Augmented Generation", "url": "https://arxiv.org/abs/2606.28344", "published": "2026-06-01", "updated": "2026-06-01", "authors": [ "Yichuan Wang", "Zhifei Li", "Zirui Wang", "Paul Teiletche", "Lesheng Jin", "Matei Zaharia", "Joseph E. Gonzalez", "Sewon Min" ], "categories": [ "cs.IR", "cs.AI", "cs.CL", "cs.CV", "cs.LG" ], "topics": [ "agent-evaluation", "rag", "tool-use" ], "score": 11, "relevance": "medium", "matched_queries": [ "agent-evaluation", "rag-agent" ], "arxiv_id": "2606.28344", "source": "arxiv", "source_id": "arxiv:2606.28344", "pdf_url": "https://arxiv.org/pdf/2606.28344", "primary_query": "agent-evaluation" }, { "id": "2606.00183", "title": "Agentic Transformers Provably Learn to Search via Reinforcement Learning", "url": "https://arxiv.org/abs/2606.00183", "published": "2026-05-29", "updated": "2026-05-29", "authors": [ "Tong Yang", "Yu Huang", "Yingbin Liang", "Yuejie Chi" ], "categories": [ "cs.LG", "cs.AI", "math.OC", "stat.ML" ], "topics": [ "memory", "reasoning", "tool-use" ], "score": 11, "relevance": "medium", "matched_queries": [ "language-agent" ], "arxiv_id": "2606.00183", "source": "arxiv", "source_id": "arxiv:2606.00183", "pdf_url": "https://arxiv.org/pdf/2606.00183", "primary_query": "language-agent" }, { "id": "2605.30916", "title": "Welfare, Improvability, and Variance: A Principal-Agent Approach to Optimal Benchmark Item Aggregation", "url": "https://arxiv.org/abs/2605.30916", "published": "2026-05-29", "updated": "2026-05-29", "authors": [ "Andreas Haupt", "Justin Hartenstein", "Anka Reuel", "Mykel Kochenderfer", "Sanmi Koyejo" ], "categories": [ "cs.LG", "cs.GT", "econ.TH" ], "topics": [ "agent-evaluation", "agent-safety", "rag" ], "score": 11, "relevance": "medium", "matched_queries": [ "agent-evaluation" ], "arxiv_id": "2605.30916", "source": "arxiv", "source_id": "arxiv:2605.30916", "pdf_url": "https://arxiv.org/pdf/2605.30916", "primary_query": "agent-evaluation" }, { "id": "2605.29786", "title": "Croissant Tasks: A Metadata Format for Reproducible Machine Learning Evaluations", "url": "https://arxiv.org/abs/2605.29786", "published": "2026-05-28", "updated": "2026-05-28", "authors": [ "Omar Benjelloun", "Leonardo Martins Bianco", "Isabelle Guyon", "Thanh Gia Hieu Khuong", "Jonathan Lebensold", "Sebastian Lobentanzer", "Luis Oala", "Benedictus Kent Rachmat", "Ihsan Ullah", "Peyman Vahidi", "Joaquin Vanschoren" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "tool-use" ], "score": 11, "relevance": "medium", "matched_queries": [ "autonomous-agent-llm" ], "arxiv_id": "2605.29786", "source": "arxiv", "source_id": "arxiv:2605.29786", "pdf_url": "https://arxiv.org/pdf/2605.29786", "primary_query": "autonomous-agent-llm" }, { "id": "2605.27995", "title": "AsyncTool: Evaluating the Asynchronous Function Calling Capability under Multi-Task Scenarios", "url": "https://arxiv.org/abs/2605.27995", "published": "2026-05-27", "updated": "2026-05-28", "authors": [ "Kou Shi", "Ziao Zhang", "Shiting Huang", "Avery Nie", "Zhen Fang", "Qiuchen Wang", "Lin Chen", "Huaian Chen", "Zehui Chen", "Feng Zhao" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "reasoning", "tool-use" ], "score": 11, "relevance": "medium", "matched_queries": [ "function-calling" ], "arxiv_id": "2605.27995", "source": "arxiv", "source_id": "arxiv:2605.27995", "pdf_url": "https://arxiv.org/pdf/2605.27995", "primary_query": "function-calling" }, { "id": "2605.23459", "title": "AI Assurance: A Comprehensive Testing Strategy for Enterprise AI Systems", "url": "https://arxiv.org/abs/2605.23459", "published": "2026-05-22", "updated": "2026-05-22", "authors": [ "Chitra Badagi", "Divye Singh", "Animesh Sen", "Adinath Shirsath" ], "categories": [ "cs.SE", "cs.AI" ], "topics": [ "agent-evaluation", "agent-safety", "computer-use", "rag", "workflow-agent" ], "score": 11, "relevance": "medium", "matched_queries": [ "autonomous-agent-llm" ], "arxiv_id": "2605.23459", "source": "arxiv", "source_id": "arxiv:2605.23459", "pdf_url": "https://arxiv.org/pdf/2605.23459", "primary_query": "autonomous-agent-llm" }, { "id": "2605.20896", "title": "GenAI-Driven Threat Detection with Microsoft Security Copilot", "url": "https://arxiv.org/abs/2605.20896", "published": "2026-05-20", "updated": "2026-05-22", "authors": [ "Scott Freitas", "Amir Gharib" ], "categories": [ "cs.CR", "cs.AI", "cs.LG" ], "topics": [ "agent-evaluation", "agent-safety", "computer-use", "planning", "rag" ], "score": 11, "relevance": "medium", "matched_queries": [ "autonomous-agent-llm" ], "arxiv_id": "2605.20896", "source": "arxiv", "source_id": "arxiv:2605.20896", "pdf_url": "https://arxiv.org/pdf/2605.20896", "primary_query": "autonomous-agent-llm" }, { "id": "2605.20477", "title": "Training Language Agents to Learn from Experience", "url": "https://arxiv.org/abs/2605.20477", "published": "2026-05-19", "updated": "2026-05-19", "authors": [ "Yuval Shalev", "Zifeng Ding", "Mateja Jamnik" ], "categories": [ "cs.LG", "cs.AI", "cs.CL" ], "topics": [ "agent-evaluation", "reasoning" ], "score": 11, "relevance": "medium", "matched_queries": [ "language-agent" ], "arxiv_id": "2605.20477", "source": "arxiv", "source_id": "arxiv:2605.20477", "pdf_url": "https://arxiv.org/pdf/2605.20477", "primary_query": "language-agent" }, { "id": "2605.15077", "title": "Concurrency without Model Changes: Future-based Asynchronous Function Calling for LLMs", "url": "https://arxiv.org/abs/2605.15077", "published": "2026-05-14", "updated": "2026-05-14", "authors": [ "Guangyu Feng", "Huanzhi Mao", "Prabal Dutta", "Joseph E. Gonzalez" ], "categories": [ "cs.CL", "cs.AI", "cs.LG" ], "topics": [ "agent-evaluation", "coding-agent", "tool-use" ], "score": 11, "relevance": "medium", "matched_queries": [ "function-calling" ], "arxiv_id": "2605.15077", "source": "arxiv", "source_id": "arxiv:2605.15077", "pdf_url": "https://arxiv.org/pdf/2605.15077", "primary_query": "function-calling" }, { "id": "2605.12460", "title": "Multi-Stream LLMs: Unblocking Language Models with Parallel Streams of Thoughts, Inputs and Outputs", "url": "https://arxiv.org/abs/2605.12460", "published": "2026-05-12", "updated": "2026-05-12", "authors": [ "Guinan Su", "Yanwu Yang", "Xueyan Li", "Jonas Geiping" ], "categories": [ "cs.LG", "cs.CL" ], "topics": [ "agent-safety", "coding-agent", "computer-use", "tool-use" ], "score": 11, "relevance": "medium", "matched_queries": [ "autonomous-agent-llm" ], "arxiv_id": "2605.12460", "source": "arxiv", "source_id": "arxiv:2605.12460", "pdf_url": "https://arxiv.org/pdf/2605.12460", "primary_query": "autonomous-agent-llm" }, { "id": "2605.11442", "title": "Can a Single Message Paralyze the AI Infrastructure? The Rise of AbO-DDoS Attacks through Targeted Mobius Injection", "url": "https://arxiv.org/abs/2605.11442", "published": "2026-05-12", "updated": "2026-05-12", "authors": [ "Zi Liang", "Ronghua Li", "Yanyun Wang", "Qingqing Ye", "Haibo Hu" ], "categories": [ "cs.CR", "cs.AI", "cs.CL" ], "topics": [ "agent-evaluation", "agent-safety", "coding-agent", "tool-use" ], "score": 11, "relevance": "medium", "matched_queries": [ "autonomous-agent-llm" ], "arxiv_id": "2605.11442", "source": "arxiv", "source_id": "arxiv:2605.11442", "pdf_url": "https://arxiv.org/pdf/2605.11442", "primary_query": "autonomous-agent-llm" }, { "id": "2605.11235", "title": "Internalizing Curriculum Judgment for LLM Reinforcement Fine-Tuning", "url": "https://arxiv.org/abs/2605.11235", "published": "2026-05-11", "updated": "2026-05-11", "authors": [ "Han Zheng", "Yining Ma", "Karthick Gunasekaran", "Bharathan Balaji", "Zheng Du", "Shiv Vitaladevuni", "Cathy Wu" ], "categories": [ "cs.LG", "cs.AI" ], "topics": [ "agent-evaluation", "agent-safety", "rag", "reasoning" ], "score": 11, "relevance": "medium", "matched_queries": [ "function-calling" ], "arxiv_id": "2605.11235", "source": "arxiv", "source_id": "arxiv:2605.11235", "pdf_url": "https://arxiv.org/pdf/2605.11235", "primary_query": "function-calling" }, { "id": "2605.09990", "title": "Merlin: Deterministic Byte-Exact Deduplication for Lossless Context Optimization in Large Language Model Inference", "url": "https://arxiv.org/abs/2605.09990", "published": "2026-05-11", "updated": "2026-05-11", "authors": [ "Sietse Schelpe" ], "categories": [ "cs.CL" ], "topics": [ "agent-evaluation", "rag", "tool-use", "workflow-agent" ], "score": 11, "relevance": "medium", "matched_queries": [ "autonomous-agent-llm" ], "arxiv_id": "2605.09990", "source": "arxiv", "source_id": "arxiv:2605.09990", "pdf_url": "https://arxiv.org/pdf/2605.09990", "primary_query": "autonomous-agent-llm" }, { "id": "2605.06320", "title": "Improving the Efficiency of Language Agent Teams with Adaptive Task Graphs", "url": "https://arxiv.org/abs/2605.06320", "published": "2026-05-07", "updated": "2026-05-07", "authors": [ "Elizabeth Mieczkowski", "Alexander Ku", "Tiwalayo Eisape", "Dilip Arumugam", "John Matters", "Katherine M. Collins", "Ilia Sucholutsky", "Thomas L. Griffiths" ], "categories": [ "cs.MA", "cs.AI", "cs.CL" ], "topics": [ "planning" ], "score": 11, "relevance": "medium", "matched_queries": [ "language-agent" ], "arxiv_id": "2605.06320", "source": "arxiv", "source_id": "arxiv:2605.06320", "pdf_url": "https://arxiv.org/pdf/2605.06320", "primary_query": "language-agent" }, { "id": "2605.06161", "title": "Beyond Accuracy: Policy Invariance as a Reliability Test for LLM Safety Judges", "url": "https://arxiv.org/abs/2605.06161", "published": "2026-05-07", "updated": "2026-05-07", "authors": [ "Shihao Weng", "Yang Feng", "Xiaofei Xie" ], "categories": [ "cs.AI", "cs.SE" ], "topics": [ "agent-evaluation", "agent-safety", "computer-use" ], "score": 11, "relevance": "medium", "matched_queries": [ "agent-safety" ], "arxiv_id": "2605.06161", "source": "arxiv", "source_id": "arxiv:2605.06161", "pdf_url": "https://arxiv.org/pdf/2605.06161", "primary_query": "agent-safety" }, { "id": "2604.27264", "title": "Self-Evolving Software Agents", "url": "https://arxiv.org/abs/2604.27264", "published": "2026-04-29", "updated": "2026-04-29", "authors": [ "Marco Robol", "Paolo Giorgini" ], "categories": [ "cs.SE", "cs.AI" ], "topics": [ "agent-evaluation", "multi-agent", "reasoning" ], "score": 11, "relevance": "medium", "matched_queries": [ "autonomous-agent-llm" ], "arxiv_id": "2604.27264", "source": "arxiv", "source_id": "arxiv:2604.27264", "pdf_url": "https://arxiv.org/pdf/2604.27264", "primary_query": "autonomous-agent-llm" }, { "id": "2604.10516", "title": "Structure-Grounded Knowledge Retrieval via Code Dependencies for Multi-Step Data Reasoning", "url": "https://arxiv.org/abs/2604.10516", "published": "2026-04-12", "updated": "2026-04-26", "authors": [ "Xinyi Huang" ], "categories": [ "cs.CL" ], "topics": [ "agent-evaluation", "coding-agent", "rag", "reasoning" ], "score": 11, "relevance": "medium", "matched_queries": [ "function-calling" ], "arxiv_id": "2604.10516", "source": "arxiv", "source_id": "arxiv:2604.10516", "pdf_url": "https://arxiv.org/pdf/2604.10516", "primary_query": "function-calling" }, { "id": "2604.07960", "title": "TOOLCAD: Exploring Tool-Using Large Language Models in Text-to-CAD Generation with Reinforcement Learning", "url": "https://arxiv.org/abs/2604.07960", "published": "2026-04-09", "updated": "2026-04-20", "authors": [ "Yifei Gong", "Xing Wu", "Wenda Liu", "Kang Tu" ], "categories": [ "cs.CV", "cs.AI", "cs.CL" ], "topics": [ "planning", "reasoning", "tool-use" ], "score": 11, "relevance": "medium", "matched_queries": [ "language-agent" ], "arxiv_id": "2604.07960", "source": "arxiv", "source_id": "arxiv:2604.07960", "pdf_url": "https://arxiv.org/pdf/2604.07960", "primary_query": "language-agent" }, { "id": "2603.25723", "title": "Natural-Language Agent Harnesses", "url": "https://arxiv.org/abs/2603.25723", "published": "2026-03-26", "updated": "2026-05-18", "authors": [ "Linyue Pan", "Lexiao Zou", "Shuo Guo", "Jingchen Ni", "Hai-Tao Zheng" ], "categories": [ "cs.CL", "cs.AI" ], "topics": [ "agent-evaluation", "coding-agent" ], "score": 11, "relevance": "medium", "matched_queries": [ "language-agent" ], "arxiv_id": "2603.25723", "source": "arxiv", "source_id": "arxiv:2603.25723", "pdf_url": "https://arxiv.org/pdf/2603.25723", "primary_query": "language-agent" }, { "id": "2603.17170", "title": "PAuth - Precise Task-Scoped Authorization For Agents", "url": "https://arxiv.org/abs/2603.17170", "published": "2026-03-17", "updated": "2026-03-17", "authors": [ "Reshabh K Sharma", "Linxi Jiang", "Zhiqiang Lin", "Shuo Chen" ], "categories": [ "cs.CR", "cs.AI", "cs.PL" ], "topics": [ "agent-evaluation", "agent-safety", "reasoning" ], "score": 11, "relevance": "medium", "matched_queries": [ "agent-safety" ], "arxiv_id": "2603.17170", "source": "arxiv", "source_id": "arxiv:2603.17170", "pdf_url": "https://arxiv.org/pdf/2603.17170", "primary_query": "agent-safety" }, { "id": "2602.22523", "title": "Cognitive Models and AI Algorithms Provide Templates for Designing Language Agents", "url": "https://arxiv.org/abs/2602.22523", "published": "2026-02-26", "updated": "2026-02-26", "authors": [ "Ryan Liu", "Dilip Arumugam", "Cedegao E. Zhang", "Sean Escola", "Xaq Pitkow", "Thomas L. Griffiths" ], "categories": [ "cs.AI", "cs.CL", "q-bio.NC" ], "topics": [ "tool-use" ], "score": 11, "relevance": "medium", "matched_queries": [ "language-agent" ], "arxiv_id": "2602.22523", "source": "arxiv", "source_id": "arxiv:2602.22523", "pdf_url": "https://arxiv.org/pdf/2602.22523", "primary_query": "language-agent" }, { "id": "2602.14364", "title": "A Trajectory-Based Safety Audit of Clawdbot (OpenClaw)", "url": "https://arxiv.org/abs/2602.14364", "published": "2026-02-16", "updated": "2026-02-16", "authors": [ "Tianyu Chen", "Dongrui Liu", "Xia Hu", "Jingyi Yu", "Wenjie Wang" ], "categories": [ "cs.CR", "cs.AI" ], "topics": [ "agent-evaluation", "agent-safety", "computer-use", "tool-use", "workflow-agent" ], "score": 11, "relevance": "medium", "matched_queries": [ "agent-safety" ], "arxiv_id": "2602.14364", "source": "arxiv", "source_id": "arxiv:2602.14364", "pdf_url": "https://arxiv.org/pdf/2602.14364", "primary_query": "agent-safety" }, { "id": "2602.10007", "title": "A Collaborative Safety Shield for Safe and Efficient CAV Lane Changes in Congested On-Ramp Merging", "url": "https://arxiv.org/abs/2602.10007", "published": "2026-02-10", "updated": "2026-02-10", "authors": [ "Bharathkumar Hegde", "Melanie Bouroche" ], "categories": [ "cs.RO", "cs.AI", "cs.MA", "eess.SY" ], "topics": [ "agent-evaluation", "agent-safety", "multi-agent", "rag", "tool-use", "world-model" ], "score": 11, "relevance": "medium", "matched_queries": [ "agent-safety" ], "arxiv_id": "2602.10007", "source": "arxiv", "source_id": "arxiv:2602.10007", "pdf_url": "https://arxiv.org/pdf/2602.10007", "primary_query": "agent-safety" }, { "id": "2604.09554", "title": "LABBench2: An Improved Benchmark for AI Systems Performing Biology Research", "url": "https://arxiv.org/abs/2604.09554", "published": "2026-02-04", "updated": "2026-05-05", "authors": [ "Jon M Laurent", "Albert Bou", "Michael Pieler", "Conor Igoe", "Alex Andonian", "Siddharth Narayanan", "James Braza", "Alexandros Sanchez Vassopoulos", "Jacob L Steenwyk", "Blake Lash", "Andrew D White", "Samuel G Rodriques" ], "categories": [ "cs.AI", "cs.CL", "cs.LG" ], "topics": [ "agent-evaluation", "reasoning", "tool-use" ], "score": 11, "relevance": "medium", "matched_queries": [ "language-agent" ], "arxiv_id": "2604.09554", "source": "arxiv", "source_id": "arxiv:2604.09554", "pdf_url": "https://arxiv.org/pdf/2604.09554", "primary_query": "language-agent" }, { "id": "2603.00030", "title": "SimpleTool: Parallel Decoding for Real-Time LLM Function Calling", "url": "https://arxiv.org/abs/2603.00030", "published": "2026-02-04", "updated": "2026-02-04", "authors": [ "Xiaoxin Shi", "Jiaxin Wan", "Linkang Dong", "Wei Jiang", "Yue Liu", "Zengfeng Huang" ], "categories": [ "cs.CL" ], "topics": [ "agent-evaluation", "coding-agent", "computer-use", "embodied-agent", "tool-use" ], "score": 11, "relevance": "medium", "matched_queries": [ "function-calling" ], "arxiv_id": "2603.00030", "source": "arxiv", "source_id": "arxiv:2603.00030", "pdf_url": "https://arxiv.org/pdf/2603.00030", "primary_query": "function-calling" }, { "id": "2601.18282", "title": "Think-Augmented Function Calling: Improving LLM Parameter Accuracy Through Embedded Reasoning", "url": "https://arxiv.org/abs/2601.18282", "published": "2026-01-26", "updated": "2026-02-06", "authors": [ "Lei Wei", "Xiao Peng", "Jinpeng Ou", "Bin Wang" ], "categories": [ "cs.AI", "cs.CL", "cs.LG" ], "topics": [ "agent-evaluation", "computer-use", "reasoning", "tool-use" ], "score": 11, "relevance": "medium", "matched_queries": [ "function-calling" ], "arxiv_id": "2601.18282", "source": "arxiv", "source_id": "arxiv:2601.18282", "pdf_url": "https://arxiv.org/pdf/2601.18282", "primary_query": "function-calling" }, { "id": "2603.21013", "title": "A Framework for Low-Latency, LLM-driven Multimodal Interaction on the Pepper Robot", "url": "https://arxiv.org/abs/2603.21013", "published": "2026-01-09", "updated": "2026-01-09", "authors": [ "Erich Studerus", "Vivienne Jia Zhong", "Stephan Vonschallen" ], "categories": [ "cs.AI", "cs.LG", "cs.RO" ], "topics": [ "computer-use", "embodied-agent", "planning", "rag", "tool-use" ], "score": 11, "relevance": "medium", "matched_queries": [ "function-calling" ], "arxiv_id": "2603.21013", "source": "arxiv", "source_id": "arxiv:2603.21013", "pdf_url": "https://arxiv.org/pdf/2603.21013", "primary_query": "function-calling" }, { "id": "2512.11277", "title": "When Actions Teach You to Think: Reasoning-Action Synergy via Reinforcement Learning in Conversational Agents", "url": "https://arxiv.org/abs/2512.11277", "published": "2025-12-12", "updated": "2025-12-12", "authors": [ "Mrinal Rawat", "Arkajyoti Chakraborty", "Neha Gupta", "Roberto Pieraccini" ], "categories": [ "cs.CL", "cs.LG" ], "topics": [ "computer-use", "rag", "reasoning", "tool-use" ], "score": 11, "relevance": "medium", "matched_queries": [ "function-calling" ], "arxiv_id": "2512.11277", "source": "arxiv", "source_id": "arxiv:2512.11277", "pdf_url": "https://arxiv.org/pdf/2512.11277", "primary_query": "function-calling" }, { "id": "2512.01270", "title": "Egent: An Autonomous Agent for Equivalent Width Measurement", "url": "https://arxiv.org/abs/2512.01270", "published": "2025-12-01", "updated": "2026-05-06", "authors": [ "Yuan-Sen Ting", "Serat Mahmud Saad", "Fan Liu", "Yuting Shen" ], "categories": [ "astro-ph.IM", "astro-ph.GA", "astro-ph.SR" ], "topics": [ "rag", "reasoning", "tool-use" ], "score": 11, "relevance": "medium", "matched_queries": [ "function-calling" ], "arxiv_id": "2512.01270", "source": "arxiv", "source_id": "arxiv:2512.01270", "pdf_url": "https://arxiv.org/pdf/2512.01270", "primary_query": "function-calling" }, { "id": "2509.00482", "title": "Talk Less, Call Right: Enhancing Role-Play LLM Agents with Automatic Prompt Optimization and Role Prompting", "url": "https://arxiv.org/abs/2509.00482", "published": "2025-08-30", "updated": "2025-10-12", "authors": [ "Saksorn Ruangtanusak", "Pittawat Taveekitworachai", "Kunat Pipatanakul" ], "categories": [ "cs.CL", "cs.AI", "cs.HC" ], "topics": [ "tool-use" ], "score": 11, "relevance": "medium", "matched_queries": [ "function-calling" ], "arxiv_id": "2509.00482", "source": "arxiv", "source_id": "arxiv:2509.00482", "pdf_url": "https://arxiv.org/pdf/2509.00482", "primary_query": "function-calling" }, { "id": "2508.20931", "title": "How Can Input Reformulation Improve Tool Usage Accuracy in a Complex Dynamic Environment? A Study on $τ$-bench", "url": "https://arxiv.org/abs/2508.20931", "published": "2025-08-28", "updated": "2025-09-01", "authors": [ "Venkatesh Mishra", "Amir Saeidi", "Satyam Raj", "Mutsumi Nakamura", "Jayanth Srinivasa", "Gaowen Liu", "Ali Payani", "Chitta Baral" ], "categories": [ "cs.CL" ], "topics": [ "multi-agent", "planning", "reasoning", "tool-use" ], "score": 11, "relevance": "medium", "matched_queries": [ "function-calling" ], "arxiv_id": "2508.20931", "source": "arxiv", "source_id": "arxiv:2508.20931", "pdf_url": "https://arxiv.org/pdf/2508.20931", "primary_query": "function-calling" }, { "id": "2607.05999", "title": "AgoraSim: A Hybrid Agent-Based Modeling Framework", "url": "https://arxiv.org/abs/2607.05999", "published": "2026-07-07", "updated": "2026-07-07", "authors": [ "Chung-Chi Chen" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "tool-use", "world-model" ], "score": 10, "relevance": "medium", "matched_queries": [ "llm-agent" ], "arxiv_id": "2607.05999", "source": "arxiv", "source_id": "arxiv:2607.05999", "pdf_url": "https://arxiv.org/pdf/2607.05999", "primary_query": "llm-agent" }, { "id": "2607.05975", "title": "MCP-Enabled Agentic AI for Autonomous IPoDWDM Network Lifecycle Automation", "url": "https://arxiv.org/abs/2607.05975", "published": "2026-07-07", "updated": "2026-07-07", "authors": [ "Chunmin Xia", "Jakub Harbaczewski", "Nikhil Dsilva", "Julie Raulin", "Dominic Schneider", "Achim Autenrieth" ], "categories": [ "cs.NI", "cs.AI", "cs.MA", "eess.SY" ], "topics": [ "tool-use", "workflow-agent" ], "score": 10, "relevance": "medium", "matched_queries": [ "agentic-ai" ], "arxiv_id": "2607.05975", "source": "arxiv", "source_id": "arxiv:2607.05975", "pdf_url": "https://arxiv.org/pdf/2607.05975", "primary_query": "agentic-ai" }, { "id": "2607.05958", "title": "Agentic AI for IPoDWDM Network Lifecycle Automation: An MCP-Enabled Architecture", "url": "https://arxiv.org/abs/2607.05958", "published": "2026-07-07", "updated": "2026-07-07", "authors": [ "Chunmin Xia", "Jakub Harbaczewski", "Nikhil Dsilva", "Julie Raulin", "Dominic Schneider", "Achim Autenrieth" ], "categories": [ "cs.NI", "cs.AI", "eess.SP", "eess.SY" ], "topics": [ "tool-use", "workflow-agent" ], "score": 10, "relevance": "medium", "matched_queries": [ "agentic-ai" ], "arxiv_id": "2607.05958", "source": "arxiv", "source_id": "arxiv:2607.05958", "pdf_url": "https://arxiv.org/pdf/2607.05958", "primary_query": "agentic-ai" }, { "id": "2607.06155", "title": "When Does Tool Use Increase the Expressive Power of Finite-Precision Recurrent Models?", "url": "https://arxiv.org/abs/2607.06155", "published": "2026-07-07", "updated": "2026-07-07", "authors": [ "Nikola Zubić", "Qian Li", "Yuyi Wang", "Davide Scaramuzza" ], "categories": [ "cs.FL", "cs.CC", "cs.CL" ], "topics": [ "memory", "tool-use", "world-model" ], "score": 10, "relevance": "medium", "matched_queries": [ "tool-use" ], "arxiv_id": "2607.06155", "source": "arxiv", "source_id": "arxiv:2607.06155", "pdf_url": "https://arxiv.org/pdf/2607.06155", "primary_query": "tool-use" }, { "id": "2607.05762", "title": "Articulating Assumptions in AI-Generated Scientific Analyses through Task Decomposition", "url": "https://arxiv.org/abs/2607.05762", "published": "2026-07-07", "updated": "2026-07-07", "authors": [ "Ahmed Hammad", "Mihoko Nojiri" ], "categories": [ "cs.SE", "hep-ex", "hep-ph" ], "topics": [ "computer-use", "multi-agent", "planning", "workflow-agent" ], "score": 10, "relevance": "medium", "matched_queries": [ "multi-agent-llm" ], "arxiv_id": "2607.05762", "source": "arxiv", "source_id": "arxiv:2607.05762", "pdf_url": "https://arxiv.org/pdf/2607.05762", "primary_query": "multi-agent-llm" }, { "id": "2607.04574", "title": "A Few Teacher Steps Go a Long Way: Cost-Efficient On-Policy Data Augmentation for Agent Post-Training", "url": "https://arxiv.org/abs/2607.04574", "published": "2026-07-06", "updated": "2026-07-06", "authors": [ "Junze Ye", "Jiayi Cheng", "Miao Lu", "Michal Mankowski", "Jose Blanchet", "Mohsen Bayati" ], "categories": [ "cs.LG", "cs.AI" ], "topics": [ "rag", "tool-use" ], "score": 10, "relevance": "medium", "matched_queries": [ "llm-agent" ], "arxiv_id": "2607.04574", "source": "arxiv", "source_id": "arxiv:2607.04574", "pdf_url": "https://arxiv.org/pdf/2607.04574", "primary_query": "llm-agent" }, { "id": "2607.05277", "title": "Untrusted Content Masking for Web Agents with Security Guarantees", "url": "https://arxiv.org/abs/2607.05277", "published": "2026-07-06", "updated": "2026-07-06", "authors": [ "Kristina Nikolić", "Egor Zverev", "Javier Rando", "Matthew Jagielski", "Edoardo Debenedetti", "Florian Tramèr" ], "categories": [ "cs.CR", "cs.LG" ], "topics": [ "agent-safety", "computer-use", "rag", "tool-use" ], "score": 10, "relevance": "medium", "matched_queries": [ "tool-use", "web-gui-agent" ], "arxiv_id": "2607.05277", "source": "arxiv", "source_id": "arxiv:2607.05277", "pdf_url": "https://arxiv.org/pdf/2607.05277", "primary_query": "tool-use" }, { "id": "2607.04613", "title": "Governed Individuation: Cryptographically Decoupling an Agent's Learning from Its Authority", "url": "https://arxiv.org/abs/2607.04613", "published": "2026-07-06", "updated": "2026-07-06", "authors": [ "Xue Qin", "Simin Luan", "Cong Yang", "Zhijun Li" ], "categories": [ "cs.AI", "cs.CR" ], "topics": [ "agent-evaluation", "agent-safety", "tool-use" ], "score": 10, "relevance": "medium", "matched_queries": [ "tool-use" ], "arxiv_id": "2607.04613", "source": "arxiv", "source_id": "arxiv:2607.04613", "pdf_url": "https://arxiv.org/pdf/2607.04613", "primary_query": "tool-use" }, { "id": "2607.04371", "title": "Nemotron-Labs-3-Puzzle-75B-A9B: Compressing Hybrid MoE LLMs", "url": "https://arxiv.org/abs/2607.04371", "published": "2026-07-05", "updated": "2026-07-07", "authors": [ "Akhiad Bercovich", "Talor Abramovich", "Daniel Afrimi", "Shay Aharon", "Nir Ailon", "Vladimir Anisimov", "Omer Ullman Argov", "Maor Ashkenazi", "Tomer Asida", "Nave Assaf", "Tomer Bar Natan", "Alexander Bukharin", "Grzegorz Chlebus", "Marcin Chochowski", "Eric Chung", "Mohammad Dabbah", "Carlo del Mundo", "Ewa Dobrowolska", "Ido Galil", "Yaniv Galron", "Amnon Geifman", "Yonatan Geifman", "Izik Golan", "Alex Gronskiy", "Tomasz Grzegorzek", "Netanel Haber", "Lior Kadoch", "Grzegorz Karch", "Tomer Keren", "Abhinav Khattar", "Amir Klein", "Tugrul Konuk", "Roi Koren", "Daniel Korzekwa", "Shaun Kotek", "Konstantinos Krommydas", "Itay Levy", "Ofri Masad", "Yoav Miron", "Pavlo Molchanov", "Shahar Mor", "Zach Moshe", "Saurav Muralidharan", "Najeeb Nabwani", "Besmira Nushi", "Mostofa Patwary", "Omri Puny", "Johannes Rausch", "Tomer Ronen", "Sepehr Sameni", "Itamar Schen", "Elad Segal", "Daniel Serebrenik", "Ido Shahaf", "Soumye Singhal", "Daniil Sorokin", "Sharath Turuvekere Sreenivas", "Marta Stepniewska-Dziubinska", "Ali Taghibakhshi", "Nima Tajbakhsh", "Oren Tropp", "Dor Tzur", "Anna Warno", "Yi-Fu Wu", "Michal Zawalski", "Jiaqi Zeng", "Yian Zhang", "Ran Zilberstein", "Amit Zuker", "Ran El-Yaniv" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "coding-agent", "reasoning" ], "score": 10, "relevance": "medium", "matched_queries": [ "agent-evaluation" ], "arxiv_id": "2607.04371", "source": "arxiv", "source_id": "arxiv:2607.04371", "pdf_url": "https://arxiv.org/pdf/2607.04371", "primary_query": "agent-evaluation" }, { "id": "2607.03238", "title": "An Empirical Study of Downstream Adaptation for Agent Skills", "url": "https://arxiv.org/abs/2607.03238", "published": "2026-07-03", "updated": "2026-07-03", "authors": [ "Xinjian Wu", "Jingzhi Gong", "Gunel Jahangirova", "Zhenpeng Chen", "Jie M. Zhang" ], "categories": [ "cs.SE" ], "topics": [ "agent-safety", "tool-use", "workflow-agent" ], "score": 10, "relevance": "medium", "matched_queries": [ "llm-agent" ], "arxiv_id": "2607.03238", "source": "arxiv", "source_id": "arxiv:2607.03238", "pdf_url": "https://arxiv.org/pdf/2607.03238", "primary_query": "llm-agent" }, { "id": "2607.03048", "title": "Compression, structure, and executor capability: a controlled real-cost decomposition of language-model agent skill optimisation", "url": "https://arxiv.org/abs/2607.03048", "published": "2026-07-03", "updated": "2026-07-03", "authors": [ "Xiaonan Xu", "Wenjing Wu" ], "categories": [ "cs.SE" ], "topics": [ "computer-use", "reasoning", "tool-use" ], "score": 10, "relevance": "medium", "matched_queries": [ "tool-use" ], "arxiv_id": "2607.03048", "source": "arxiv", "source_id": "arxiv:2607.03048", "pdf_url": "https://arxiv.org/pdf/2607.03048", "primary_query": "tool-use" }, { "id": "2607.02873", "title": "Determinants and Limits of LLM Security-Tool Orchestration: A Study with HexStrike-AI", "url": "https://arxiv.org/abs/2607.02873", "published": "2026-07-03", "updated": "2026-07-03", "authors": [ "Romain Gerard", "Assmaa Zeghaider", "Yan Guo" ], "categories": [ "cs.SE", "cs.AI", "cs.CR" ], "topics": [ "agent-evaluation", "agent-safety", "reasoning", "tool-use" ], "score": 10, "relevance": "medium", "matched_queries": [ "tool-use" ], "arxiv_id": "2607.02873", "source": "arxiv", "source_id": "arxiv:2607.02873", "pdf_url": "https://arxiv.org/pdf/2607.02873", "primary_query": "tool-use" }, { "id": "2607.03028", "title": "HETERQA: Benchmarking Record Retrieval over Multiple Heterogeneous Sources", "url": "https://arxiv.org/abs/2607.03028", "published": "2026-07-03", "updated": "2026-07-03", "authors": [ "Yaodong Su", "Hanchang Li", "Quanqing Xu", "Chuanhui Yang", "Yixiang Fang" ], "categories": [ "cs.IR" ], "topics": [ "agent-evaluation", "rag", "tool-use" ], "score": 10, "relevance": "medium", "matched_queries": [ "rag-agent" ], "arxiv_id": "2607.03028", "source": "arxiv", "source_id": "arxiv:2607.03028", "pdf_url": "https://arxiv.org/pdf/2607.03028", "primary_query": "rag-agent" }, { "id": "2607.02459", "title": "Language Models as Measurement Apparatus for Culture", "url": "https://arxiv.org/abs/2607.02459", "published": "2026-07-02", "updated": "2026-07-02", "authors": [ "Kent K. Chang" ], "categories": [ "cs.CL" ], "topics": [ "agent-evaluation", "tool-use", "workflow-agent" ], "score": 10, "relevance": "medium", "matched_queries": [ "agentic-ai" ], "arxiv_id": "2607.02459", "source": "arxiv", "source_id": "arxiv:2607.02459", "pdf_url": "https://arxiv.org/pdf/2607.02459", "primary_query": "agentic-ai" }, { "id": "2607.02376", "title": "Hardware-Enforced Semantic Coordination for Safety-Critical Real-Time Autonomous Systems", "url": "https://arxiv.org/abs/2607.02376", "published": "2026-07-02", "updated": "2026-07-02", "authors": [ "Uwe M. Borghoff", "Paolo Bottoni", "Remo Pareschi" ], "categories": [ "cs.AI", "cs.MA" ], "topics": [ "agent-safety", "reasoning", "tool-use", "world-model" ], "score": 10, "relevance": "medium", "matched_queries": [ "agentic-ai" ], "arxiv_id": "2607.02376", "source": "arxiv", "source_id": "arxiv:2607.02376", "pdf_url": "https://arxiv.org/pdf/2607.02376", "primary_query": "agentic-ai" }, { "id": "2607.02370", "title": "Understanding Agent-Based Patching of Compiler Missed Optimizations", "url": "https://arxiv.org/abs/2607.02370", "published": "2026-07-02", "updated": "2026-07-03", "authors": [ "Batu Guan", "Zirui Wang", "Shaohua Li" ], "categories": [ "cs.SE", "cs.AI" ], "topics": [ "agent-evaluation", "coding-agent", "rag" ], "score": 10, "relevance": "medium", "matched_queries": [ "coding-agent" ], "arxiv_id": "2607.02370", "source": "arxiv", "source_id": "arxiv:2607.02370", "pdf_url": "https://arxiv.org/pdf/2607.02370", "primary_query": "coding-agent" }, { "id": "2607.02134", "title": "Coding-agents can replicate scientific machine learning papers", "url": "https://arxiv.org/abs/2607.02134", "published": "2026-07-02", "updated": "2026-07-02", "authors": [ "Atharva Hans", "Ilias Bilionis" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "coding-agent", "rag", "workflow-agent" ], "score": 10, "relevance": "medium", "matched_queries": [ "coding-agent" ], "arxiv_id": "2607.02134", "source": "arxiv", "source_id": "arxiv:2607.02134", "pdf_url": "https://arxiv.org/pdf/2607.02134", "primary_query": "coding-agent" }, { "id": "2607.01597", "title": "A Single Patch Is Not Enough: Deterministic Fusion of Repair Candidates", "url": "https://arxiv.org/abs/2607.01597", "published": "2026-07-02", "updated": "2026-07-02", "authors": [ "Boyang Yang", "Xiangliang Hu", "Luyao Ren", "Yanjun Chen", "Bach Le", "Tegawendé F. Bissyandé", "Haoye Tian" ], "categories": [ "cs.SE" ], "topics": [ "agent-evaluation", "coding-agent" ], "score": 10, "relevance": "medium", "matched_queries": [ "coding-agent" ], "arxiv_id": "2607.01597", "source": "arxiv", "source_id": "arxiv:2607.01597", "pdf_url": "https://arxiv.org/pdf/2607.01597", "primary_query": "coding-agent" }, { "id": "2607.01557", "title": "DiPS: Dialogue Policy Selection for High-Stakes Persuasion Agents", "url": "https://arxiv.org/abs/2607.01557", "published": "2026-07-02", "updated": "2026-07-03", "authors": [ "Tianyi Zhang", "Mousumi Das", "Abrar Anwar", "Jesse Thomason", "David Traum" ], "categories": [ "cs.CL", "cs.AI" ], "topics": [ "agent-evaluation", "rag", "tool-use" ], "score": 10, "relevance": "medium", "matched_queries": [ "rag-agent" ], "arxiv_id": "2607.01557", "source": "arxiv", "source_id": "arxiv:2607.01557", "pdf_url": "https://arxiv.org/pdf/2607.01557", "primary_query": "rag-agent" }, { "id": "2607.01456", "title": "From Anatomy to Smells: An Empirical Study of SKILL.md in Agent Skills", "url": "https://arxiv.org/abs/2607.01456", "published": "2026-07-01", "updated": "2026-07-03", "authors": [ "David Boram Hong", "Aaron Imani", "Iftekhar Ahmed" ], "categories": [ "cs.SE" ], "topics": [ "computer-use", "tool-use" ], "score": 10, "relevance": "medium", "matched_queries": [ "llm-agent" ], "arxiv_id": "2607.01456", "source": "arxiv", "source_id": "arxiv:2607.01456", "pdf_url": "https://arxiv.org/pdf/2607.01456", "primary_query": "llm-agent" }, { "id": "2607.00394", "title": "When Classic Cache Policies Fail: Learning-Augmented Replacement for Semantic Retrieval Buffers", "url": "https://arxiv.org/abs/2607.00394", "published": "2026-07-01", "updated": "2026-07-01", "authors": [ "Yushi Sun", "Bowen Cao", "Wai Lam" ], "categories": [ "cs.DB", "cs.CL" ], "topics": [ "agent-evaluation", "memory", "rag" ], "score": 10, "relevance": "medium", "matched_queries": [ "llm-agent" ], "arxiv_id": "2607.00394", "source": "arxiv", "source_id": "arxiv:2607.00394", "pdf_url": "https://arxiv.org/pdf/2607.00394", "primary_query": "llm-agent" }, { "id": "2607.01507", "title": "The Agentic Garden of Forking Paths", "url": "https://arxiv.org/abs/2607.01507", "published": "2026-07-01", "updated": "2026-07-01", "authors": [ "Jiacheng Miao", "Jonathan K Pritchard", "James Zou" ], "categories": [ "cs.AI", "stat.ME" ], "topics": [ "agent-evaluation" ], "score": 10, "relevance": "medium", "matched_queries": [ "ai-agent" ], "arxiv_id": "2607.01507", "source": "arxiv", "source_id": "arxiv:2607.01507", "pdf_url": "https://arxiv.org/pdf/2607.01507", "primary_query": "ai-agent" }, { "id": "2607.00751", "title": "SessionBound: Turning Enterprise Task Approval into Budgeted Database Sessions", "url": "https://arxiv.org/abs/2607.00751", "published": "2026-07-01", "updated": "2026-07-01", "authors": [ "Minmin Wu" ], "categories": [ "cs.DB", "cs.CR" ], "topics": [ "agent-evaluation", "planning", "workflow-agent" ], "score": 10, "relevance": "medium", "matched_queries": [ "ai-agent" ], "arxiv_id": "2607.00751", "source": "arxiv", "source_id": "arxiv:2607.00751", "pdf_url": "https://arxiv.org/pdf/2607.00751", "primary_query": "ai-agent" }, { "id": "2607.01465", "title": "Beyond Next-Token Prediction: An RLVR Proof of Concept for Tool-Use Agents on Atlassian Workflows", "url": "https://arxiv.org/abs/2607.01465", "published": "2026-07-01", "updated": "2026-07-01", "authors": [ "Karthikeya Aditya Vissa", "Sankalp Mane", "Ananya Mantravadi", "Harshit Rajgarhia", "Abhishek Mukherji" ], "categories": [ "cs.AI" ], "topics": [ "rag", "tool-use", "workflow-agent" ], "score": 10, "relevance": "medium", "matched_queries": [ "tool-use" ], "arxiv_id": "2607.01465", "source": "arxiv", "source_id": "arxiv:2607.01465", "pdf_url": "https://arxiv.org/pdf/2607.01465", "primary_query": "tool-use" }, { "id": "2607.00871", "title": "Self-Evolving Agents with Anytime-Valid Certificates", "url": "https://arxiv.org/abs/2607.00871", "published": "2026-07-01", "updated": "2026-07-01", "authors": [ "Biswa Sengupta" ], "categories": [ "cs.AI", "cs.CL" ], "topics": [ "agent-evaluation", "coding-agent", "reasoning" ], "score": 10, "relevance": "medium", "matched_queries": [ "coding-agent" ], "arxiv_id": "2607.00871", "source": "arxiv", "source_id": "arxiv:2607.00871", "pdf_url": "https://arxiv.org/pdf/2607.00871", "primary_query": "coding-agent" }, { "id": "2607.01044", "title": "Robots Ask the Way: Communication-Enabled Social Navigation", "url": "https://arxiv.org/abs/2607.01044", "published": "2026-07-01", "updated": "2026-07-01", "authors": [ "Valentino Sacco", "Luca Scofano", "Indro Spinelli", "Fabio Galasso" ], "categories": [ "cs.RO" ], "topics": [ "agent-evaluation", "embodied-agent", "multi-agent", "tool-use" ], "score": 10, "relevance": "medium", "matched_queries": [ "multi-agent-llm" ], "arxiv_id": "2607.01044", "source": "arxiv", "source_id": "arxiv:2607.01044", "pdf_url": "https://arxiv.org/pdf/2607.01044", "primary_query": "multi-agent-llm" }, { "id": "2607.00245", "title": "Agent-to-Agent Finance: Blockchain Payments and Trust Infrastructure for Autonomous AI Agents", "url": "https://arxiv.org/abs/2607.00245", "published": "2026-06-30", "updated": "2026-06-30", "authors": [ "Hui Gong" ], "categories": [ "q-fin.GN" ], "topics": [ "rag", "tool-use" ], "score": 10, "relevance": "medium", "matched_queries": [ "ai-agent" ], "arxiv_id": "2607.00245", "source": "arxiv", "source_id": "arxiv:2607.00245", "pdf_url": "https://arxiv.org/pdf/2607.00245", "primary_query": "ai-agent" }, { "id": "2606.31272", "title": "The Decomposition Is the Fingerprint: Per-Component Identity for Agent Skills", "url": "https://arxiv.org/abs/2606.31272", "published": "2026-06-30", "updated": "2026-06-30", "authors": [ "Hongliang Liu", "Yuhao Wu", "Tung-Ling Li" ], "categories": [ "cs.CR", "cs.CL", "cs.LG" ], "topics": [ "agent-evaluation", "agent-safety", "tool-use" ], "score": 10, "relevance": "medium", "matched_queries": [ "ai-agent" ], "arxiv_id": "2606.31272", "source": "arxiv", "source_id": "arxiv:2606.31272", "pdf_url": "https://arxiv.org/pdf/2606.31272", "primary_query": "ai-agent" }, { "id": "2606.31036", "title": "Teaching LLMs to Recommend and Defer in Underrepresented Epilepsy Care", "url": "https://arxiv.org/abs/2606.31036", "published": "2026-06-30", "updated": "2026-06-30", "authors": [ "Shreyas Rajesh", "Kartik Sharma", "Tonmoy Monsoor", "Mehmet Yigit Turali", "Richard Idro", "Juliana Kayaga", "Robert Sebunya", "Tracy Tushabe Namata", "Jessica Nichole Pasqua", "Vwani Roychowdhury", "Rajarshi Mazumder" ], "categories": [ "cs.LG" ], "topics": [ "agent-evaluation", "computer-use", "multi-agent", "rag" ], "score": 10, "relevance": "medium", "matched_queries": [ "multi-agent-llm" ], "arxiv_id": "2606.31036", "source": "arxiv", "source_id": "arxiv:2606.31036", "pdf_url": "https://arxiv.org/pdf/2606.31036", "primary_query": "multi-agent-llm" }, { "id": "2606.30801", "title": "Using AI Agents to Automate Black-Box Audits of Personalization Algorithms at Scale", "url": "https://arxiv.org/abs/2606.30801", "published": "2026-06-29", "updated": "2026-06-29", "authors": [ "Alessandro Morosini", "Sarah H. Cen", "Andrew Ilyas", "Hedi Driss", "Aleksander Mądry", "Chara Podimata" ], "categories": [ "cs.CL", "cs.CY", "cs.LG", "cs.SI" ], "topics": [ "reasoning", "tool-use" ], "score": 10, "relevance": "medium", "matched_queries": [ "ai-agent" ], "arxiv_id": "2606.30801", "source": "arxiv", "source_id": "arxiv:2606.30801", "pdf_url": "https://arxiv.org/pdf/2606.30801", "primary_query": "ai-agent" }, { "id": "2606.30182", "title": "MirrorCode: AI can rebuild entire programs from behavior alone", "url": "https://arxiv.org/abs/2606.30182", "published": "2026-06-29", "updated": "2026-06-29", "authors": [ "Tom Adamczewski", "David Owen", "David Rein", "Florian Brand", "Giles Edkins", "Allen Hart", "Daniel O'Connell" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "coding-agent", "computer-use", "planning", "tool-use" ], "score": 10, "relevance": "medium", "matched_queries": [ "ai-agent" ], "arxiv_id": "2606.30182", "source": "arxiv", "source_id": "arxiv:2606.30182", "pdf_url": "https://arxiv.org/pdf/2606.30182", "primary_query": "ai-agent" }, { "id": "2606.30911", "title": "Why Solve It Twice? Hierarchical Accumulation of Skills for Transfer-Efficient ML Engineering", "url": "https://arxiv.org/abs/2606.30911", "published": "2026-06-29", "updated": "2026-07-01", "authors": [ "Yongbin Kim", "Yashar Talebirad", "Osmar R. Zaiane" ], "categories": [ "cs.AI", "cs.LG", "cs.MA" ], "topics": [ "agent-evaluation", "multi-agent", "tool-use" ], "score": 10, "relevance": "medium", "matched_queries": [ "multi-agent-llm" ], "arxiv_id": "2606.30911", "source": "arxiv", "source_id": "arxiv:2606.30911", "pdf_url": "https://arxiv.org/pdf/2606.30911", "primary_query": "multi-agent-llm" }, { "id": "2606.29279", "title": "Manufactured Confidence: How Memory Consolidation Turns Hearsay into Confident Facts", "url": "https://arxiv.org/abs/2606.29279", "published": "2026-06-28", "updated": "2026-06-28", "authors": [ "Alex Kwon" ], "categories": [ "cs.CR", "cs.AI", "cs.CL" ], "topics": [ "memory" ], "score": 10, "relevance": "medium", "matched_queries": [ "llm-agent" ], "arxiv_id": "2606.29279", "source": "arxiv", "source_id": "arxiv:2606.29279", "pdf_url": "https://arxiv.org/pdf/2606.29279", "primary_query": "llm-agent" }, { "id": "2606.29194", "title": "AI Trading's Alpha Singularity: Emergent Market Reasoning through Agent-to-Agent Self-Evolution", "url": "https://arxiv.org/abs/2606.29194", "published": "2026-06-28", "updated": "2026-06-28", "authors": [ "Yuqi Li", "Siyuan Liu", "Bingjun Liu" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "reasoning" ], "score": 10, "relevance": "medium", "matched_queries": [ "llm-agent" ], "arxiv_id": "2606.29194", "source": "arxiv", "source_id": "arxiv:2606.29194", "pdf_url": "https://arxiv.org/pdf/2606.29194", "primary_query": "llm-agent" }, { "id": "2606.29280", "title": "Deterministic Decisions for High-Stakes AI. A Zero-Egress Pipeline with the Deployability of RAG and the Accuracy of Machine Learning", "url": "https://arxiv.org/abs/2606.29280", "published": "2026-06-28", "updated": "2026-06-28", "authors": [ "Craig Atkinson" ], "categories": [ "cs.LG", "cs.AI", "cs.CL" ], "topics": [ "agent-evaluation", "agent-safety", "rag", "tool-use" ], "score": 10, "relevance": "medium", "matched_queries": [ "rag-agent" ], "arxiv_id": "2606.29280", "source": "arxiv", "source_id": "arxiv:2606.29280", "pdf_url": "https://arxiv.org/pdf/2606.29280", "primary_query": "rag-agent" }, { "id": "2606.29113", "title": "LLM Semantic Signaling Game and Mechanism Design: Systematic Blindness, Awareness Shaping, and Mindset Dynamics", "url": "https://arxiv.org/abs/2606.29113", "published": "2026-06-27", "updated": "2026-06-27", "authors": [ "Quanyan Zhu" ], "categories": [ "cs.GT", "cs.AI", "cs.MA" ], "topics": [ "agent-evaluation", "agent-safety", "computer-use", "tool-use" ], "score": 10, "relevance": "medium", "matched_queries": [ "agentic-ai" ], "arxiv_id": "2606.29113", "source": "arxiv", "source_id": "arxiv:2606.29113", "pdf_url": "https://arxiv.org/pdf/2606.29113", "primary_query": "agentic-ai" }, { "id": "2606.27845", "title": "LLM Agents as Static Level-k Players in Behavioural Games", "url": "https://arxiv.org/abs/2606.27845", "published": "2026-06-26", "updated": "2026-06-26", "authors": [ "Po Han Teo" ], "categories": [ "econ.GN", "econ.TH" ], "topics": [ "agent" ], "score": 10, "relevance": "medium", "matched_queries": [ "llm-agent" ], "arxiv_id": "2606.27845", "source": "arxiv", "source_id": "arxiv:2606.27845", "pdf_url": "https://arxiv.org/pdf/2606.27845", "primary_query": "llm-agent" }, { "id": "2606.27974", "title": "ProMSA:Progressive Multimodal Search Agents for Knowledge-Based Visual Question Answering", "url": "https://arxiv.org/abs/2606.27974", "published": "2026-06-26", "updated": "2026-06-26", "authors": [ "ZhengXian Wu", "Hangrui Xu", "Kai Shi", "Zhuohong Chen", "Yunyao Yu", "Chuanrui Zhang", "Zirui Liao", "Jun Yang", "Zhenyu Yang", "Haonan Lu", "Haoqian Wang" ], "categories": [ "cs.CV", "cs.AI" ], "topics": [ "agent-evaluation", "rag", "reasoning", "tool-use" ], "score": 10, "relevance": "medium", "matched_queries": [ "rag-agent", "tool-use" ], "arxiv_id": "2606.27974", "source": "arxiv", "source_id": "arxiv:2606.27974", "pdf_url": "https://arxiv.org/pdf/2606.27974", "primary_query": "rag-agent" }, { "id": "2606.26959", "title": "The Shift to Agentic AI: Evidence from Codex", "url": "https://arxiv.org/abs/2606.26959", "published": "2026-06-25", "updated": "2026-06-25", "authors": [ "Drew Johnston", "David Holtz", "Alex Martin Richmond", "Christopher Ong", "Prasanna Tambe", "Aaron Chatterji" ], "categories": [ "econ.GN" ], "topics": [ "tool-use", "workflow-agent" ], "score": 10, "relevance": "medium", "matched_queries": [ "agentic-ai" ], "arxiv_id": "2606.26959", "source": "arxiv", "source_id": "arxiv:2606.26959", "pdf_url": "https://arxiv.org/pdf/2606.26959", "primary_query": "agentic-ai" }, { "id": "2606.25605", "title": "Constraint Tax in Open-Weight LLMs: An Empirical Study of Tool Calling Suppression Under Structured Output Constraints", "url": "https://arxiv.org/abs/2606.25605", "published": "2026-06-24", "updated": "2026-06-24", "authors": [ "Fangzheng Li", "Aimin Zhang", "Chen Lv" ], "categories": [ "cs.CL" ], "topics": [ "agent-evaluation", "coding-agent", "planning", "tool-use" ], "score": 10, "relevance": "medium", "matched_queries": [ "tool-use" ], "arxiv_id": "2606.25605", "source": "arxiv", "source_id": "arxiv:2606.25605", "pdf_url": "https://arxiv.org/pdf/2606.25605", "primary_query": "tool-use" }, { "id": "2606.24424", "title": "Explainable AI for Next-Generation Wireless Physical Layer: Basics, State-of-the-Art, and Open Challenges", "url": "https://arxiv.org/abs/2606.24424", "published": "2026-06-23", "updated": "2026-06-23", "authors": [ "Bingnan Xiao", "Shuyan Hu", "Xiaojing Chen", "Zhiyuan Zhai", "Bingcong Li", "Wei Ni", "Xin Wang", "Ekram Hossain" ], "categories": [ "eess.SP" ], "topics": [ "agent-evaluation", "agent-safety", "planning" ], "score": 10, "relevance": "medium", "matched_queries": [ "agentic-ai" ], "arxiv_id": "2606.24424", "source": "arxiv", "source_id": "arxiv:2606.24424", "pdf_url": "https://arxiv.org/pdf/2606.24424", "primary_query": "agentic-ai" }, { "id": "2606.24470", "title": "The Latent Bridge: A Continuous Slow-Fast Channel for Real-Time Game Agents", "url": "https://arxiv.org/abs/2606.24470", "published": "2026-06-23", "updated": "2026-06-23", "authors": [ "Bojie Li", "Noah Shi" ], "categories": [ "cs.AI" ], "topics": [ "computer-use", "planning", "reasoning", "tool-use" ], "score": 10, "relevance": "medium", "matched_queries": [ "web-gui-agent" ], "arxiv_id": "2606.24470", "source": "arxiv", "source_id": "arxiv:2606.24470", "pdf_url": "https://arxiv.org/pdf/2606.24470", "primary_query": "web-gui-agent" }, { "id": "2606.22813", "title": "Active Inference as the Test-Time Scaling Law for Physical AI Agents", "url": "https://arxiv.org/abs/2606.22813", "published": "2026-06-22", "updated": "2026-06-22", "authors": [ "Omar Hashash", "Christo Kurisummoottil Thomas", "Walid Saad", "Merouane Debbah", "Karl Friston", "Adeel Razi" ], "categories": [ "cs.AI" ], "topics": [ "reasoning", "world-model" ], "score": 10, "relevance": "medium", "matched_queries": [ "ai-agent" ], "arxiv_id": "2606.22813", "source": "arxiv", "source_id": "arxiv:2606.22813", "pdf_url": "https://arxiv.org/pdf/2606.22813", "primary_query": "ai-agent" }, { "id": "2606.23138", "title": "Rising From the Ashes: How Agentic AI is Unblocking Challenges in Cybersecurity", "url": "https://arxiv.org/abs/2606.23138", "published": "2026-06-22", "updated": "2026-06-22", "authors": [ "Gabriela F. Ciocarlie", "Kathrin Grosse", "Somesh Jha", "Daryna Oliynyk", "Andrew Paverd", "Christian Wressnegger" ], "categories": [ "cs.CR" ], "topics": [ "agent-safety", "reasoning" ], "score": 10, "relevance": "medium", "matched_queries": [ "agentic-ai" ], "arxiv_id": "2606.23138", "source": "arxiv", "source_id": "arxiv:2606.23138", "pdf_url": "https://arxiv.org/pdf/2606.23138", "primary_query": "agentic-ai" }, { "id": "2606.23175", "title": "Position: Correct Answer, Wrong Mechanism -- When AI Scientists Defend General Claims Their Own Data Contradicts", "url": "https://arxiv.org/abs/2606.23175", "published": "2026-06-22", "updated": "2026-06-22", "authors": [ "Steven Young Eulig" ], "categories": [ "cs.LG" ], "topics": [ "agent-evaluation", "coding-agent", "reasoning", "tool-use", "world-model" ], "score": 10, "relevance": "medium", "matched_queries": [ "coding-agent" ], "arxiv_id": "2606.23175", "source": "arxiv", "source_id": "arxiv:2606.23175", "pdf_url": "https://arxiv.org/pdf/2606.23175", "primary_query": "coding-agent" }, { "id": "2606.22906", "title": "From Fragments to Paths: Task-Level Context Recovery for Large Industrial Codebases", "url": "https://arxiv.org/abs/2606.22906", "published": "2026-06-22", "updated": "2026-06-22", "authors": [ "Jiawei He", "Weisong Sun", "Mengyu Shi", "Jie Jia", "Tong Bian", "Xikai Yang", "Dong Sun" ], "categories": [ "cs.SE", "cs.AI" ], "topics": [ "agent-evaluation", "coding-agent", "rag" ], "score": 10, "relevance": "medium", "matched_queries": [ "coding-agent" ], "arxiv_id": "2606.22906", "source": "arxiv", "source_id": "arxiv:2606.22906", "pdf_url": "https://arxiv.org/pdf/2606.22906", "primary_query": "coding-agent" }, { "id": "2606.23189", "title": "Capable but Careless: Do Computer-Use Agents Follow Contextual Integrity?", "url": "https://arxiv.org/abs/2606.23189", "published": "2026-06-22", "updated": "2026-06-22", "authors": [ "Anmol Goel", "Iryna Gurevych" ], "categories": [ "cs.AI", "cs.CL" ], "topics": [ "agent-evaluation", "agent-safety", "computer-use", "rag" ], "score": 10, "relevance": "medium", "matched_queries": [ "web-gui-agent" ], "arxiv_id": "2606.23189", "source": "arxiv", "source_id": "arxiv:2606.23189", "pdf_url": "https://arxiv.org/pdf/2606.23189", "primary_query": "web-gui-agent" }, { "id": "2606.21959", "title": "OpenBioRQ: Unsolved Biomedical Research Questions for Agents", "url": "https://arxiv.org/abs/2606.21959", "published": "2026-06-20", "updated": "2026-06-20", "authors": [ "Minbyul Jeong" ], "categories": [ "cs.CL" ], "topics": [ "agent-evaluation", "rag", "tool-use" ], "score": 10, "relevance": "medium", "matched_queries": [ "agent-evaluation" ], "arxiv_id": "2606.21959", "source": "arxiv", "source_id": "arxiv:2606.21959", "pdf_url": "https://arxiv.org/pdf/2606.21959", "primary_query": "agent-evaluation" }, { "id": "2606.21151", "title": "Context-Aware Generative AI for Automated Telecom Test Script Generation", "url": "https://arxiv.org/abs/2606.21151", "published": "2026-06-19", "updated": "2026-06-19", "authors": [ "Gautam Prasad", "Chandramohan T. N.", "Joy Bose" ], "categories": [ "cs.SE", "cs.AI", "cs.NI" ], "topics": [ "agent-evaluation", "computer-use", "rag", "reasoning", "tool-use" ], "score": 10, "relevance": "medium", "matched_queries": [ "ai-agent", "rag-agent" ], "arxiv_id": "2606.21151", "source": "arxiv", "source_id": "arxiv:2606.21151", "pdf_url": "https://arxiv.org/pdf/2606.21151", "primary_query": "ai-agent" }, { "id": "2606.21037", "title": "Honeyquest for LLMs: Rethinking Cyber Deception for AI Attackers", "url": "https://arxiv.org/abs/2606.21037", "published": "2026-06-19", "updated": "2026-06-19", "authors": [ "Kerri Prinos", "Lilianne Brush", "Cameron Denton" ], "categories": [ "cs.CR", "cs.CL" ], "topics": [ "agent-evaluation", "reasoning", "tool-use" ], "score": 10, "relevance": "medium", "matched_queries": [ "ai-agent" ], "arxiv_id": "2606.21037", "source": "arxiv", "source_id": "arxiv:2606.21037", "pdf_url": "https://arxiv.org/pdf/2606.21037", "primary_query": "ai-agent" }, { "id": "2606.21804", "title": "Is Agent Code Less Maintainable Than Human Code?", "url": "https://arxiv.org/abs/2606.21804", "published": "2026-06-19", "updated": "2026-06-19", "authors": [ "Shaswat Patel", "Betty Li Hou", "Arun Purohit", "Kai Xu", "Jane Pan", "He He", "Valerie Chen" ], "categories": [ "cs.SE", "cs.AI", "cs.CL" ], "topics": [ "agent-evaluation", "coding-agent", "tool-use" ], "score": 10, "relevance": "medium", "matched_queries": [ "coding-agent" ], "arxiv_id": "2606.21804", "source": "arxiv", "source_id": "arxiv:2606.21804", "pdf_url": "https://arxiv.org/pdf/2606.21804", "primary_query": "coding-agent" }, { "id": "2606.20002", "title": "Connect the Dots: Training LLMs for Long-Lifecycle Agents with Cross-Domain Generalization Via Reinforcement Learning", "url": "https://arxiv.org/abs/2606.20002", "published": "2026-06-18", "updated": "2026-06-18", "authors": [ "Yanxi Chen", "Weijie Shi", "Yuexiang Xie", "Boyi Hu", "Yaliang Li", "Bolin Ding", "Jingren Zhou" ], "categories": [ "cs.LG", "cs.AI", "cs.CL" ], "topics": [ "agent-evaluation" ], "score": 10, "relevance": "medium", "matched_queries": [ "ai-agent" ], "arxiv_id": "2606.20002", "source": "arxiv", "source_id": "arxiv:2606.20002", "pdf_url": "https://arxiv.org/pdf/2606.20002", "primary_query": "ai-agent" }, { "id": "2606.20158", "title": "N-Version Programming with Coding Agents", "url": "https://arxiv.org/abs/2606.20158", "published": "2026-06-18", "updated": "2026-06-18", "authors": [ "Javier Ron", "Benoit Baudry", "Martin Monperrus" ], "categories": [ "cs.SE" ], "topics": [ "agent-evaluation", "coding-agent" ], "score": 10, "relevance": "medium", "matched_queries": [ "coding-agent" ], "arxiv_id": "2606.20158", "source": "arxiv", "source_id": "arxiv:2606.20158", "pdf_url": "https://arxiv.org/pdf/2606.20158", "primary_query": "coding-agent" }, { "id": "2606.19830", "title": "JAMER: Project-Level Code Framework Dataset and Benchmark on Professional Game Engines", "url": "https://arxiv.org/abs/2606.19830", "published": "2026-06-18", "updated": "2026-06-21", "authors": [ "Jianwen Sun", "Chuanhao Li", "Zizhen Li", "Yukang Feng", "Fanrui Zhang", "Yifei Huang", "Yu Dai", "Kaipeng Zhang" ], "categories": [ "cs.SE", "cs.CL" ], "topics": [ "agent-evaluation", "agent-safety", "coding-agent" ], "score": 10, "relevance": "medium", "matched_queries": [ "coding-agent" ], "arxiv_id": "2606.19830", "source": "arxiv", "source_id": "arxiv:2606.19830", "pdf_url": "https://arxiv.org/pdf/2606.19830", "primary_query": "coding-agent" }, { "id": "2606.20363", "title": "Automating SKILL.md Generation for Computer-Using Agents via Interaction Trajectory Mining", "url": "https://arxiv.org/abs/2606.20363", "published": "2026-06-18", "updated": "2026-06-18", "authors": [ "Yuexing Hao", "Xiaomin Li" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "computer-use", "tool-use", "workflow-agent" ], "score": 10, "relevance": "medium", "matched_queries": [ "web-gui-agent" ], "arxiv_id": "2606.20363", "source": "arxiv", "source_id": "arxiv:2606.20363", "pdf_url": "https://arxiv.org/pdf/2606.20363", "primary_query": "web-gui-agent" }, { "id": "2606.20537", "title": "Execution-State Capsules: Graph-Bound Execution-State Checkpoint and Restore for Low-Latency, Small-Batch, On-Device Physical-AI Serving", "url": "https://arxiv.org/abs/2606.20537", "published": "2026-06-18", "updated": "2026-06-18", "authors": [ "Liang Su" ], "categories": [ "cs.LG", "cs.DC" ], "topics": [ "agent-evaluation", "embodied-agent", "planning", "rag" ], "score": 10, "relevance": "medium", "matched_queries": [ "planning-agent" ], "arxiv_id": "2606.20537", "source": "arxiv", "source_id": "arxiv:2606.20537", "pdf_url": "https://arxiv.org/pdf/2606.20537", "primary_query": "planning-agent" }, { "id": "2606.19116", "title": "Towards an Agent-First Web: Redesigning the Web for AI Agents", "url": "https://arxiv.org/abs/2606.19116", "published": "2026-06-17", "updated": "2026-06-17", "authors": [ "Eranga Bandara", "Ross Gore", "Ravi Mukkamala", "Asanga Gunaratna", "Safdar H. Bouk", "Xueping Liang", "Peter Foytik", "Abdul Rahman", "Sachini Rajapakse", "Isurunima Kularathna", "Pramoda Karunarathna", "Chalani Rajapakse", "Ng Wee Keong", "Kasun De Zoysa", "Tharaka Hewa", "Amin Hass", "Wathsala Herath", "Aruna Withanage", "Nilaan Loganathan", "Atmaram Yarlagadda", "Sachin Shetty" ], "categories": [ "cs.AI", "cs.CY" ], "topics": [ "computer-use", "tool-use" ], "score": 10, "relevance": "medium", "matched_queries": [ "ai-agent" ], "arxiv_id": "2606.19116", "source": "arxiv", "source_id": "arxiv:2606.19116", "pdf_url": "https://arxiv.org/pdf/2606.19116", "primary_query": "ai-agent" }, { "id": "2606.18716", "title": "Human-AI Agent Interaction in a Business Context", "url": "https://arxiv.org/abs/2606.18716", "published": "2026-06-17", "updated": "2026-06-17", "authors": [ "Kathrin Paimann", "Elizangela Valarini", "Sebastian Juhl" ], "categories": [ "cs.HC", "cs.AI" ], "topics": [ "agent-evaluation", "tool-use" ], "score": 10, "relevance": "medium", "matched_queries": [ "ai-agent" ], "arxiv_id": "2606.18716", "source": "arxiv", "source_id": "arxiv:2606.18716", "pdf_url": "https://arxiv.org/pdf/2606.18716", "primary_query": "ai-agent" }, { "id": "2606.18831", "title": "Beyond Reward Engineering: A Data Recipe for Long-Context Reinforcement Learning", "url": "https://arxiv.org/abs/2606.18831", "published": "2026-06-17", "updated": "2026-06-17", "authors": [ "Xiaoyue Xu", "Sikui Zhang", "Xiaorong Wang", "Xu Han", "Chaojun Xiao" ], "categories": [ "cs.CL", "cs.AI" ], "topics": [ "agent-evaluation", "rag", "reasoning" ], "score": 10, "relevance": "medium", "matched_queries": [ "autonomous-agent-llm" ], "arxiv_id": "2606.18831", "source": "arxiv", "source_id": "arxiv:2606.18831", "pdf_url": "https://arxiv.org/pdf/2606.18831", "primary_query": "autonomous-agent-llm" }, { "id": "2606.17645", "title": "Beyond Domains: Reusing Web Skills via Transferable Interaction Patterns", "url": "https://arxiv.org/abs/2606.17645", "published": "2026-06-16", "updated": "2026-06-16", "authors": [ "Shiqi He", "Yue Cui", "Feijie Wu", "Xinyu Ma", "Jiaheng Lu", "Yaliang Li", "Bolin Ding", "Mosharaf Chowdhury" ], "categories": [ "cs.AI", "cs.CL", "cs.LG" ], "topics": [ "agent-evaluation", "computer-use", "rag", "tool-use" ], "score": 10, "relevance": "medium", "matched_queries": [ "web-gui-agent" ], "arxiv_id": "2606.17645", "source": "arxiv", "source_id": "arxiv:2606.17645", "pdf_url": "https://arxiv.org/pdf/2606.17645", "primary_query": "web-gui-agent" }, { "id": "2606.28370", "title": "Conversational Query Engine for Mixed-Modality Heterogeneous Enterprise Data Sources", "url": "https://arxiv.org/abs/2606.28370", "published": "2026-06-15", "updated": "2026-06-15", "authors": [ "Darshita Rathore", "Vineet Kumar", "Vaibhav Singal", "Ankur Vivek Singh", "Anindya Moitra" ], "categories": [ "cs.IR", "cs.AI" ], "topics": [ "agent-evaluation", "rag", "tool-use", "workflow-agent" ], "score": 10, "relevance": "medium", "matched_queries": [ "rag-agent" ], "arxiv_id": "2606.28370", "source": "arxiv", "source_id": "arxiv:2606.28370", "pdf_url": "https://arxiv.org/pdf/2606.28370", "primary_query": "rag-agent" }, { "id": "2606.12908", "title": "SENTINEL: Failure-Driven Reinforcement Learning for Training Tool-Using Language Model Agents", "url": "https://arxiv.org/abs/2606.12908", "published": "2026-06-11", "updated": "2026-06-11", "authors": [ "Ziyi Wang", "Yuxuan Lu", "Yimeng Zhang", "Qun Liu", "Chen Luo", "Jiri Gesi", "Hanqing Lu", "Yisi Sang", "Manling Li", "Jing Huang", "Dakuo Wang" ], "categories": [ "cs.CL" ], "topics": [ "agent-evaluation", "tool-use" ], "score": 10, "relevance": "medium", "matched_queries": [ "tool-use" ], "arxiv_id": "2606.12908", "source": "arxiv", "source_id": "arxiv:2606.12908", "pdf_url": "https://arxiv.org/pdf/2606.12908", "primary_query": "tool-use" }, { "id": "2606.13949", "title": "Minim: Privacy-Aware Minimal View for Agents via Trusted Local Sanitization", "url": "https://arxiv.org/abs/2606.13949", "published": "2026-06-11", "updated": "2026-06-11", "authors": [ "Hexuan Yu", "Chaoyu Zhang", "Heng Jin", "Shanghao Shi", "Ning Zhang", "Y. Thomas Hou", "Wenjing Lou" ], "categories": [ "cs.AI" ], "topics": [ "agent-safety", "rag", "tool-use" ], "score": 10, "relevance": "medium", "matched_queries": [ "autonomous-agent-llm" ], "arxiv_id": "2606.13949", "source": "arxiv", "source_id": "arxiv:2606.13949", "pdf_url": "https://arxiv.org/pdf/2606.13949", "primary_query": "autonomous-agent-llm" }, { "id": "2606.10299", "title": "What Spatial Memory Must Store: Occlusion as the Test for Language-Agent Memory", "url": "https://arxiv.org/abs/2606.10299", "published": "2026-06-09", "updated": "2026-06-09", "authors": [ "Doeon Kwon", "Junho Bang" ], "categories": [ "cs.AI", "cs.CV", "cs.MA" ], "topics": [ "memory" ], "score": 10, "relevance": "medium", "matched_queries": [ "agent-memory", "language-agent" ], "arxiv_id": "2606.10299", "source": "arxiv", "source_id": "arxiv:2606.10299", "pdf_url": "https://arxiv.org/pdf/2606.10299", "primary_query": "agent-memory" }, { "id": "2606.11350", "title": "When More Documents Hurt RAG: Mitigating Vector Search Dilution with Domain-Scoped, Model-Agnostic Retrieval", "url": "https://arxiv.org/abs/2606.11350", "published": "2026-06-09", "updated": "2026-06-09", "authors": [ "Nabaraj Subedi", "Ahmed Abdelaty", "Shivanand Venkanna Sheshappanavar" ], "categories": [ "cs.CL", "cs.IR" ], "topics": [ "agent-evaluation", "multi-agent", "rag", "tool-use" ], "score": 10, "relevance": "medium", "matched_queries": [ "rag-agent" ], "arxiv_id": "2606.11350", "source": "arxiv", "source_id": "arxiv:2606.11350", "pdf_url": "https://arxiv.org/pdf/2606.11350", "primary_query": "rag-agent" }, { "id": "2606.09935", "title": "GitInject: Real-World Prompt Injection Attacks in AI-Powered CI/CD Pipelines", "url": "https://arxiv.org/abs/2606.09935", "published": "2026-06-07", "updated": "2026-06-07", "authors": [ "Jafar Isbarov", "Umid Suleymanov", "Ilia Shumailov", "Murat Kantarcioglu" ], "categories": [ "cs.CR", "cs.AI" ], "topics": [ "agent-evaluation", "agent-safety", "coding-agent", "rag", "tool-use", "workflow-agent" ], "score": 10, "relevance": "medium", "matched_queries": [ "agent-safety" ], "arxiv_id": "2606.09935", "source": "arxiv", "source_id": "arxiv:2606.09935", "pdf_url": "https://arxiv.org/pdf/2606.09935", "primary_query": "agent-safety" }, { "id": "2606.07017", "title": "The Sim-to-Real Gap of Foundation Model Agents: A Unified MDP Perspective", "url": "https://arxiv.org/abs/2606.07017", "published": "2026-06-05", "updated": "2026-06-05", "authors": [ "Xiaoou Liu", "Tiejin Chen", "Weibo Li", "Xiyang Hu", "Hua Wei" ], "categories": [ "cs.AI", "cs.CL", "cs.ET" ], "topics": [ "agent-evaluation", "embodied-agent", "tool-use" ], "score": 10, "relevance": "medium", "matched_queries": [ "agent-evaluation" ], "arxiv_id": "2606.07017", "source": "arxiv", "source_id": "arxiv:2606.07017", "pdf_url": "https://arxiv.org/pdf/2606.07017", "primary_query": "agent-evaluation" }, { "id": "2606.03889", "title": "RealClawBench: Live OpenClaw Benchmarks from Real Developer-Agent Sessions", "url": "https://arxiv.org/abs/2606.03889", "published": "2026-06-02", "updated": "2026-06-05", "authors": [ "Zongwei Lv", "Zhewen Tan", "Yaoming Li", "Yilun Yao", "Yuxuan Tian", "Lin Sun", "Xiangzheng Zhang", "Weihong Lin", "Tong Yang", "Guangxiang Zhao" ], "categories": [ "cs.CL" ], "topics": [ "agent-evaluation" ], "score": 10, "relevance": "medium", "matched_queries": [ "agent-evaluation" ], "arxiv_id": "2606.03889", "source": "arxiv", "source_id": "arxiv:2606.03889", "pdf_url": "https://arxiv.org/pdf/2606.03889", "primary_query": "agent-evaluation" }, { "id": "2606.01751", "title": "SparseX: Efficient Segment-Level KV Cache Sharing for Interleaved LLM Serving", "url": "https://arxiv.org/abs/2606.01751", "published": "2026-06-01", "updated": "2026-06-07", "authors": [ "Quqing Zhang", "Kai Chen", "Ning Liao", "Zehao Lin", "Bo Tang", "Feiyu Xiong", "Zhiyu Li", "Xiaoxing Wang" ], "categories": [ "cs.PF" ], "topics": [ "agent-evaluation", "agent-safety", "rag", "tool-use", "workflow-agent" ], "score": 10, "relevance": "medium", "matched_queries": [ "rag-agent" ], "arxiv_id": "2606.01751", "source": "arxiv", "source_id": "arxiv:2606.01751", "pdf_url": "https://arxiv.org/pdf/2606.01751", "primary_query": "rag-agent" }, { "id": "2606.02643", "title": "Inference Cost Attacks for Retrieval-Augmented Large Language Models", "url": "https://arxiv.org/abs/2606.02643", "published": "2026-05-31", "updated": "2026-05-31", "authors": [ "Chengliang Liu", "Liangbo Ning", "Yujuan Ding", "Wenqi Fan" ], "categories": [ "cs.CR", "cs.AI", "cs.DB" ], "topics": [ "agent-evaluation", "memory", "rag" ], "score": 10, "relevance": "medium", "matched_queries": [ "rag-agent" ], "arxiv_id": "2606.02643", "source": "arxiv", "source_id": "arxiv:2606.02643", "pdf_url": "https://arxiv.org/pdf/2606.02643", "primary_query": "rag-agent" }, { "id": "2606.00750", "title": "I-WebGenBench : Evaluating Interactivity in LLM-Generated Scientific Web Applications", "url": "https://arxiv.org/abs/2606.00750", "published": "2026-05-30", "updated": "2026-05-30", "authors": [ "Dasen Dai", "Biao Wu", "Meng Fang", "Shuoqi Li", "Wenhao Wang" ], "categories": [ "cs.CL" ], "topics": [ "agent-evaluation", "reasoning", "tool-use" ], "score": 10, "relevance": "medium", "matched_queries": [ "autonomous-agent-llm" ], "arxiv_id": "2606.00750", "source": "arxiv", "source_id": "arxiv:2606.00750", "pdf_url": "https://arxiv.org/pdf/2606.00750", "primary_query": "autonomous-agent-llm" }, { "id": "2605.29559", "title": "LiteCoder-Terminal: Scaling Long-Horizon Terminal Environments for Learning Language Agents", "url": "https://arxiv.org/abs/2605.29559", "published": "2026-05-28", "updated": "2026-05-28", "authors": [ "Xiaoxuan Peng", "Kaiqi Zhang", "Xinyu Lu", "Boxi Cao", "Yaojie Lu", "Hongyu Lin", "Xianpei Han", "Le Sun" ], "categories": [ "cs.CL" ], "topics": [ "planning", "workflow-agent" ], "score": 10, "relevance": "medium", "matched_queries": [ "language-agent" ], "arxiv_id": "2605.29559", "source": "arxiv", "source_id": "arxiv:2605.29559", "pdf_url": "https://arxiv.org/pdf/2605.29559", "primary_query": "language-agent" }, { "id": "2605.29491", "title": "The Curse of Helpfulness: Inverse Scaling Law in Robustness to Distractor Instructions via DistractionIF", "url": "https://arxiv.org/abs/2605.29491", "published": "2026-05-28", "updated": "2026-05-28", "authors": [ "Zeli Su", "Zhankai Xu", "Tianlei Chen", "Longfei Zheng", "Xiaolu Zhang", "Jun Zhou", "Wentao Zhang" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "rag", "tool-use" ], "score": 10, "relevance": "medium", "matched_queries": [ "rag-agent" ], "arxiv_id": "2605.29491", "source": "arxiv", "source_id": "arxiv:2605.29491", "pdf_url": "https://arxiv.org/pdf/2605.29491", "primary_query": "rag-agent" }, { "id": "2605.26352", "title": "RICE-PO: Turning Retrieval Interactions into Credit Signals for Reasoning Agents", "url": "https://arxiv.org/abs/2605.26352", "published": "2026-05-25", "updated": "2026-05-25", "authors": [ "Mingchen Li", "Hansi Zeng", "Zhuo Qian", "Jiatan Huang", "Hamed Zamani", "Hong Yu" ], "categories": [ "cs.CL" ], "topics": [ "agent-evaluation", "rag", "reasoning", "tool-use" ], "score": 10, "relevance": "medium", "matched_queries": [ "language-agent" ], "arxiv_id": "2605.26352", "source": "arxiv", "source_id": "arxiv:2605.26352", "pdf_url": "https://arxiv.org/pdf/2605.26352", "primary_query": "language-agent" }, { "id": "2605.25988", "title": "What Makes a Medical Checker Trainable? Diagnosing Signal Collapse and Reward Hacking in Checker-Guided RAG for Biomedical QA", "url": "https://arxiv.org/abs/2605.25988", "published": "2026-05-25", "updated": "2026-05-25", "authors": [ "Yuelyu Ji", "Min Gu Kwak", "Hang Zhang", "Xizhi Wu", "Chenyu Li", "Yanshan Wan" ], "categories": [ "cs.CL" ], "topics": [ "agent-evaluation", "computer-use", "rag", "reasoning" ], "score": 10, "relevance": "medium", "matched_queries": [ "rag-agent" ], "arxiv_id": "2605.25988", "source": "arxiv", "source_id": "arxiv:2605.25988", "pdf_url": "https://arxiv.org/pdf/2605.25988", "primary_query": "rag-agent" }, { "id": "2605.23281", "title": "DepthAgent: Towards Better Universal Depth Estimation via Sample-wise Expert Selection", "url": "https://arxiv.org/abs/2605.23281", "published": "2026-05-22", "updated": "2026-05-22", "authors": [ "Jie Zhu", "Girish Chandar Ganesan", "Xiaoming Liu" ], "categories": [ "cs.CV" ], "topics": [ "agent-evaluation", "rag", "tool-use" ], "score": 10, "relevance": "medium", "matched_queries": [ "language-agent" ], "arxiv_id": "2605.23281", "source": "arxiv", "source_id": "arxiv:2605.23281", "pdf_url": "https://arxiv.org/pdf/2605.23281", "primary_query": "language-agent" }, { "id": "2605.20312", "title": "Pramana: A Protocol-Layer Treatment of Claim Verification in Autonomous Agent Networks", "url": "https://arxiv.org/abs/2605.20312", "published": "2026-05-19", "updated": "2026-05-19", "authors": [ "Ravi Kiran Kadaboina" ], "categories": [ "cs.CR", "cs.LO", "cs.MA" ], "topics": [ "rag", "reasoning", "tool-use" ], "score": 10, "relevance": "medium", "matched_queries": [ "autonomous-agent-llm" ], "arxiv_id": "2605.20312", "source": "arxiv", "source_id": "arxiv:2605.20312", "pdf_url": "https://arxiv.org/pdf/2605.20312", "primary_query": "autonomous-agent-llm" }, { "id": "2605.13438", "title": "CogniFold: Always-On Proactive Memory via Cognitive Folding", "url": "https://arxiv.org/abs/2605.13438", "published": "2026-05-13", "updated": "2026-06-17", "authors": [ "Suli Wang", "Yiqun Duan", "Yu Deng", "Rundong Zhao", "Dai Shi", "Minghua Deng", "Chen Chen", "Xinliang Zhou" ], "categories": [ "cs.AI", "cs.CL" ], "topics": [ "agent-evaluation", "memory", "rag" ], "score": 10, "relevance": "medium", "matched_queries": [ "agent-memory" ], "arxiv_id": "2605.13438", "source": "arxiv", "source_id": "arxiv:2605.13438", "pdf_url": "https://arxiv.org/pdf/2605.13438", "primary_query": "agent-memory" }, { "id": "2605.11359", "title": "CVEvolve: Autonomous Algorithm Discovery for Unstructured Scientific Data Processing", "url": "https://arxiv.org/abs/2605.11359", "published": "2026-05-12", "updated": "2026-06-01", "authors": [ "Ming Du", "Xiangyu Yin", "Yanqi Luo", "Dishant Beniwal", "Songyuan Tang", "Hemant Sharma", "Mathew J. Cherukara" ], "categories": [ "cs.AI", "physics.data-an" ], "topics": [ "agent-evaluation", "rag", "tool-use" ], "score": 10, "relevance": "medium", "matched_queries": [ "autonomous-agent-llm" ], "arxiv_id": "2605.11359", "source": "arxiv", "source_id": "arxiv:2605.11359", "pdf_url": "https://arxiv.org/pdf/2605.11359", "primary_query": "autonomous-agent-llm" }, { "id": "2605.08721", "title": "Breaking the Impasse: Dual-Scale Evolutionary Policy Training for Social Language Agents", "url": "https://arxiv.org/abs/2605.08721", "published": "2026-05-09", "updated": "2026-05-09", "authors": [ "Minzheng Wang", "Run Luo", "Yanbo Wang", "Zichen Liu", "Yuqiao Tan", "Tao Tan", "Xu Nan", "Yinhe Zheng", "Wenji Mao" ], "categories": [ "cs.CL" ], "topics": [ "agent-evaluation", "tool-use" ], "score": 10, "relevance": "medium", "matched_queries": [ "language-agent" ], "arxiv_id": "2605.08721", "source": "arxiv", "source_id": "arxiv:2605.08721", "pdf_url": "https://arxiv.org/pdf/2605.08721", "primary_query": "language-agent" }, { "id": "2605.03159", "title": "Learning Correct Behavior from Examples: Validating Sequential Execution in Autonomous Agents", "url": "https://arxiv.org/abs/2605.03159", "published": "2026-05-04", "updated": "2026-05-04", "authors": [ "Reshabh K Sharma", "Gaurav Mittal", "Yu Hu" ], "categories": [ "cs.AI", "cs.SE" ], "topics": [ "agent-evaluation", "embodied-agent", "rag" ], "score": 10, "relevance": "medium", "matched_queries": [ "autonomous-agent-llm" ], "arxiv_id": "2605.03159", "source": "arxiv", "source_id": "arxiv:2605.03159", "pdf_url": "https://arxiv.org/pdf/2605.03159", "primary_query": "autonomous-agent-llm" }, { "id": "2604.11839", "title": "Beyond Static Sandboxing: Learned Capability Governance for Autonomous AI Agents", "url": "https://arxiv.org/abs/2604.11839", "published": "2026-04-12", "updated": "2026-05-03", "authors": [ "Bronislav Sidik", "Lior Rokach" ], "categories": [ "cs.CR", "cs.AI" ], "topics": [ "agent-safety", "tool-use" ], "score": 10, "relevance": "medium", "matched_queries": [ "agent-safety" ], "arxiv_id": "2604.11839", "source": "arxiv", "source_id": "arxiv:2604.11839", "pdf_url": "https://arxiv.org/pdf/2604.11839", "primary_query": "agent-safety" }, { "id": "2603.27742", "title": "TIR-Agent: Training an Explorative and Efficient Agent for Image Restoration", "url": "https://arxiv.org/abs/2603.27742", "published": "2026-03-29", "updated": "2026-03-29", "authors": [ "Yisheng Zhang", "Guoli Jia", "Haote Hu", "Shanxu Zhao", "Kaikai Zhao", "Long Sun", "Xinwei Long", "Kai Tian", "Che Jiang", "Zhaoxiang Liu", "Kai Wang", "Shiguo Lian", "Kaiyan Zhang", "Bowen Zhou" ], "categories": [ "cs.CV" ], "topics": [ "agent-evaluation", "tool-use" ], "score": 10, "relevance": "medium", "matched_queries": [ "language-agent" ], "arxiv_id": "2603.27742", "source": "arxiv", "source_id": "arxiv:2603.27742", "pdf_url": "https://arxiv.org/pdf/2603.27742", "primary_query": "language-agent" }, { "id": "2603.05413", "title": "Building Enterprise Realtime Voice Agents from Scratch: A Technical Tutorial", "url": "https://arxiv.org/abs/2603.05413", "published": "2026-03-05", "updated": "2026-03-17", "authors": [ "Jielin Qiu", "Zixiang Chen", "Liangwei Yang", "Ming Zhu", "Zhiwei Liu", "Juntao Tan", "Wenting Zhao", "Rithesh Murthy", "Roshan Ram", "Akshara Prabhakar", "Shelby Heinecke", "Caiming Xiong", "Silvio Savarese", "Huan Wang" ], "categories": [ "cs.SD" ], "topics": [ "agent-evaluation", "tool-use", "workflow-agent" ], "score": 10, "relevance": "medium", "matched_queries": [ "function-calling" ], "arxiv_id": "2603.05413", "source": "arxiv", "source_id": "arxiv:2603.05413", "pdf_url": "https://arxiv.org/pdf/2603.05413", "primary_query": "function-calling" }, { "id": "2603.00991", "title": "Tracking Capabilities for Safer Agents", "url": "https://arxiv.org/abs/2603.00991", "published": "2026-03-01", "updated": "2026-05-07", "authors": [ "Martin Odersky", "Yaoyu Zhao", "Yichen Xu", "Oliver Bračevac", "Cao Nguyen Pham" ], "categories": [ "cs.AI", "cs.PL" ], "topics": [ "agent-safety", "rag", "tool-use" ], "score": 10, "relevance": "medium", "matched_queries": [ "agent-safety" ], "arxiv_id": "2603.00991", "source": "arxiv", "source_id": "arxiv:2603.00991", "pdf_url": "https://arxiv.org/pdf/2603.00991", "primary_query": "agent-safety" }, { "id": "2602.15197", "title": "OpaqueToolsBench: Learning Nuances of Tool Behavior Through Interaction", "url": "https://arxiv.org/abs/2602.15197", "published": "2026-02-16", "updated": "2026-02-16", "authors": [ "Skyler Hallinan", "Thejas Venkatesh", "Xiang Ren", "Sai Praneeth Karimireddy", "Ashwin Paranjape", "Yuhao Zhang", "Jack Hessel" ], "categories": [ "cs.CL", "cs.AI" ], "topics": [ "agent-evaluation", "tool-use" ], "score": 10, "relevance": "medium", "matched_queries": [ "function-calling" ], "arxiv_id": "2602.15197", "source": "arxiv", "source_id": "arxiv:2602.15197", "pdf_url": "https://arxiv.org/pdf/2602.15197", "primary_query": "function-calling" }, { "id": "2602.03025", "title": "RC-GRPO: Reward-Conditioned Group Relative Policy Optimization for Multi-Turn Tool Calling Agents", "url": "https://arxiv.org/abs/2602.03025", "published": "2026-02-03", "updated": "2026-02-03", "authors": [ "Haitian Zhong", "Jixiu Zhai", "Lei Song", "Jiang Bian", "Qiang Liu", "Tieniu Tan" ], "categories": [ "cs.AI", "cs.CL" ], "topics": [ "agent-evaluation", "tool-use" ], "score": 10, "relevance": "medium", "matched_queries": [ "function-calling" ], "arxiv_id": "2602.03025", "source": "arxiv", "source_id": "arxiv:2602.03025", "pdf_url": "https://arxiv.org/pdf/2602.03025", "primary_query": "function-calling" }, { "id": "2602.03022", "title": "STAR: Similarity-guided Teacher-Assisted Refinement for Super-Tiny Function Calling Models", "url": "https://arxiv.org/abs/2602.03022", "published": "2026-02-03", "updated": "2026-02-24", "authors": [ "Jiliang Ni", "Jiachen Pu", "Zhongyi Yang", "Jingfeng Luo", "Conggang Hu" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "computer-use", "tool-use" ], "score": 10, "relevance": "medium", "matched_queries": [ "function-calling" ], "arxiv_id": "2602.03022", "source": "arxiv", "source_id": "arxiv:2602.03022", "pdf_url": "https://arxiv.org/pdf/2602.03022", "primary_query": "function-calling" }, { "id": "2601.17829", "title": "Linguistic and Argument Diversity in Synthetic Data for Function-Calling Agents", "url": "https://arxiv.org/abs/2601.17829", "published": "2026-01-25", "updated": "2026-01-25", "authors": [ "Dan Greenstein", "Zohar Karnin", "Chen Amiraz", "Oren Somekh" ], "categories": [ "cs.CL", "cs.AI" ], "topics": [ "agent-evaluation", "computer-use", "rag", "tool-use" ], "score": 10, "relevance": "medium", "matched_queries": [ "function-calling" ], "arxiv_id": "2601.17829", "source": "arxiv", "source_id": "arxiv:2601.17829", "pdf_url": "https://arxiv.org/pdf/2601.17829", "primary_query": "function-calling" }, { "id": "2512.17052", "title": "Dynamic Tool Dependency Retrieval for Lightweight Function Calling", "url": "https://arxiv.org/abs/2512.17052", "published": "2025-12-18", "updated": "2026-04-17", "authors": [ "Bhrij Patel", "Davide Belli", "Amir Jalalirad", "Maximilian Arnold", "Aleksandr Ermolov", "Bence Major" ], "categories": [ "cs.LG" ], "topics": [ "agent-evaluation", "planning", "rag", "tool-use" ], "score": 10, "relevance": "medium", "matched_queries": [ "function-calling" ], "arxiv_id": "2512.17052", "source": "arxiv", "source_id": "arxiv:2512.17052", "pdf_url": "https://arxiv.org/pdf/2512.17052", "primary_query": "function-calling" }, { "id": "2510.10197", "title": "Don't Just Fine-tune the Agent, Tune the Environment", "url": "https://arxiv.org/abs/2510.10197", "published": "2025-10-11", "updated": "2026-01-30", "authors": [ "Siyuan Lu", "Zechuan Wang", "Hongxuan Zhang", "Qintong Wu", "Leilei Gan", "Chenyi Zhuang", "Jinjie Gu", "Tao Lin" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "tool-use" ], "score": 10, "relevance": "medium", "matched_queries": [ "function-calling" ], "arxiv_id": "2510.10197", "source": "arxiv", "source_id": "arxiv:2510.10197", "pdf_url": "https://arxiv.org/pdf/2510.10197", "primary_query": "function-calling" }, { "id": "2510.06727", "title": "Scaling LLM Multi-turn RL with End-to-end Summarization-based Context Management", "url": "https://arxiv.org/abs/2510.06727", "published": "2025-10-08", "updated": "2025-10-08", "authors": [ "Miao Lu", "Weiwei Sun", "Weihua Du", "Zhan Ling", "Xuesong Yao", "Kang Liu", "Jiecao Chen" ], "categories": [ "cs.CL", "cs.AI", "cs.LG" ], "topics": [ "agent-evaluation", "planning", "tool-use" ], "score": 10, "relevance": "medium", "matched_queries": [ "function-calling" ], "arxiv_id": "2510.06727", "source": "arxiv", "source_id": "arxiv:2510.06727", "pdf_url": "https://arxiv.org/pdf/2510.06727", "primary_query": "function-calling" }, { "id": "2607.06126", "title": "Causal Inference with Video Features as Treatments", "url": "https://arxiv.org/abs/2607.06126", "published": "2026-07-07", "updated": "2026-07-07", "authors": [ "Kentaro Nakamura", "Adam Breuer", "Michael H. Crespin", "Bryce J. Dietrich", "Kosuke Imai" ], "categories": [ "stat.AP" ], "topics": [ "agent-evaluation", "rag", "tool-use" ], "score": 9, "relevance": "medium", "matched_queries": [ "ai-agent" ], "arxiv_id": "2607.06126", "source": "arxiv", "source_id": "arxiv:2607.06126", "pdf_url": "https://arxiv.org/pdf/2607.06126", "primary_query": "ai-agent" }, { "id": "2607.06214", "title": "A toy framework for single and multi-agent human-AI curiosity ecosystems", "url": "https://arxiv.org/abs/2607.06214", "published": "2026-07-07", "updated": "2026-07-07", "authors": [ "Ilya E. Monosov" ], "categories": [ "cs.AI" ], "topics": [ "multi-agent" ], "score": 9, "relevance": "medium", "matched_queries": [ "agentic-ai" ], "arxiv_id": "2607.06214", "source": "arxiv", "source_id": "arxiv:2607.06214", "pdf_url": "https://arxiv.org/pdf/2607.06214", "primary_query": "agentic-ai" }, { "id": "2607.04763", "title": "Multi-Turn On-Policy Distillation with Prefix Replay", "url": "https://arxiv.org/abs/2607.04763", "published": "2026-07-06", "updated": "2026-07-06", "authors": [ "Baohao Liao", "Hanze Dong", "Christof Monz", "Xinxing Xu", "Li Dong", "Furu Wei" ], "categories": [ "cs.LG", "cs.AI", "cs.CL", "stat.ML" ], "topics": [ "reasoning", "tool-use" ], "score": 9, "relevance": "medium", "matched_queries": [ "llm-agent" ], "arxiv_id": "2607.04763", "source": "arxiv", "source_id": "arxiv:2607.04763", "pdf_url": "https://arxiv.org/pdf/2607.04763", "primary_query": "llm-agent" }, { "id": "2607.04631", "title": "Formal Disco: Scalable Open-Ended Generation of Formally Verified Programs", "url": "https://arxiv.org/abs/2607.04631", "published": "2026-07-06", "updated": "2026-07-06", "authors": [ "Gabriel Poesia", "Simon Henniger", "Tzu-Han Hsu", "Yilun Du", "Nada Amin" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "reasoning", "tool-use" ], "score": 9, "relevance": "medium", "matched_queries": [ "ai-agent" ], "arxiv_id": "2607.04631", "source": "arxiv", "source_id": "arxiv:2607.04631", "pdf_url": "https://arxiv.org/pdf/2607.04631", "primary_query": "ai-agent" }, { "id": "2607.04708", "title": "Strategic Buying Agents", "url": "https://arxiv.org/abs/2607.04708", "published": "2026-07-06", "updated": "2026-07-06", "authors": [ "Mingyang Fu", "Ming Hu" ], "categories": [ "econ.TH", "cs.AI", "cs.CY", "cs.GT", "cs.HC" ], "topics": [ "agent-evaluation" ], "score": 9, "relevance": "medium", "matched_queries": [ "agentic-ai" ], "arxiv_id": "2607.04708", "source": "arxiv", "source_id": "arxiv:2607.04708", "pdf_url": "https://arxiv.org/pdf/2607.04708", "primary_query": "agentic-ai" }, { "id": "2607.05471", "title": "KAT-Coder-V2.5 Technical Report", "url": "https://arxiv.org/abs/2607.05471", "published": "2026-07-06", "updated": "2026-07-06", "authors": [ "Bo Huang", "Fengxiang Li", "Hao Xu", "Haoyang Huang", "Hongyi Fu", "Jinhua Hao", "Kun Yuan", "Minglei Zhang", "Pengcheng Xu", "Shiyang Liu", "Wenhao Zhuang", "Yuze Shi", "Zongxian Feng", "Chao Wang", "Cheng He", "Chongling Rao", "Deyu Cao", "Fan Yang", "Gang Xiong", "Haochen Liu", "Jiabao Li", "Jian Liang", "Jinghui Jia", "Jingwen Chang", "Jun Du", "Junyu Shi", "Min Li", "Mingqi Wu", "Qiang Gao", "Shangpeng Yan", "Shaotong Qi", "Shu Xu", "Shuo Zhou", "Tiankuo Xu", "Tong Zheng", "Weilun Zhao", "Xiancheng Meng", "Xianda Sun", "Xiaoyu Jiang", "Xunhao Jia", "Yao Xia", "Yimeng Xu", "Yinghan Cui", "Yingpeng Chen", "Yiwen Ning", "Yong Wang", "Yuxuan Sun", "Zhongsheng Liu", "Ming Sun", "Cheng Luo", "Chen Yang", "Han Li", "Kun Gai" ], "categories": [ "cs.SE", "cs.AI" ], "topics": [ "agent-evaluation", "coding-agent", "tool-use" ], "score": 9, "relevance": "medium", "matched_queries": [ "agent-evaluation", "tool-use" ], "arxiv_id": "2607.05471", "source": "arxiv", "source_id": "arxiv:2607.05471", "pdf_url": "https://arxiv.org/pdf/2607.05471", "primary_query": "agent-evaluation" }, { "id": "2607.05465", "title": "CanvasAgent: Enabling Complex Image Creation and Editing via Visual Tool Orchestration", "url": "https://arxiv.org/abs/2607.05465", "published": "2026-07-06", "updated": "2026-07-06", "authors": [ "Hairui Zhu", "Yiying Yang", "Tengjin Weng", "Ziyu Lu", "Xiao Yao", "Xiaoyang Ye", "Lin Ma", "Wenhao Jiang" ], "categories": [ "cs.CV", "cs.AI" ], "topics": [ "agent-evaluation", "reasoning", "tool-use", "workflow-agent" ], "score": 9, "relevance": "medium", "matched_queries": [ "tool-use" ], "arxiv_id": "2607.05465", "source": "arxiv", "source_id": "arxiv:2607.05465", "pdf_url": "https://arxiv.org/pdf/2607.05465", "primary_query": "tool-use" }, { "id": "2607.04508", "title": "Compressing the Validation Bottleneck: An Agentic Self-Driving Lab for Scientific Discovery", "url": "https://arxiv.org/abs/2607.04508", "published": "2026-07-05", "updated": "2026-07-05", "authors": [ "Kyunghoon Hur", "Chihun Lee" ], "categories": [ "cs.AI", "cs.RO" ], "topics": [ "planning" ], "score": 9, "relevance": "medium", "matched_queries": [ "agentic-ai" ], "arxiv_id": "2607.04508", "source": "arxiv", "source_id": "arxiv:2607.04508", "pdf_url": "https://arxiv.org/pdf/2607.04508", "primary_query": "agentic-ai" }, { "id": "2607.02514", "title": "Distributed Attacks in Persistent-State AI Control", "url": "https://arxiv.org/abs/2607.02514", "published": "2026-07-02", "updated": "2026-07-02", "authors": [ "Josh Hills", "Ida Caspary", "Asa Cooper Stickland" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "coding-agent", "memory", "tool-use" ], "score": 9, "relevance": "medium", "matched_queries": [ "coding-agent" ], "arxiv_id": "2607.02514", "source": "arxiv", "source_id": "arxiv:2607.02514", "pdf_url": "https://arxiv.org/pdf/2607.02514", "primary_query": "coding-agent" }, { "id": "2607.02357", "title": "Cloak and Detonate: Scanner Evasion and Dynamic Detection of Agent Skill Malware", "url": "https://arxiv.org/abs/2607.02357", "published": "2026-07-02", "updated": "2026-07-03", "authors": [ "Zimo Ji", "Congying Xu", "Zongjie Li", "Yudong Gao", "Xin Wei", "Shuai Wang", "Shing-Chi Cheung" ], "categories": [ "cs.CR", "cs.SE" ], "topics": [ "coding-agent" ], "score": 9, "relevance": "medium", "matched_queries": [ "coding-agent" ], "arxiv_id": "2607.02357", "source": "arxiv", "source_id": "arxiv:2607.02357", "pdf_url": "https://arxiv.org/pdf/2607.02357", "primary_query": "coding-agent" }, { "id": "2607.02057", "title": "Prompt Coverage Adequacy", "url": "https://arxiv.org/abs/2607.02057", "published": "2026-07-02", "updated": "2026-07-02", "authors": [ "Florian Tambon", "Michael Konstantinou", "Cedric Richter", "Charles Chenouard", "Mark Harman", "Mike Papadakis" ], "categories": [ "cs.SE", "cs.AI" ], "topics": [ "agent-evaluation", "computer-use", "rag", "tool-use" ], "score": 9, "relevance": "medium", "matched_queries": [ "autonomous-agent-llm" ], "arxiv_id": "2607.02057", "source": "arxiv", "source_id": "arxiv:2607.02057", "pdf_url": "https://arxiv.org/pdf/2607.02057", "primary_query": "autonomous-agent-llm" }, { "id": "2607.01148", "title": "Emergence of Preferential Attachment and Glass-Ceiling Effects in Autonomous Networks of LLMs", "url": "https://arxiv.org/abs/2607.01148", "published": "2026-07-01", "updated": "2026-07-06", "authors": [ "Yiming Zhang", "Vikram Krishnamurthy" ], "categories": [ "cs.SI", "eess.SY" ], "topics": [ "agent-safety", "reasoning", "tool-use" ], "score": 9, "relevance": "medium", "matched_queries": [ "llm-agent" ], "arxiv_id": "2607.01148", "source": "arxiv", "source_id": "arxiv:2607.01148", "pdf_url": "https://arxiv.org/pdf/2607.01148", "primary_query": "llm-agent" }, { "id": "2607.01426", "title": "When Should Service Agents Reconsider? Difficulty-Routed Control in Customer-Service Operations", "url": "https://arxiv.org/abs/2607.01426", "published": "2026-07-01", "updated": "2026-07-01", "authors": [ "Qian Chen", "Chengyuan Liu", "Xin Yu" ], "categories": [ "cs.AI", "cs.CL" ], "topics": [ "agent-evaluation", "planning", "tool-use", "workflow-agent" ], "score": 9, "relevance": "medium", "matched_queries": [ "tool-use" ], "arxiv_id": "2607.01426", "source": "arxiv", "source_id": "arxiv:2607.01426", "pdf_url": "https://arxiv.org/pdf/2607.01426", "primary_query": "tool-use" }, { "id": "2606.31422", "title": "Ask the World Before Acting: Environment Probing for Calibrated Agent World Models", "url": "https://arxiv.org/abs/2606.31422", "published": "2026-06-30", "updated": "2026-07-05", "authors": [ "Xinyuan Song", "Zekun Cai" ], "categories": [ "cs.AI" ], "topics": [ "reasoning", "tool-use", "world-model" ], "score": 9, "relevance": "medium", "matched_queries": [ "language-agent" ], "arxiv_id": "2606.31422", "source": "arxiv", "source_id": "arxiv:2606.31422", "pdf_url": "https://arxiv.org/pdf/2606.31422", "primary_query": "language-agent" }, { "id": "2606.31055", "title": "Reference-Based Prosody and Rhythm Evaluation for Spoken Dialogue Systems", "url": "https://arxiv.org/abs/2606.31055", "published": "2026-06-30", "updated": "2026-06-30", "authors": [ "Ashish Hallur", "Thomas Thebaud", "Georgi Tinchev", "Venkatesh Ravichandran", "Laureano Moro-Velazquez" ], "categories": [ "cs.CL", "cs.SD", "eess.AS" ], "topics": [ "agent-evaluation", "tool-use" ], "score": 9, "relevance": "medium", "matched_queries": [ "ai-agent" ], "arxiv_id": "2606.31055", "source": "arxiv", "source_id": "arxiv:2606.31055", "pdf_url": "https://arxiv.org/pdf/2606.31055", "primary_query": "ai-agent" }, { "id": "2606.30775", "title": "A Single Rewrite Suffices: Empirical Lessons from Production Skill Description Optimization", "url": "https://arxiv.org/abs/2606.30775", "published": "2026-06-29", "updated": "2026-06-29", "authors": [ "Yangqiaoyu Zhou", "Mohammad Alqudah", "Kwei-Herng Lai", "Aaron Halfaker", "Yingqi Xiong", "Yaar Harari" ], "categories": [ "cs.CL", "cs.AI" ], "topics": [ "rag", "tool-use", "workflow-agent" ], "score": 9, "relevance": "medium", "matched_queries": [ "ai-agent" ], "arxiv_id": "2606.30775", "source": "arxiv", "source_id": "arxiv:2606.30775", "pdf_url": "https://arxiv.org/pdf/2606.30775", "primary_query": "ai-agent" }, { "id": "2606.30441", "title": "Translating Natural Language to Strategic Temporal Specifications via LLMs", "url": "https://arxiv.org/abs/2606.30441", "published": "2026-06-29", "updated": "2026-07-02", "authors": [ "Marco Aruta", "Francesco Improta", "Vadim Malvone", "Aniello Murano", "Vladana Perlić" ], "categories": [ "cs.MA", "cs.AI" ], "topics": [ "agent-evaluation", "multi-agent", "tool-use" ], "score": 9, "relevance": "medium", "matched_queries": [ "multi-agent-llm" ], "arxiv_id": "2606.30441", "source": "arxiv", "source_id": "arxiv:2606.30441", "pdf_url": "https://arxiv.org/pdf/2606.30441", "primary_query": "multi-agent-llm" }, { "id": "2606.28805", "title": "Physics Models for Sim-to-Real Transfer in Professional-Level Robot Table Tennis", "url": "https://arxiv.org/abs/2606.28805", "published": "2026-06-27", "updated": "2026-07-01", "authors": [ "Christian Conti", "Bilan Yang", "Alexander Sigrist", "Lorenzo Miele", "Yamen Saraiji", "Peter Dürr", "Naoya Takahashi" ], "categories": [ "cs.RO" ], "topics": [ "agent-evaluation", "embodied-agent", "rag", "tool-use", "world-model" ], "score": 9, "relevance": "medium", "matched_queries": [ "ai-agent" ], "arxiv_id": "2606.28805", "source": "arxiv", "source_id": "arxiv:2606.28805", "pdf_url": "https://arxiv.org/pdf/2606.28805", "primary_query": "ai-agent" }, { "id": "2606.27944", "title": "It Lied to a Doctor to Buy Poison Ingredients: Quantifying Real-World Misuse of Phone-use Agents", "url": "https://arxiv.org/abs/2606.27944", "published": "2026-06-26", "updated": "2026-06-26", "authors": [ "Yiming Sun", "Chen Chen", "Zifan Zhou", "Mi Zhang" ], "categories": [ "cs.MM", "cs.AI", "cs.CR" ], "topics": [ "agent-safety", "computer-use", "rag" ], "score": 9, "relevance": "medium", "matched_queries": [ "ai-agent" ], "arxiv_id": "2606.27944", "source": "arxiv", "source_id": "arxiv:2606.27944", "pdf_url": "https://arxiv.org/pdf/2606.27944", "primary_query": "ai-agent" }, { "id": "2606.28277", "title": "Towards Automating Scientific Review with Google's Paper Assistant Tool", "url": "https://arxiv.org/abs/2606.28277", "published": "2026-06-26", "updated": "2026-06-26", "authors": [ "Rajesh Jayaram", "Drew Tyler", "David Woodruff", "Corinna Cortes", "Yossi Matias", "Vahab Mirrokni", "Vincent Cohen-Addad" ], "categories": [ "cs.LG", "cs.AI", "cs.CL", "cs.CY" ], "topics": [ "agent-evaluation", "multi-agent", "tool-use" ], "score": 9, "relevance": "medium", "matched_queries": [ "agentic-ai" ], "arxiv_id": "2606.28277", "source": "arxiv", "source_id": "arxiv:2606.28277", "pdf_url": "https://arxiv.org/pdf/2606.28277", "primary_query": "agentic-ai" }, { "id": "2606.28125", "title": "How Humans, Bots, and Agents Communicate About Vulnerabilities in Pull Requests", "url": "https://arxiv.org/abs/2606.28125", "published": "2026-06-26", "updated": "2026-06-26", "authors": [ "Pien Rooijendijk", "Christoph Treude", "Mairieli Wessel" ], "categories": [ "cs.SE", "cs.CR" ], "topics": [ "agent-safety", "coding-agent", "planning" ], "score": 9, "relevance": "medium", "matched_queries": [ "coding-agent" ], "arxiv_id": "2606.28125", "source": "arxiv", "source_id": "arxiv:2606.28125", "pdf_url": "https://arxiv.org/pdf/2606.28125", "primary_query": "coding-agent" }, { "id": "2606.26978", "title": "To Run or Not to Run: Analyzing the Cost-Effectiveness of Code Execution in LLM-Based Program Repair", "url": "https://arxiv.org/abs/2606.26978", "published": "2026-06-25", "updated": "2026-06-25", "authors": [ "Zhihao Lin", "Junhua Zhu", "Mingyi Zhou", "Xin Wang", "Zhensu Sun", "Renyu Yang", "David Lo", "Li Li" ], "categories": [ "cs.SE" ], "topics": [ "agent-evaluation", "coding-agent", "rag" ], "score": 9, "relevance": "medium", "matched_queries": [ "coding-agent" ], "arxiv_id": "2606.26978", "source": "arxiv", "source_id": "arxiv:2606.26978", "pdf_url": "https://arxiv.org/pdf/2606.26978", "primary_query": "coding-agent" }, { "id": "2606.26028", "title": "Can Trustless Agents Be Trusted? An Empirical Study of the ERC-8004 Decentralized AI Agent Ecosystem", "url": "https://arxiv.org/abs/2606.26028", "published": "2026-06-24", "updated": "2026-06-24", "authors": [ "Xihan Xiong", "Zelin Li", "Wei Wei", "Qin Wang", "William Knottenbelt", "Zhipeng Wang" ], "categories": [ "cs.CR", "cs.AI", "cs.MA" ], "topics": [ "tool-use" ], "score": 9, "relevance": "medium", "matched_queries": [ "ai-agent" ], "arxiv_id": "2606.26028", "source": "arxiv", "source_id": "arxiv:2606.26028", "pdf_url": "https://arxiv.org/pdf/2606.26028", "primary_query": "ai-agent" }, { "id": "2606.25342", "title": "Lifelong In-Context Learning with Transformers Requires Parametric Forms of Attention", "url": "https://arxiv.org/abs/2606.25342", "published": "2026-06-24", "updated": "2026-06-24", "authors": [ "Luke McDermott", "Robert W. Heath", "Rahul Parhi" ], "categories": [ "cs.LG" ], "topics": [ "agent-evaluation", "computer-use", "memory", "planning" ], "score": 9, "relevance": "medium", "matched_queries": [ "ai-agent" ], "arxiv_id": "2606.25342", "source": "arxiv", "source_id": "arxiv:2606.25342", "pdf_url": "https://arxiv.org/pdf/2606.25342", "primary_query": "ai-agent" }, { "id": "2606.26175", "title": "RMTL: Reinforced Micro-task Learning for Long-Horizon Manipulation with VLM Rewards", "url": "https://arxiv.org/abs/2606.26175", "published": "2026-06-24", "updated": "2026-06-24", "authors": [ "Anıl Can Ateş", "Orhan Kahraman", "Cihan Topal" ], "categories": [ "cs.RO" ], "topics": [ "computer-use", "embodied-agent", "planning", "rag" ], "score": 9, "relevance": "medium", "matched_queries": [ "web-gui-agent" ], "arxiv_id": "2606.26175", "source": "arxiv", "source_id": "arxiv:2606.26175", "pdf_url": "https://arxiv.org/pdf/2606.26175", "primary_query": "web-gui-agent" }, { "id": "2606.22798", "title": "Does the Same Token Mean the Same State? MoE Routing as Signal for Reasoning Control", "url": "https://arxiv.org/abs/2606.22798", "published": "2026-06-22", "updated": "2026-06-22", "authors": [ "Kang Chen", "Minshen Yu", "Junjie Nian", "Yaoning Wang", "Yixin Cao", "Yugang Jiang" ], "categories": [ "cs.CL" ], "topics": [ "coding-agent", "reasoning" ], "score": 9, "relevance": "medium", "matched_queries": [ "coding-agent" ], "arxiv_id": "2606.22798", "source": "arxiv", "source_id": "arxiv:2606.22798", "pdf_url": "https://arxiv.org/pdf/2606.22798", "primary_query": "coding-agent" }, { "id": "2606.22425", "title": "SVGym (SciVerseGym): An Environment for Reinforcement Learning and Bayesian Optimization in Crystal Discovery", "url": "https://arxiv.org/abs/2606.22425", "published": "2026-06-21", "updated": "2026-06-21", "authors": [ "Bin Cao" ], "categories": [ "cs.AI", "cond-mat.mtrl-sci" ], "topics": [ "agent-evaluation", "rag", "tool-use", "workflow-agent" ], "score": 9, "relevance": "medium", "matched_queries": [ "agentic-ai", "language-agent" ], "arxiv_id": "2606.22425", "source": "arxiv", "source_id": "arxiv:2606.22425", "pdf_url": "https://arxiv.org/pdf/2606.22425", "primary_query": "agentic-ai" }, { "id": "2606.21811", "title": "Steer, Don't Solve: Training Small Critic Models for Large Code Agents", "url": "https://arxiv.org/abs/2606.21811", "published": "2026-06-20", "updated": "2026-06-20", "authors": [ "Shubham Gandhi", "Yiqing Xie", "Atharva Naik", "Ruichen Zhu", "Carolyn Rose" ], "categories": [ "cs.SE", "cs.AI", "cs.LG" ], "topics": [ "coding-agent", "computer-use", "reasoning" ], "score": 9, "relevance": "medium", "matched_queries": [ "coding-agent" ], "arxiv_id": "2606.21811", "source": "arxiv", "source_id": "arxiv:2606.21811", "pdf_url": "https://arxiv.org/pdf/2606.21811", "primary_query": "coding-agent" }, { "id": "2606.21315", "title": "Social World Model for Lifelong Social Intelligence", "url": "https://arxiv.org/abs/2606.21315", "published": "2026-06-19", "updated": "2026-06-19", "authors": [ "Yu Luo" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "tool-use", "world-model" ], "score": 9, "relevance": "medium", "matched_queries": [ "language-agent" ], "arxiv_id": "2606.21315", "source": "arxiv", "source_id": "arxiv:2606.21315", "pdf_url": "https://arxiv.org/pdf/2606.21315", "primary_query": "language-agent" }, { "id": "2606.21453", "title": "CORTIS: Text-Only Adaptation of Spoken Language Models for Task-Oriented Voice Agents", "url": "https://arxiv.org/abs/2606.21453", "published": "2026-06-19", "updated": "2026-06-19", "authors": [ "Youngwon Choi", "Hyeonyu Kim", "Taeyoun Kwon", "Donghyuk Jung", "Myeongkyun Cho" ], "categories": [ "cs.HC", "cs.AI", "cs.SD", "eess.AS" ], "topics": [ "agent-evaluation", "tool-use" ], "score": 9, "relevance": "medium", "matched_queries": [ "function-calling" ], "arxiv_id": "2606.21453", "source": "arxiv", "source_id": "arxiv:2606.21453", "pdf_url": "https://arxiv.org/pdf/2606.21453", "primary_query": "function-calling" }, { "id": "2606.21654", "title": "ChainWorld: Composing Long-Horizon Desktop Workloads from Atomic OSWorld Tasks", "url": "https://arxiv.org/abs/2606.21654", "published": "2026-06-19", "updated": "2026-06-19", "authors": [ "Vincent Siu", "Manasi Sharma", "Dawn Song", "Daniel Yue Zhang", "Chenguang Wang" ], "categories": [ "cs.AI", "cs.CL" ], "topics": [ "agent-evaluation", "computer-use", "planning", "rag" ], "score": 9, "relevance": "medium", "matched_queries": [ "web-gui-agent" ], "arxiv_id": "2606.21654", "source": "arxiv", "source_id": "arxiv:2606.21654", "pdf_url": "https://arxiv.org/pdf/2606.21654", "primary_query": "web-gui-agent" }, { "id": "2606.20753", "title": "Empowering Polymeric Materials Discovery by Artificial Intelligence", "url": "https://arxiv.org/abs/2606.20753", "published": "2026-06-18", "updated": "2026-06-18", "authors": [ "Chenyao Ma", "Linda Zhang", "Yuheng Chen", "Wei Du", "Shangwen Fang", "Zihao Jiang", "Chuanyu Liu", "Xinyu Ma", "Rui Su", "Gang Wang", "Muyao Yu", "Dong Zhong", "Jie Zhu", "Weibo Gong", "Huan Gu", "Limin Li", "Chen Shen", "Rui Wu", "Zhenghao Wu", "Kan Xu", "Min Zhou", "Donglin He", "Xiayun Huang", "Shan Jiang", "Pengfei Ou", "Jiayu Peng", "Yuwei Zhang", "Jie Zhao", "Di Zhang", "Piao Ma", "Zhenghao Li", "Hao Li" ], "categories": [ "physics.chem-ph", "cs.AI" ], "topics": [ "rag", "reasoning", "tool-use", "workflow-agent", "world-model" ], "score": 9, "relevance": "medium", "matched_queries": [ "ai-agent" ], "arxiv_id": "2606.20753", "source": "arxiv", "source_id": "arxiv:2606.20753", "pdf_url": "https://arxiv.org/pdf/2606.20753", "primary_query": "ai-agent" }, { "id": "2606.18733", "title": "SWE-Future: Forecast-Conditioned Data Synthesis for Future-Oriented Software Engineering Agents", "url": "https://arxiv.org/abs/2606.18733", "published": "2026-06-17", "updated": "2026-06-17", "authors": [ "Qiao Zhao", "JianYing Qu", "Jun Zhang", "Yehua Yang", "Hanwen Du", "Zhongkai Sun" ], "categories": [ "cs.SE", "cs.AI" ], "topics": [ "agent-evaluation", "coding-agent", "computer-use" ], "score": 9, "relevance": "medium", "matched_queries": [ "agent-evaluation" ], "arxiv_id": "2606.18733", "source": "arxiv", "source_id": "arxiv:2606.18733", "pdf_url": "https://arxiv.org/pdf/2606.18733", "primary_query": "agent-evaluation" }, { "id": "2606.18746", "title": "What Must Generalist Agents Remember?", "url": "https://arxiv.org/abs/2606.18746", "published": "2026-06-17", "updated": "2026-06-17", "authors": [ "Khurram Yamin", "Namrata Deka", "Maitreyi Swaroop", "Albert Ting", "Jeff Schneider", "Bryan Wilder" ], "categories": [ "cs.AI" ], "topics": [ "memory", "planning", "tool-use" ], "score": 9, "relevance": "medium", "matched_queries": [ "agent-memory" ], "arxiv_id": "2606.18746", "source": "arxiv", "source_id": "arxiv:2606.18746", "pdf_url": "https://arxiv.org/pdf/2606.18746", "primary_query": "agent-memory" }, { "id": "2606.18996", "title": "TRAP: Benchmark for Task-completion and Resistance to Active Privacy-extraction", "url": "https://arxiv.org/abs/2606.18996", "published": "2026-06-17", "updated": "2026-06-18", "authors": [ "Moon Ye-Bin", "Nam Hyeon-Woo", "Baek Seong-Eun", "Yejin Yeo", "Tae-Hyun Oh" ], "categories": [ "cs.CR", "cs.AI" ], "topics": [ "agent-evaluation", "tool-use", "workflow-agent" ], "score": 9, "relevance": "medium", "matched_queries": [ "tool-use" ], "arxiv_id": "2606.18996", "source": "arxiv", "source_id": "arxiv:2606.18996", "pdf_url": "https://arxiv.org/pdf/2606.18996", "primary_query": "tool-use" }, { "id": "2606.18385", "title": "CaVe-VLM-CoT: An Interpretable Vision-Language Model Framework", "url": "https://arxiv.org/abs/2606.18385", "published": "2026-06-16", "updated": "2026-06-16", "authors": [ "Sneha Rao", "Shaina Raza", "Dhanesh Ramachandram" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "rag", "reasoning" ], "score": 9, "relevance": "medium", "matched_queries": [ "rag-agent" ], "arxiv_id": "2606.18385", "source": "arxiv", "source_id": "arxiv:2606.18385", "pdf_url": "https://arxiv.org/pdf/2606.18385", "primary_query": "rag-agent" }, { "id": "2606.18005", "title": "LLM Consumer Behavior Theory: Foundations of a Novel Research Field", "url": "https://arxiv.org/abs/2606.18005", "published": "2026-06-16", "updated": "2026-06-16", "authors": [ "Manon Reusens", "Sofie Goethals", "David Martens" ], "categories": [ "cs.AI", "econ.GN" ], "topics": [ "agent-safety", "rag", "world-model" ], "score": 9, "relevance": "medium", "matched_queries": [ "autonomous-agent-llm" ], "arxiv_id": "2606.18005", "source": "arxiv", "source_id": "arxiv:2606.18005", "pdf_url": "https://arxiv.org/pdf/2606.18005", "primary_query": "autonomous-agent-llm" }, { "id": "2606.16215", "title": "PACT: Privileged Trace Co-Training for Multi-Turn Tool-Use Agents", "url": "https://arxiv.org/abs/2606.16215", "published": "2026-06-15", "updated": "2026-06-15", "authors": [ "Zhenbang Du", "Jun Luo", "Zhiwei Zheng", "Xiangchi Yuan", "Kejing Xia", "Dachuan Shi", "Qirui Jin", "Qijia He", "Shaofeng Zou", "Yingbin Liang", "Wenke Lee" ], "categories": [ "cs.CL", "cs.AI", "cs.LG" ], "topics": [ "agent-evaluation", "computer-use", "reasoning", "tool-use" ], "score": 9, "relevance": "medium", "matched_queries": [ "tool-use" ], "arxiv_id": "2606.16215", "source": "arxiv", "source_id": "arxiv:2606.16215", "pdf_url": "https://arxiv.org/pdf/2606.16215", "primary_query": "tool-use" }, { "id": "2606.15971", "title": "SAG: SQL-Retrieval Augmented Generation with Query-Time Dynamic Hyperedges", "url": "https://arxiv.org/abs/2606.15971", "published": "2026-06-14", "updated": "2026-06-14", "authors": [ "Yuchao Wu", "Junqin Li", "XingCheng Liang", "Yongjie Chen", "Yinghao Liang", "Linyuan Mo", "Guanxian Li" ], "categories": [ "cs.CL" ], "topics": [ "agent-evaluation", "rag", "reasoning" ], "score": 9, "relevance": "medium", "matched_queries": [ "rag-agent" ], "arxiv_id": "2606.15971", "source": "arxiv", "source_id": "arxiv:2606.15971", "pdf_url": "https://arxiv.org/pdf/2606.15971", "primary_query": "rag-agent" }, { "id": "2606.13239", "title": "ComAct: Reframing Professional Software Manipulation via COM-as-Action Paradigm", "url": "https://arxiv.org/abs/2606.13239", "published": "2026-06-11", "updated": "2026-06-30", "authors": [ "Jiaxin Ai", "Tao Hu", "Xuemeng Yang", "Shu Zou", "Hairong Zhang", "Daocheng Fu", "Yu Yang", "Hongbin Zhou", "Nianchen Deng", "Pinlong Cai", "Zhongyuan Wang", "Botian Shi", "Kaipeng Zhang", "Licheng Wen" ], "categories": [ "cs.SE", "cs.AI", "cs.CL", "cs.CV" ], "topics": [ "agent-evaluation", "computer-use", "planning", "rag", "tool-use" ], "score": 9, "relevance": "medium", "matched_queries": [ "web-gui-agent" ], "arxiv_id": "2606.13239", "source": "arxiv", "source_id": "arxiv:2606.13239", "pdf_url": "https://arxiv.org/pdf/2606.13239", "primary_query": "web-gui-agent" }, { "id": "2606.10478", "title": "3D-CoS: A New 3D Reconstruction Paradigm Based on VLM Code Synthesis", "url": "https://arxiv.org/abs/2606.10478", "published": "2026-06-09", "updated": "2026-06-09", "authors": [ "Yuhao Wang", "Puyi Wang", "Linjie Li", "Zhengyuan Yang", "Kevin Qinghong Lin", "Yu Cheng" ], "categories": [ "cs.CV" ], "topics": [ "agent-evaluation", "planning", "rag", "tool-use", "workflow-agent" ], "score": 9, "relevance": "medium", "matched_queries": [ "rag-agent" ], "arxiv_id": "2606.10478", "source": "arxiv", "source_id": "arxiv:2606.10478", "pdf_url": "https://arxiv.org/pdf/2606.10478", "primary_query": "rag-agent" }, { "id": "2606.10875", "title": "Pushing the Limits of LLM Tool Calling via Experiential Knowledge Integration and Activation", "url": "https://arxiv.org/abs/2606.10875", "published": "2026-06-09", "updated": "2026-06-09", "authors": [ "Yupu Hao", "Zhuoran Jin", "Huanxuan Liao", "Kang Liu", "Jun Zhao" ], "categories": [ "cs.CL" ], "topics": [ "agent-evaluation", "reasoning", "tool-use" ], "score": 9, "relevance": "medium", "matched_queries": [ "autonomous-agent-llm" ], "arxiv_id": "2606.10875", "source": "arxiv", "source_id": "arxiv:2606.10875", "pdf_url": "https://arxiv.org/pdf/2606.10875", "primary_query": "autonomous-agent-llm" }, { "id": "2606.07924", "title": "Decoupling Semantics and Logic: A Training-Free Coarse-to-Fine Pipeline for Video Retrieval-Augmented Generation", "url": "https://arxiv.org/abs/2606.07924", "published": "2026-06-06", "updated": "2026-06-06", "authors": [ "Jiaxin Dai", "Zehang Wei", "Jiamin Yan", "Xiang Xiang" ], "categories": [ "cs.CV", "cs.AI", "cs.CL", "cs.LG", "cs.MM" ], "topics": [ "agent-evaluation", "agent-safety", "rag", "reasoning" ], "score": 9, "relevance": "medium", "matched_queries": [ "rag-agent" ], "arxiv_id": "2606.07924", "source": "arxiv", "source_id": "arxiv:2606.07924", "pdf_url": "https://arxiv.org/pdf/2606.07924", "primary_query": "rag-agent" }, { "id": "2606.06566", "title": "NTILC: Neural Tool Invocation via Learned Compression", "url": "https://arxiv.org/abs/2606.06566", "published": "2026-06-04", "updated": "2026-06-04", "authors": [ "Andrew Krikorian", "Yayuan Li", "Jason J. Corso" ], "categories": [ "cs.SE", "cs.AI" ], "topics": [ "agent-evaluation", "rag", "tool-use" ], "score": 9, "relevance": "medium", "matched_queries": [ "function-calling" ], "arxiv_id": "2606.06566", "source": "arxiv", "source_id": "arxiv:2606.06566", "pdf_url": "https://arxiv.org/pdf/2606.06566", "primary_query": "function-calling" }, { "id": "2606.03800", "title": "Trading Human Curation for Synthetic Augmentation in RLVR", "url": "https://arxiv.org/abs/2606.03800", "published": "2026-06-02", "updated": "2026-06-02", "authors": [ "Akshansh", "Leonardo Rosa Rodrigues", "Michael Korostelev", "Youssef Hassan", "Mark E. Whiting" ], "categories": [ "cs.LG", "cs.AI" ], "topics": [ "agent-evaluation", "reasoning" ], "score": 9, "relevance": "medium", "matched_queries": [ "function-calling" ], "arxiv_id": "2606.03800", "source": "arxiv", "source_id": "arxiv:2606.03800", "pdf_url": "https://arxiv.org/pdf/2606.03800", "primary_query": "function-calling" }, { "id": "2606.01722", "title": "Post-Deterministic Distributed Systems: A New Foundation for Trustworthy Autonomous Infrastructure", "url": "https://arxiv.org/abs/2606.01722", "published": "2026-06-01", "updated": "2026-06-01", "authors": [ "Jun He", "Deying Yu" ], "categories": [ "cs.LG", "cs.AI", "cs.DC" ], "topics": [ "computer-use", "memory", "planning", "reasoning" ], "score": 9, "relevance": "medium", "matched_queries": [ "agent-memory" ], "arxiv_id": "2606.01722", "source": "arxiv", "source_id": "arxiv:2606.01722", "pdf_url": "https://arxiv.org/pdf/2606.01722", "primary_query": "agent-memory" }, { "id": "2606.02245", "title": "When Knowledge Is Not Free: Cost-Aware Evidence Selection in Retrieval-Augmented Generation", "url": "https://arxiv.org/abs/2606.02245", "published": "2026-06-01", "updated": "2026-06-01", "authors": [ "Mingyan Wu", "Han Yang", "Omer Ben-Porat", "Yftah Ziser" ], "categories": [ "cs.CL" ], "topics": [ "agent-evaluation", "rag" ], "score": 9, "relevance": "medium", "matched_queries": [ "rag-agent" ], "arxiv_id": "2606.02245", "source": "arxiv", "source_id": "arxiv:2606.02245", "pdf_url": "https://arxiv.org/pdf/2606.02245", "primary_query": "rag-agent" }, { "id": "2606.00734", "title": "EMA: Approximate Nearest Neighbor Search with General Attribute Filtering and Dynamic Updates", "url": "https://arxiv.org/abs/2606.00734", "published": "2026-05-30", "updated": "2026-05-30", "authors": [ "Mocheng Li", "Baotong Lu", "James Cheng", "Chenhao Ma" ], "categories": [ "cs.DB" ], "topics": [ "agent-evaluation", "computer-use", "memory", "rag" ], "score": 9, "relevance": "medium", "matched_queries": [ "agent-memory", "rag-agent" ], "arxiv_id": "2606.00734", "source": "arxiv", "source_id": "arxiv:2606.00734", "pdf_url": "https://arxiv.org/pdf/2606.00734", "primary_query": "agent-memory" }, { "id": "2605.31064", "title": "Fighting Numerical Hallucinations via Data-centric Compilation for Online Financial QA", "url": "https://arxiv.org/abs/2605.31064", "published": "2026-05-29", "updated": "2026-05-29", "authors": [ "Hao Chen", "Xing Tang", "Qirui Liu", "Weijie Shi", "Shiwei Li", "Fuyuan Lyu", "Weihong Luo", "Xiku Du", "Xiuqiang He" ], "categories": [ "cs.IR", "cs.AI" ], "topics": [ "agent-evaluation", "rag", "reasoning" ], "score": 9, "relevance": "medium", "matched_queries": [ "rag-agent" ], "arxiv_id": "2605.31064", "source": "arxiv", "source_id": "arxiv:2605.31064", "pdf_url": "https://arxiv.org/pdf/2605.31064", "primary_query": "rag-agent" }, { "id": "2605.30628", "title": "The Architecture of Errors: From Universal Impossibility to Patch-Local LLM Reliability", "url": "https://arxiv.org/abs/2605.30628", "published": "2026-05-28", "updated": "2026-05-28", "authors": [ "Mikhail L. Arbuzov", "Lee Mosbacker", "Sisong Bei", "Ziwei Dong", "Dmitri Kalaev", "Alexey Shvets" ], "categories": [ "cs.CL", "cs.AI", "cs.LG" ], "topics": [ "agent-evaluation", "computer-use", "rag", "tool-use" ], "score": 9, "relevance": "medium", "matched_queries": [ "rag-agent" ], "arxiv_id": "2605.30628", "source": "arxiv", "source_id": "arxiv:2605.30628", "pdf_url": "https://arxiv.org/pdf/2605.30628", "primary_query": "rag-agent" }, { "id": "2605.28914", "title": "AIRGuard: Guarding Agent Actions with Runtime Authority Control", "url": "https://arxiv.org/abs/2605.28914", "published": "2026-05-27", "updated": "2026-05-27", "authors": [ "Suliu Qin", "Haomin Zhuang", "Yujun Zhou", "Yufei Han", "Xiangliang Zhang" ], "categories": [ "cs.CR", "cs.AI" ], "topics": [ "agent-safety", "reasoning", "tool-use" ], "score": 9, "relevance": "medium", "matched_queries": [ "language-agent" ], "arxiv_id": "2605.28914", "source": "arxiv", "source_id": "arxiv:2605.28914", "pdf_url": "https://arxiv.org/pdf/2605.28914", "primary_query": "language-agent" }, { "id": "2605.25162", "title": "STREAM: A Data-Centric Framework for Mining High-Value Task-Oriented Dialogues from Streaming Media", "url": "https://arxiv.org/abs/2605.25162", "published": "2026-05-24", "updated": "2026-05-24", "authors": [ "Liang Xue", "Haoyu Liu", "Cheng Wang", "Pengyu Chen", "Haozhuo Zheng", "Yang Liu" ], "categories": [ "cs.CL", "cs.AI" ], "topics": [ "agent-evaluation", "rag", "tool-use" ], "score": 9, "relevance": "medium", "matched_queries": [ "rag-agent" ], "arxiv_id": "2605.25162", "source": "arxiv", "source_id": "arxiv:2605.25162", "pdf_url": "https://arxiv.org/pdf/2605.25162", "primary_query": "rag-agent" }, { "id": "2605.21956", "title": "Detecting Offensive Cyber Agents: A Detection-in-Depth Approach", "url": "https://arxiv.org/abs/2605.21956", "published": "2026-05-21", "updated": "2026-05-21", "authors": [ "Matt Mittelsteadt", "Jam Kraprayoon", "Robin Staes-Polet", "Oskar Galeev", "Jan Wehner", "Christopher Covino", "Shaun Ee" ], "categories": [ "cs.CY" ], "topics": [ "agent-safety", "computer-use", "tool-use" ], "score": 9, "relevance": "medium", "matched_queries": [ "agent-safety" ], "arxiv_id": "2605.21956", "source": "arxiv", "source_id": "arxiv:2605.21956", "pdf_url": "https://arxiv.org/pdf/2605.21956", "primary_query": "agent-safety" }, { "id": "2605.22177", "title": "Maestro: Reinforcement Learning to Orchestrate Hierarchical Model-Skill Ensembles", "url": "https://arxiv.org/abs/2605.22177", "published": "2026-05-21", "updated": "2026-05-21", "authors": [ "Jinyang Wu", "Guocheng Zhai", "Ruihan Jin", "Yuhao Shen", "Zhengxi Lu", "Fan Zhang", "Haoran Luo", "Zheng Lian", "Zhengqi Wen", "Jianhua Tao" ], "categories": [ "cs.LG", "cs.CL" ], "topics": [ "agent-evaluation", "rag", "reasoning" ], "score": 9, "relevance": "medium", "matched_queries": [ "autonomous-agent-llm" ], "arxiv_id": "2605.22177", "source": "arxiv", "source_id": "arxiv:2605.22177", "pdf_url": "https://arxiv.org/pdf/2605.22177", "primary_query": "autonomous-agent-llm" }, { "id": "2605.18988", "title": "Surviving the Unseen: Predictive Defense for Novel Multi-Turn Multimodal Attacks", "url": "https://arxiv.org/abs/2605.18988", "published": "2026-05-18", "updated": "2026-05-18", "authors": [ "Doohee You" ], "categories": [ "cs.CR", "cs.AI" ], "topics": [ "agent-evaluation", "agent-safety", "workflow-agent" ], "score": 9, "relevance": "medium", "matched_queries": [ "autonomous-agent-llm" ], "arxiv_id": "2605.18988", "source": "arxiv", "source_id": "arxiv:2605.18988", "pdf_url": "https://arxiv.org/pdf/2605.18988", "primary_query": "autonomous-agent-llm" }, { "id": "2605.12978", "title": "Useful Memories Become Faulty When Continuously Updated by LLMs", "url": "https://arxiv.org/abs/2605.12978", "published": "2026-05-13", "updated": "2026-05-13", "authors": [ "Dylan Zhang", "Yanshan Lin", "Zhengkun Wu", "Yihang Sun", "Bingxuan Li", "Dianqi Li", "Hao Peng" ], "categories": [ "cs.AI" ], "topics": [ "memory", "tool-use" ], "score": 9, "relevance": "medium", "matched_queries": [ "agent-memory" ], "arxiv_id": "2605.12978", "source": "arxiv", "source_id": "arxiv:2605.12978", "pdf_url": "https://arxiv.org/pdf/2605.12978", "primary_query": "agent-memory" }, { "id": "2605.13918", "title": "CA2: Code-Aware Agent for Automated Game Testing", "url": "https://arxiv.org/abs/2605.13918", "published": "2026-05-13", "updated": "2026-05-13", "authors": [ "Valliappan Chidambaram Adaikkappan", "Vincent Martineau", "Joshua Romoff", "David Meger" ], "categories": [ "cs.SE", "cs.LG" ], "topics": [ "agent-evaluation", "rag", "tool-use" ], "score": 9, "relevance": "medium", "matched_queries": [ "function-calling" ], "arxiv_id": "2605.13918", "source": "arxiv", "source_id": "arxiv:2605.13918", "pdf_url": "https://arxiv.org/pdf/2605.13918", "primary_query": "function-calling" }, { "id": "2605.14038", "title": "Model-Adaptive Tool Necessity Reveals the Knowing-Doing Gap in LLM Tool Use", "url": "https://arxiv.org/abs/2605.14038", "published": "2026-05-13", "updated": "2026-05-17", "authors": [ "Yize Cheng", "Chenrui Fan", "Mahdi JafariRaviz", "Keivan Rezaei", "Soheil Feizi" ], "categories": [ "cs.AI" ], "topics": [ "tool-use" ], "score": 9, "relevance": "medium", "matched_queries": [ "autonomous-agent-llm" ], "arxiv_id": "2605.14038", "source": "arxiv", "source_id": "arxiv:2605.14038", "pdf_url": "https://arxiv.org/pdf/2605.14038", "primary_query": "autonomous-agent-llm" }, { "id": "2605.13414", "title": "TRIAGE: Evaluating Prospective Metacognitive Control in LLMs under Resource Constraints", "url": "https://arxiv.org/abs/2605.13414", "published": "2026-05-13", "updated": "2026-05-13", "authors": [ "Zabir Al Nazi", "Shubhashis Roy Dipta" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "planning", "reasoning" ], "score": 9, "relevance": "medium", "matched_queries": [ "autonomous-agent-llm" ], "arxiv_id": "2605.13414", "source": "arxiv", "source_id": "arxiv:2605.13414", "pdf_url": "https://arxiv.org/pdf/2605.13414", "primary_query": "autonomous-agent-llm" }, { "id": "2605.13898", "title": "Bidirectional Empowerment of Metamorphic Testing and Large Language Models: A Systematic Survey", "url": "https://arxiv.org/abs/2605.13898", "published": "2026-05-12", "updated": "2026-05-12", "authors": [ "Zheng Zheng", "Zenghui Zhou", "Yinwang Xu", "Daixu Ren", "Tsong Yueh Chen" ], "categories": [ "cs.SE" ], "topics": [ "agent-evaluation", "rag", "reasoning" ], "score": 9, "relevance": "medium", "matched_queries": [ "autonomous-agent-llm" ], "arxiv_id": "2605.13898", "source": "arxiv", "source_id": "arxiv:2605.13898", "pdf_url": "https://arxiv.org/pdf/2605.13898", "primary_query": "autonomous-agent-llm" }, { "id": "2605.06524", "title": "Process Matters more than Output for Distinguishing Humans from Machines", "url": "https://arxiv.org/abs/2605.06524", "published": "2026-05-07", "updated": "2026-05-09", "authors": [ "Milena Rmus", "Mathew D. Hardy", "Thomas L. Griffiths", "Mayank Agrawal" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "computer-use", "tool-use" ], "score": 9, "relevance": "medium", "matched_queries": [ "autonomous-agent-llm" ], "arxiv_id": "2605.06524", "source": "arxiv", "source_id": "arxiv:2605.06524", "pdf_url": "https://arxiv.org/pdf/2605.06524", "primary_query": "autonomous-agent-llm" }, { "id": "2603.09036", "title": "SCALAR: Learning and Composing Skills through LLM Guided Symbolic Planning and Deep RL Grounding", "url": "https://arxiv.org/abs/2603.09036", "published": "2026-03-10", "updated": "2026-03-10", "authors": [ "Renos Zabounidis", "Yue Wu", "Simon Stepputtis", "Woojun Kim", "Yuanzhi Li", "Tom Mitchell", "Katia Sycara" ], "categories": [ "cs.LG" ], "topics": [ "computer-use", "planning", "rag", "tool-use" ], "score": 9, "relevance": "medium", "matched_queries": [ "language-agent" ], "arxiv_id": "2603.09036", "source": "arxiv", "source_id": "arxiv:2603.09036", "pdf_url": "https://arxiv.org/pdf/2603.09036", "primary_query": "language-agent" }, { "id": "2603.16901", "title": "From Language to Action in Arabic: Reliable Structured Tool Calling via Data-Centric Fine-Tuning", "url": "https://arxiv.org/abs/2603.16901", "published": "2026-03-04", "updated": "2026-03-04", "authors": [ "Omer Nacar", "Deema Alquffari", "Saleh Alsharideh", "Adeem AlOtaibi", "Abdulaziz Alabdulkarim", "Leen Alhazmi", "Nada Alomar", "Wareef Alzubaidi", "Nada Alsultan", "Ahmed Alrabghi", "Demah Alhoshan", "Rana Alsayyari", "Hamed Alruwaili", "Albaraa Jaafar", "Khaled Alusmani", "Abdulaziz Alsohimy", "Munirah Alsubaie", "Shahd Aldukhayil", "Arwa Alali", "Yazeed BinShihah", "Razan Alsulaymi", "Nourah Alhumaid", "Razan Abdulsalam", "Reem Alamoudi", "Mohammed Alkhalifa" ], "categories": [ "cs.LG", "cs.AI" ], "topics": [ "agent-safety", "reasoning", "tool-use" ], "score": 9, "relevance": "medium", "matched_queries": [ "function-calling" ], "arxiv_id": "2603.16901", "source": "arxiv", "source_id": "arxiv:2603.16901", "pdf_url": "https://arxiv.org/pdf/2603.16901", "primary_query": "function-calling" }, { "id": "2602.09372", "title": "AgentSkiller: Scaling Generalist Agent Intelligence through Semantically Integrated Cross-Domain Data Synthesis", "url": "https://arxiv.org/abs/2602.09372", "published": "2026-02-10", "updated": "2026-02-10", "authors": [ "Zexu Sun", "Bokai Ji", "Hengyi Cai", "Shuaiqiang Wang", "Lei Wang", "Guangxia Li", "Xu Chen" ], "categories": [ "cs.CL" ], "topics": [ "planning", "tool-use" ], "score": 9, "relevance": "medium", "matched_queries": [ "function-calling" ], "arxiv_id": "2602.09372", "source": "arxiv", "source_id": "arxiv:2602.09372", "pdf_url": "https://arxiv.org/pdf/2602.09372", "primary_query": "function-calling" }, { "id": "2601.09292", "title": "Blue Teaming Function-Calling Agents", "url": "https://arxiv.org/abs/2601.09292", "published": "2026-01-14", "updated": "2026-01-14", "authors": [ "Greta Dolcetti", "Giulio Zizzo", "Sergio Maffeis" ], "categories": [ "cs.CR", "cs.AI" ], "topics": [ "agent-evaluation" ], "score": 9, "relevance": "medium", "matched_queries": [ "function-calling" ], "arxiv_id": "2601.09292", "source": "arxiv", "source_id": "arxiv:2601.09292", "pdf_url": "https://arxiv.org/pdf/2601.09292", "primary_query": "function-calling" }, { "id": "2601.05366", "title": "Lost in Execution: On the Multilingual Robustness of Tool Calling in Large Language Models", "url": "https://arxiv.org/abs/2601.05366", "published": "2026-01-08", "updated": "2026-06-28", "authors": [ "Zheng Luo", "T Pranav Kutralingam", "Ogochukwu N Okoani", "Wanpeng Xu", "Hua Wei", "Xiyang Hu" ], "categories": [ "cs.CL", "cs.AI", "cs.LG" ], "topics": [ "agent-evaluation", "tool-use" ], "score": 9, "relevance": "medium", "matched_queries": [ "function-calling" ], "arxiv_id": "2601.05366", "source": "arxiv", "source_id": "arxiv:2601.05366", "pdf_url": "https://arxiv.org/pdf/2601.05366", "primary_query": "function-calling" }, { "id": "2509.18076", "title": "Improving Large Language Models Function Calling and Interpretability via Guided-Structured Templates", "url": "https://arxiv.org/abs/2509.18076", "published": "2025-09-22", "updated": "2025-09-22", "authors": [ "Hy Dang", "Tianyi Liu", "Zhuofeng Wu", "Jingfeng Yang", "Haoming Jiang", "Tao Yang", "Pei Chen", "Zhengyang Wang", "Helen Wang", "Huasheng Li", "Bing Yin", "Meng Jiang" ], "categories": [ "cs.AI" ], "topics": [ "computer-use", "rag", "reasoning", "tool-use" ], "score": 9, "relevance": "medium", "matched_queries": [ "function-calling" ], "arxiv_id": "2509.18076", "source": "arxiv", "source_id": "arxiv:2509.18076", "pdf_url": "https://arxiv.org/pdf/2509.18076", "primary_query": "function-calling" }, { "id": "2508.09125", "title": "Complex Logical Instruction Generation", "url": "https://arxiv.org/abs/2508.09125", "published": "2025-08-12", "updated": "2026-01-27", "authors": [ "Mian Zhang", "Shujian Liu", "Sixun Dong", "Ming Yin", "Yebowen Hu", "Xun Wang", "Steven Ma", "Song Wang", "Sathish Reddy Indurthi", "Haoyun Deng", "Zhiyu Zoey Chen", "Kaiqiang Song" ], "categories": [ "cs.CL", "cs.LG" ], "topics": [ "agent-evaluation", "reasoning" ], "score": 9, "relevance": "medium", "matched_queries": [ "function-calling" ], "arxiv_id": "2508.09125", "source": "arxiv", "source_id": "arxiv:2508.09125", "pdf_url": "https://arxiv.org/pdf/2508.09125", "primary_query": "function-calling" }, { "id": "2607.05804", "title": "TurnOPD: Making On-Policy Distillation Turn-Aware for Efficient Long-Horizon Agent Training", "url": "https://arxiv.org/abs/2607.05804", "published": "2026-07-07", "updated": "2026-07-07", "authors": [ "Yuhang Zhou", "Kai Zheng", "Haoling Li", "Dengyun Peng", "Can Xu", "Jingjing Chen" ], "categories": [ "cs.AI", "cs.CL" ], "topics": [ "planning" ], "score": 8, "relevance": "medium", "matched_queries": [ "language-agent" ], "arxiv_id": "2607.05804", "source": "arxiv", "source_id": "arxiv:2607.05804", "pdf_url": "https://arxiv.org/pdf/2607.05804", "primary_query": "language-agent" }, { "id": "2607.04728", "title": "Turning Off-Policy Tokens On-Policy: A Plug-in Approach for Improving LLM Alignment", "url": "https://arxiv.org/abs/2607.04728", "published": "2026-07-06", "updated": "2026-07-06", "authors": [ "Yu Li", "Xiuyu Li", "Mingyang Yi", "Jiaxing Wang", "zhangliangxu", "Zhaolong Xing", "Zhen Chen" ], "categories": [ "cs.CL", "cs.AI", "cs.LG" ], "topics": [ "agent-evaluation", "agent-safety" ], "score": 8, "relevance": "medium", "matched_queries": [ "agent-evaluation" ], "arxiv_id": "2607.04728", "source": "arxiv", "source_id": "arxiv:2607.04728", "pdf_url": "https://arxiv.org/pdf/2607.04728", "primary_query": "agent-evaluation" }, { "id": "2607.04542", "title": "Auto: The AGI Compiler", "url": "https://arxiv.org/abs/2607.04542", "published": "2026-07-05", "updated": "2026-07-05", "authors": [ "Jaber Jaber", "Osama Jaber" ], "categories": [ "cs.LG", "cs.AI", "cs.SE" ], "topics": [ "agent-evaluation" ], "score": 8, "relevance": "medium", "matched_queries": [ "llm-agent" ], "arxiv_id": "2607.04542", "source": "arxiv", "source_id": "arxiv:2607.04542", "pdf_url": "https://arxiv.org/pdf/2607.04542", "primary_query": "llm-agent" }, { "id": "2607.03193", "title": "Self-Specializing Vision-Language Transmon Chip Calibration in a Physics-Grounded Environment", "url": "https://arxiv.org/abs/2607.03193", "published": "2026-07-03", "updated": "2026-07-03", "authors": [ "Animesh Tripathy", "Aswanth Krishnan" ], "categories": [ "quant-ph", "cs.AI", "cs.LG" ], "topics": [ "planning", "rag", "tool-use", "world-model" ], "score": 8, "relevance": "medium", "matched_queries": [ "language-agent" ], "arxiv_id": "2607.03193", "source": "arxiv", "source_id": "arxiv:2607.03193", "pdf_url": "https://arxiv.org/pdf/2607.03193", "primary_query": "language-agent" }, { "id": "2607.03451", "title": "SkillOpt-Lite: Better and Faster Agent Self-evolution via One Line of Vibe", "url": "https://arxiv.org/abs/2607.03451", "published": "2026-07-03", "updated": "2026-07-03", "authors": [ "Yifei Shen", "Bo Li", "Xinjie Zhang" ], "categories": [ "cs.SE", "cs.AI", "cs.LG" ], "topics": [ "coding-agent", "tool-use" ], "score": 8, "relevance": "medium", "matched_queries": [ "coding-agent" ], "arxiv_id": "2607.03451", "source": "arxiv", "source_id": "arxiv:2607.03451", "pdf_url": "https://arxiv.org/pdf/2607.03451", "primary_query": "coding-agent" }, { "id": "2607.02931", "title": "VERITAS: Towards a General-Purpose Replication Tool for Scientific Research", "url": "https://arxiv.org/abs/2607.02931", "published": "2026-07-03", "updated": "2026-07-03", "authors": [ "Haokun Liu", "Filbert Aurelian Tjiaranata", "Chenhao Tan" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "coding-agent", "tool-use" ], "score": 8, "relevance": "medium", "matched_queries": [ "coding-agent" ], "arxiv_id": "2607.02931", "source": "arxiv", "source_id": "arxiv:2607.02931", "pdf_url": "https://arxiv.org/pdf/2607.02931", "primary_query": "coding-agent" }, { "id": "2607.02217", "title": "Affinage: genome-scale mechanistic gene annotation from the published literature", "url": "https://arxiv.org/abs/2607.02217", "published": "2026-07-02", "updated": "2026-07-02", "authors": [ "Matteo Di Bernardo", "Iain M. Cheeseman" ], "categories": [ "q-bio.GN" ], "topics": [ "agent-evaluation", "coding-agent", "computer-use", "rag", "reasoning" ], "score": 8, "relevance": "medium", "matched_queries": [ "agentic-ai" ], "arxiv_id": "2607.02217", "source": "arxiv", "source_id": "arxiv:2607.02217", "pdf_url": "https://arxiv.org/pdf/2607.02217", "primary_query": "agentic-ai" }, { "id": "2607.01639", "title": "Autonomous discovery of traffic laws with AI traffic scientists", "url": "https://arxiv.org/abs/2607.01639", "published": "2026-07-02", "updated": "2026-07-02", "authors": [ "Xingyuan Dai", "Yue Liu", "Xiaoyan Gong", "Qinghai Miao", "Junyou Shang", "Yutong Wang", "Chao Guo", "Yonglin Tian", "Yizhang Chai", "Chao Xiang", "Yisheng Lv", "Fei-Yue Wang" ], "categories": [ "cs.AI" ], "topics": [ "memory", "planning", "workflow-agent" ], "score": 8, "relevance": "medium", "matched_queries": [ "agentic-ai" ], "arxiv_id": "2607.01639", "source": "arxiv", "source_id": "arxiv:2607.01639", "pdf_url": "https://arxiv.org/pdf/2607.01639", "primary_query": "agentic-ai" }, { "id": "2607.00910", "title": "Calibrating the Instrument: Controllability of an LLM-Driven Synthetic Population", "url": "https://arxiv.org/abs/2607.00910", "published": "2026-07-01", "updated": "2026-07-01", "authors": [ "Mirko Degli Esposti" ], "categories": [ "cs.MA" ], "topics": [ "agent-evaluation", "world-model" ], "score": 8, "relevance": "medium", "matched_queries": [ "llm-agent" ], "arxiv_id": "2607.00910", "source": "arxiv", "source_id": "arxiv:2607.00910", "pdf_url": "https://arxiv.org/pdf/2607.00910", "primary_query": "llm-agent" }, { "id": "2607.00941", "title": "From Runtime Records to Legal Findings: An Evidentiary-Adequacy Criterion for Agentic AI Oversight", "url": "https://arxiv.org/abs/2607.00941", "published": "2026-07-01", "updated": "2026-07-01", "authors": [ "Jeroen Janssen" ], "categories": [ "cs.CY" ], "topics": [ "agent" ], "score": 8, "relevance": "medium", "matched_queries": [ "agentic-ai" ], "arxiv_id": "2607.00941", "source": "arxiv", "source_id": "arxiv:2607.00941", "pdf_url": "https://arxiv.org/pdf/2607.00941", "primary_query": "agentic-ai" }, { "id": "2607.00316", "title": "Evolving Intelligent Complex Systems via Intellicise Networks: Architecture, Technologies, and Pathways", "url": "https://arxiv.org/abs/2607.00316", "published": "2026-07-01", "updated": "2026-07-01", "authors": [ "Ping Zhang", "Rui Meng", "Xiaodong Xu", "Song Gao", "Zixuan Huang", "Yaheng Wang", "Yinqiu Liu", "Ruichen Zhang", "Yiming Liu", "Kaiwen Yu", "Yaping Sun", "Han Meng", "Haonan Tong", "Huishi Song", "Qianqian Yang", "Shuoyao Wang", "Lexi Xu", "Qinghe Du", "Geng Sun", "Jiawen Kang", "Gang Wu", "Yiqing Zhou", "Haixia Zhang", "Zesong Fei", "Aimin Hao", "Ming Li" ], "categories": [ "eess.SP" ], "topics": [ "agent-safety", "embodied-agent", "planning", "tool-use" ], "score": 8, "relevance": "medium", "matched_queries": [ "agentic-ai" ], "arxiv_id": "2607.00316", "source": "arxiv", "source_id": "arxiv:2607.00316", "pdf_url": "https://arxiv.org/pdf/2607.00316", "primary_query": "agentic-ai" }, { "id": "2607.01299", "title": "HYPIC: Accelerating Hybrid-Attention LLM Serving with Position-Independent Caching", "url": "https://arxiv.org/abs/2607.01299", "published": "2026-07-01", "updated": "2026-07-01", "authors": [ "Yifei Liu", "Juntong Wu", "Yang Liu", "Junhao Hu", "Minghao Li", "Xiaoxu Chen", "Weihang Chen" ], "categories": [ "cs.DC" ], "topics": [ "agent-evaluation", "rag" ], "score": 8, "relevance": "medium", "matched_queries": [ "rag-agent" ], "arxiv_id": "2607.01299", "source": "arxiv", "source_id": "arxiv:2607.01299", "pdf_url": "https://arxiv.org/pdf/2607.01299", "primary_query": "rag-agent" }, { "id": "2606.31492", "title": "Higher-order hopping-parameter expansion by human-AI collaboration", "url": "https://arxiv.org/abs/2606.31492", "published": "2026-06-30", "updated": "2026-07-06", "authors": [ "Masakiyo Kitazawa", "Tatsuya Wada" ], "categories": [ "hep-lat", "hep-ph" ], "topics": [ "agent-evaluation", "coding-agent", "multi-agent" ], "score": 8, "relevance": "medium", "matched_queries": [ "coding-agent" ], "arxiv_id": "2606.31492", "source": "arxiv", "source_id": "arxiv:2606.31492", "pdf_url": "https://arxiv.org/pdf/2606.31492", "primary_query": "coding-agent" }, { "id": "2606.30774", "title": "What Drives Interactive Improvement from Feedback?", "url": "https://arxiv.org/abs/2606.30774", "published": "2026-06-29", "updated": "2026-06-29", "authors": [ "Bartłomiej Cupiał", "Jan Łojek", "Mikołaj Garstecki", "Szymon Pobłocki", "Alicja Ziarko", "Piotr Miłoś" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "computer-use", "tool-use" ], "score": 8, "relevance": "medium", "matched_queries": [ "language-agent" ], "arxiv_id": "2606.30774", "source": "arxiv", "source_id": "arxiv:2606.30774", "pdf_url": "https://arxiv.org/pdf/2606.30774", "primary_query": "language-agent" }, { "id": "2606.29916", "title": "EVAF: A Test-Retest Protocol for Selective Parametric Consolidation", "url": "https://arxiv.org/abs/2606.29916", "published": "2026-06-29", "updated": "2026-06-29", "authors": [ "Haoliang Han" ], "categories": [ "cs.LG", "cs.AI" ], "topics": [ "agent-evaluation", "memory", "rag" ], "score": 8, "relevance": "medium", "matched_queries": [ "language-agent" ], "arxiv_id": "2606.29916", "source": "arxiv", "source_id": "arxiv:2606.29916", "pdf_url": "https://arxiv.org/pdf/2606.29916", "primary_query": "language-agent" }, { "id": "2606.30963", "title": "Loc2Repair: A Framework for Evaluating the Impact of File-Level Issue Localization in Repo-Level LLM Repair", "url": "https://arxiv.org/abs/2606.30963", "published": "2026-06-29", "updated": "2026-06-29", "authors": [ "Mohammad Nour Al Awad", "Sergey Ivanov" ], "categories": [ "cs.SE", "cs.AI" ], "topics": [ "agent-evaluation", "coding-agent", "computer-use" ], "score": 8, "relevance": "medium", "matched_queries": [ "coding-agent" ], "arxiv_id": "2606.30963", "source": "arxiv", "source_id": "arxiv:2606.30963", "pdf_url": "https://arxiv.org/pdf/2606.30963", "primary_query": "coding-agent" }, { "id": "2606.29556", "title": "Persona-Trained Monte Carlo: Estimating Market-Outcome Distributions via Swarms of Persona-Conditioned Neural Policy Bots in a Limit Order Book", "url": "https://arxiv.org/abs/2606.29556", "published": "2026-06-28", "updated": "2026-06-28", "authors": [ "Salavat Ishbulatov" ], "categories": [ "cs.LG", "cs.MA" ], "topics": [ "agent-safety", "computer-use", "tool-use" ], "score": 8, "relevance": "medium", "matched_queries": [ "coding-agent" ], "arxiv_id": "2606.29556", "source": "arxiv", "source_id": "arxiv:2606.29556", "pdf_url": "https://arxiv.org/pdf/2606.29556", "primary_query": "coding-agent" }, { "id": "2606.28690", "title": "Formal Security Analysis of Agent Protocol Composition", "url": "https://arxiv.org/abs/2606.28690", "published": "2026-06-27", "updated": "2026-06-27", "authors": [ "Shenghan Zheng", "Qifan Zhang", "Zheng Zhang", "Haonan Li", "Christophe Hauser" ], "categories": [ "cs.CR" ], "topics": [ "agent-safety", "tool-use" ], "score": 8, "relevance": "medium", "matched_queries": [ "ai-agent" ], "arxiv_id": "2606.28690", "source": "arxiv", "source_id": "arxiv:2606.28690", "pdf_url": "https://arxiv.org/pdf/2606.28690", "primary_query": "ai-agent" }, { "id": "2606.30678", "title": "NanoVer: An open-source framework for interactive molecular dynamics in extended reality (iMD-XR) on commodity hardware", "url": "https://arxiv.org/abs/2606.30678", "published": "2026-06-26", "updated": "2026-06-26", "authors": [ "Mark D. Wonnacott", "Luis Ernesto Toledo Castro", "Harry J. Stroud", "Ludovica Aisa", "Mohamed Dhouioui", "Rhoslyn Roebuck Williams", "Denis Protopopov", "Sila Sobrado", "David R. Glowacki" ], "categories": [ "physics.chem-ph", "physics.bio-ph", "physics.ed-ph" ], "topics": [ "computer-use", "reasoning", "tool-use", "world-model" ], "score": 8, "relevance": "medium", "matched_queries": [ "ai-agent" ], "arxiv_id": "2606.30678", "source": "arxiv", "source_id": "arxiv:2606.30678", "pdf_url": "https://arxiv.org/pdf/2606.30678", "primary_query": "ai-agent" }, { "id": "2606.25550", "title": "On the Viability of Requirements Generation From Code: An Experience Report", "url": "https://arxiv.org/abs/2606.25550", "published": "2026-06-24", "updated": "2026-06-24", "authors": [ "Alexander Korn", "Jone Bartel", "Max Unterbusch", "Andreas Vogelsang" ], "categories": [ "cs.SE" ], "topics": [ "agent-evaluation", "rag" ], "score": 8, "relevance": "medium", "matched_queries": [ "rag-agent" ], "arxiv_id": "2606.25550", "source": "arxiv", "source_id": "arxiv:2606.25550", "pdf_url": "https://arxiv.org/pdf/2606.25550", "primary_query": "rag-agent" }, { "id": "2606.23679", "title": "Semantic Browsing: Controllable Diversity for Image Generation", "url": "https://arxiv.org/abs/2606.23679", "published": "2026-06-22", "updated": "2026-06-22", "authors": [ "Sara Dorfman", "Maya Vishnevsky", "Omer Dahary", "Or Patashnik", "Daniel Cohen-Or" ], "categories": [ "cs.CV", "cs.AI", "cs.GR", "cs.LG" ], "topics": [ "rag", "workflow-agent" ], "score": 8, "relevance": "medium", "matched_queries": [ "agentic-ai" ], "arxiv_id": "2606.23679", "source": "arxiv", "source_id": "arxiv:2606.23679", "pdf_url": "https://arxiv.org/pdf/2606.23679", "primary_query": "agentic-ai" }, { "id": "2606.20161", "title": "ARTEMIS: Agent-guided Reliability-aware Temporal Mask Evolution for Imperfectly Supervised Video Polyp Segmentation", "url": "https://arxiv.org/abs/2606.20161", "published": "2026-06-18", "updated": "2026-06-18", "authors": [ "Tong Wang", "Siwen Wang", "Yaolei Qi", "Jinxing Zhou", "Yuting He", "Guanyu Yang", "Yutong Xie" ], "categories": [ "cs.CV" ], "topics": [ "computer-use", "multi-agent" ], "score": 8, "relevance": "medium", "matched_queries": [ "language-agent" ], "arxiv_id": "2606.20161", "source": "arxiv", "source_id": "arxiv:2606.20161", "pdf_url": "https://arxiv.org/pdf/2606.20161", "primary_query": "language-agent" }, { "id": "2606.20453", "title": "Directors Duties in the Age of Agentic Artificial Intelligence", "url": "https://arxiv.org/abs/2606.20453", "published": "2026-06-18", "updated": "2026-06-18", "authors": [ "Deirdre Ahern" ], "categories": [ "cs.CY", "cs.HC" ], "topics": [ "agent" ], "score": 8, "relevance": "medium", "matched_queries": [ "agentic-ai" ], "arxiv_id": "2606.20453", "source": "arxiv", "source_id": "arxiv:2606.20453", "pdf_url": "https://arxiv.org/pdf/2606.20453", "primary_query": "agentic-ai" }, { "id": "2606.19670", "title": "PiMiX 2.0: AI-enhanced Data Fusion for Radiographic Imaging and Tomography", "url": "https://arxiv.org/abs/2606.19670", "published": "2026-06-18", "updated": "2026-06-20", "authors": [ "Zhehui Wang", "Shanny Lin", "Nicholas Amano", "Susan S. Glenn", "Ramya Gurunathan", "Katie Liu", "Nathan E. Peterson", "Michelle A. Espy", "Adam Thompson", "Amy J. Clarke", "Ray T. Chen" ], "categories": [ "physics.ins-det", "physics.data-an" ], "topics": [ "computer-use", "reasoning", "tool-use", "workflow-agent" ], "score": 8, "relevance": "medium", "matched_queries": [ "agentic-ai" ], "arxiv_id": "2606.19670", "source": "arxiv", "source_id": "arxiv:2606.19670", "pdf_url": "https://arxiv.org/pdf/2606.19670", "primary_query": "agentic-ai" }, { "id": "2606.20708", "title": "Simulated Customers Never Walk Away: Decision Fidelity of LLM User Simulators Measured Against Real Purchase Outcomes", "url": "https://arxiv.org/abs/2606.20708", "published": "2026-06-16", "updated": "2026-06-16", "authors": [ "Liang Chen" ], "categories": [ "cs.AI", "cs.CL", "cs.HC" ], "topics": [ "agent-evaluation", "world-model" ], "score": 8, "relevance": "medium", "matched_queries": [ "agent-evaluation" ], "arxiv_id": "2606.20708", "source": "arxiv", "source_id": "arxiv:2606.20708", "pdf_url": "https://arxiv.org/pdf/2606.20708", "primary_query": "agent-evaluation" }, { "id": "2606.16903", "title": "Directory-Aware Query and Maintenance in Vector Databases", "url": "https://arxiv.org/abs/2606.16903", "published": "2026-06-15", "updated": "2026-06-15", "authors": [ "Mengzhao Wang", "Zheng Gong", "Jingpei Hu", "Jiajie Fu", "Maojia Sheng", "Junwen Chen", "Yifan Zhu" ], "categories": [ "cs.DB" ], "topics": [ "agent-evaluation", "rag", "workflow-agent" ], "score": 8, "relevance": "medium", "matched_queries": [ "agent-memory" ], "arxiv_id": "2606.16903", "source": "arxiv", "source_id": "arxiv:2606.16903", "pdf_url": "https://arxiv.org/pdf/2606.16903", "primary_query": "agent-memory" }, { "id": "2606.12485", "title": "Speculative Rollback Correction for Quality-Diverse Web Agent Imitation", "url": "https://arxiv.org/abs/2606.12485", "published": "2026-06-10", "updated": "2026-06-10", "authors": [ "Longkun Hao", "Hongyu Lin", "Hao Li", "Zhichao Yang", "Haojie Hao", "Dongshuo Huang", "Haitao Yang", "Hongyu Ge", "Ming jie Xie", "Yanjun Wu", "Zi Hao Yin", "Yan Bai", "Yihang Lou" ], "categories": [ "cs.LG", "cs.AI" ], "topics": [ "computer-use", "reasoning", "tool-use" ], "score": 8, "relevance": "medium", "matched_queries": [ "web-gui-agent" ], "arxiv_id": "2606.12485", "source": "arxiv", "source_id": "arxiv:2606.12485", "pdf_url": "https://arxiv.org/pdf/2606.12485", "primary_query": "web-gui-agent" }, { "id": "2606.08944", "title": "LongRTL: Graph-Similarity-Guided LLM-driven Long Context RTL Optimization", "url": "https://arxiv.org/abs/2606.08944", "published": "2026-06-08", "updated": "2026-06-08", "authors": [ "Yuyang Ye", "Che-Kuan Shen", "Xiangfei Hu", "Yuchen Liu", "Shuo Yin", "Xufeng Yao", "Bei Yu", "Tsung-Yi Ho" ], "categories": [ "cs.AR", "cs.PL" ], "topics": [ "agent-evaluation", "computer-use", "rag" ], "score": 8, "relevance": "medium", "matched_queries": [ "rag-agent" ], "arxiv_id": "2606.08944", "source": "arxiv", "source_id": "arxiv:2606.08944", "pdf_url": "https://arxiv.org/pdf/2606.08944", "primary_query": "rag-agent" }, { "id": "2606.08755", "title": "Co-Evolving Skill Generation and Policy Optimization", "url": "https://arxiv.org/abs/2606.08755", "published": "2026-06-07", "updated": "2026-06-07", "authors": [ "Zhiwei Zhang", "Yudi Lin", "Nikki Lijing Kuang", "Linlin Wu", "Xiaomin Li", "Songtao Liu", "Fenglong Ma" ], "categories": [ "cs.CL" ], "topics": [ "agent-evaluation", "rag" ], "score": 8, "relevance": "medium", "matched_queries": [ "language-agent" ], "arxiv_id": "2606.08755", "source": "arxiv", "source_id": "arxiv:2606.08755", "pdf_url": "https://arxiv.org/pdf/2606.08755", "primary_query": "language-agent" }, { "id": "2606.04751", "title": "FALSIFYBENCH: Evaluating Inductive Reasoning in LLMs with Rule Discovery Games", "url": "https://arxiv.org/abs/2606.04751", "published": "2026-06-03", "updated": "2026-06-03", "authors": [ "Leonardo Bertolazzi", "Katya Tentori", "Raffaella Bernardi" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "reasoning" ], "score": 8, "relevance": "medium", "matched_queries": [ "autonomous-agent-llm" ], "arxiv_id": "2606.04751", "source": "arxiv", "source_id": "arxiv:2606.04751", "pdf_url": "https://arxiv.org/pdf/2606.04751", "primary_query": "autonomous-agent-llm" }, { "id": "2606.03777", "title": "From Control Boundary to Insurance Claim: Reconstructing AI-Mediated Losses Through the CER Framework", "url": "https://arxiv.org/abs/2606.03777", "published": "2026-06-02", "updated": "2026-06-02", "authors": [ "Alex Leung", "Rex Zhang", "Kentaroh Toyoda", "SiewMei Loh" ], "categories": [ "cs.AI", "cs.CR", "q-fin.RM" ], "topics": [ "agent-evaluation", "agent-safety", "rag", "tool-use" ], "score": 8, "relevance": "medium", "matched_queries": [ "rag-agent" ], "arxiv_id": "2606.03777", "source": "arxiv", "source_id": "arxiv:2606.03777", "pdf_url": "https://arxiv.org/pdf/2606.03777", "primary_query": "rag-agent" }, { "id": "2606.02483", "title": "Ghost Tool Calls: Issue-Time Privacy for Speculative Agent Tools", "url": "https://arxiv.org/abs/2606.02483", "published": "2026-06-01", "updated": "2026-06-01", "authors": [ "Bardia Mohammadi", "Lars Klein", "Akhil Arora", "Laurent Bindschaedler" ], "categories": [ "cs.CR", "cs.AI", "cs.CL" ], "topics": [ "agent-evaluation", "tool-use" ], "score": 8, "relevance": "medium", "matched_queries": [ "language-agent" ], "arxiv_id": "2606.02483", "source": "arxiv", "source_id": "arxiv:2606.02483", "pdf_url": "https://arxiv.org/pdf/2606.02483", "primary_query": "language-agent" }, { "id": "2606.01212", "title": "DiscourseFlip: An Oblique Discourse-Level Opinion Manipulation Attack against Black-box Retrieval-Augmented Generation", "url": "https://arxiv.org/abs/2606.01212", "published": "2026-05-31", "updated": "2026-06-03", "authors": [ "Yuyang Gong", "Miaokun Chen", "Jiawei Liu", "Zhuo Chen", "Guoxiu He", "Wei Lu", "XiaoFeng Wang", "Xiaozhong Liu" ], "categories": [ "cs.CL", "cs.AI", "cs.CR", "cs.IR" ], "topics": [ "agent-evaluation", "agent-safety", "computer-use", "rag" ], "score": 8, "relevance": "medium", "matched_queries": [ "rag-agent" ], "arxiv_id": "2606.01212", "source": "arxiv", "source_id": "arxiv:2606.01212", "pdf_url": "https://arxiv.org/pdf/2606.01212", "primary_query": "rag-agent" }, { "id": "2605.26754", "title": "Cordon-MAS: Defending RAG against Knowledge Poisoning via Information-Flow Control", "url": "https://arxiv.org/abs/2605.26754", "published": "2026-05-26", "updated": "2026-05-26", "authors": [ "Zhe Yu", "Wenpeng Xing", "Gaolei Li", "Shuguang Xiong", "Hongzhi Wang", "Xuyang Teng", "Meng Han" ], "categories": [ "cs.CR", "cs.AI" ], "topics": [ "agent-evaluation", "memory", "rag", "tool-use" ], "score": 8, "relevance": "medium", "matched_queries": [ "rag-agent" ], "arxiv_id": "2605.26754", "source": "arxiv", "source_id": "arxiv:2605.26754", "pdf_url": "https://arxiv.org/pdf/2605.26754", "primary_query": "rag-agent" }, { "id": "2605.25379", "title": "EfficientGraph-RAG: Structured Retrieval-State Management for Cross-Task Retrieval-Augmented Generation", "url": "https://arxiv.org/abs/2605.25379", "published": "2026-05-25", "updated": "2026-05-25", "authors": [ "Miaohe Niu", "Lianlei Shan", "Zhengtao Yu", "Jingbo Zhu", "Tong Xiao" ], "categories": [ "cs.CL" ], "topics": [ "agent-evaluation", "rag" ], "score": 8, "relevance": "medium", "matched_queries": [ "rag-agent" ], "arxiv_id": "2605.25379", "source": "arxiv", "source_id": "arxiv:2605.25379", "pdf_url": "https://arxiv.org/pdf/2605.25379", "primary_query": "rag-agent" }, { "id": "2605.21401", "title": "Open-source LLMs administer maximum electric shocks in a Milgram-like obedience experiment", "url": "https://arxiv.org/abs/2605.21401", "published": "2026-05-20", "updated": "2026-06-23", "authors": [ "Roland Pihlakas", "Jan Llenzl Dagohoy" ], "categories": [ "cs.CY", "cs.AI" ], "topics": [ "agent-safety", "tool-use" ], "score": 8, "relevance": "medium", "matched_queries": [ "autonomous-agent-llm" ], "arxiv_id": "2605.21401", "source": "arxiv", "source_id": "arxiv:2605.21401", "pdf_url": "https://arxiv.org/pdf/2605.21401", "primary_query": "autonomous-agent-llm" }, { "id": "2605.18414", "title": "Prompts Don't Protect: Architectural Enforcement via MCP Proxy for LLM Tool Access Control", "url": "https://arxiv.org/abs/2605.18414", "published": "2026-05-18", "updated": "2026-05-18", "authors": [ "Rohith Uppala" ], "categories": [ "cs.CR", "cs.AI" ], "topics": [ "agent-safety", "tool-use" ], "score": 8, "relevance": "medium", "matched_queries": [ "autonomous-agent-llm" ], "arxiv_id": "2605.18414", "source": "arxiv", "source_id": "arxiv:2605.18414", "pdf_url": "https://arxiv.org/pdf/2605.18414", "primary_query": "autonomous-agent-llm" }, { "id": "2605.14588", "title": "Silent Collapse in Recursive Learning Systems", "url": "https://arxiv.org/abs/2605.14588", "published": "2026-05-14", "updated": "2026-05-19", "authors": [ "Zhipeng Zhang" ], "categories": [ "cs.LG" ], "topics": [ "agent-evaluation", "rag", "tool-use" ], "score": 8, "relevance": "medium", "matched_queries": [ "autonomous-agent-llm" ], "arxiv_id": "2605.14588", "source": "arxiv", "source_id": "arxiv:2605.14588", "pdf_url": "https://arxiv.org/pdf/2605.14588", "primary_query": "autonomous-agent-llm" }, { "id": "2605.00651", "title": "EQSANS-CLI: A natural-language, agent-ready command-line tool for small-angle neutron scattering data reduction at EQ-SANS", "url": "https://arxiv.org/abs/2605.00651", "published": "2026-05-01", "updated": "2026-05-01", "authors": [ "Changwoo Do" ], "categories": [ "physics.ins-det" ], "topics": [ "computer-use", "tool-use", "workflow-agent" ], "score": 8, "relevance": "medium", "matched_queries": [ "language-agent" ], "arxiv_id": "2605.00651", "source": "arxiv", "source_id": "arxiv:2605.00651", "pdf_url": "https://arxiv.org/pdf/2605.00651", "primary_query": "language-agent" }, { "id": "2605.01047", "title": "LLM Ghostbusters: Surgical Hallucination Suppression via Adaptive Unlearning", "url": "https://arxiv.org/abs/2605.01047", "published": "2026-05-01", "updated": "2026-05-01", "authors": [ "Joseph Spracklen", "Pedram Aghazadeh", "Farinaz Koushanfar", "Murtuza Jadliwala" ], "categories": [ "cs.CR", "cs.AI", "cs.CL", "cs.LG" ], "topics": [ "agent-evaluation", "coding-agent" ], "score": 8, "relevance": "medium", "matched_queries": [ "autonomous-agent-llm" ], "arxiv_id": "2605.01047", "source": "arxiv", "source_id": "arxiv:2605.01047", "pdf_url": "https://arxiv.org/pdf/2605.01047", "primary_query": "autonomous-agent-llm" }, { "id": "2604.28138", "title": "Crab: A Semantics-Aware Checkpoint/Restore Runtime for Agent Sandboxes", "url": "https://arxiv.org/abs/2604.28138", "published": "2026-04-30", "updated": "2026-04-30", "authors": [ "Tianyuan Wu", "Chaokun Chang", "Lunxi Cao", "Wei Gao", "Wei Wang" ], "categories": [ "cs.OS", "cs.AI" ], "topics": [ "tool-use" ], "score": 8, "relevance": "medium", "matched_queries": [ "autonomous-agent-llm" ], "arxiv_id": "2604.28138", "source": "arxiv", "source_id": "arxiv:2604.28138", "pdf_url": "https://arxiv.org/pdf/2604.28138", "primary_query": "autonomous-agent-llm" }, { "id": "2603.10285", "title": "Conversational AI-Enhanced Exploration System to Query Large-Scale Digitised Collections of Natural History Museums", "url": "https://arxiv.org/abs/2603.10285", "published": "2026-03-11", "updated": "2026-03-11", "authors": [ "Yiyuan Wang", "Andrew Johnston", "Zoë Sadokierski", "Rhiannon Stephens", "Shane T. Ahyong" ], "categories": [ "cs.HC", "cs.AI", "cs.CY", "cs.DL", "cs.ET" ], "topics": [ "rag", "tool-use" ], "score": 8, "relevance": "medium", "matched_queries": [ "function-calling" ], "arxiv_id": "2603.10285", "source": "arxiv", "source_id": "arxiv:2603.10285", "pdf_url": "https://arxiv.org/pdf/2603.10285", "primary_query": "function-calling" }, { "id": "2602.08121", "title": "Initial Risk Probing and Feasibility Testing of Glow: a Generative AI-Powered Dialectical Behavior Therapy Skills Coach for Substance Use Recovery and HIV Prevention", "url": "https://arxiv.org/abs/2602.08121", "published": "2026-02-08", "updated": "2026-02-08", "authors": [ "Liying Wang", "Madison Lee", "Yunzhang Jiang", "Steven Chen", "Kewei Sha", "Yunhe Feng", "Frank Wong", "Lisa Hightow-Weidman", "Weichao Yuwen" ], "categories": [ "cs.AI" ], "topics": [ "agent-evaluation", "agent-safety", "rag", "tool-use" ], "score": 8, "relevance": "medium", "matched_queries": [ "agent-safety" ], "arxiv_id": "2602.08121", "source": "arxiv", "source_id": "arxiv:2602.08121", "pdf_url": "https://arxiv.org/pdf/2602.08121", "primary_query": "agent-safety" }, { "id": "2601.06937", "title": "mind_call: A Dataset for Mental Health Function Calling with Large Language Models", "url": "https://arxiv.org/abs/2601.06937", "published": "2026-01-11", "updated": "2026-01-11", "authors": [ "Fozle Rabbi Shafi", "M. Anwar Hossain", "Salimur Choudhury" ], "categories": [ "cs.AI", "cs.LG" ], "topics": [ "rag", "reasoning", "tool-use" ], "score": 8, "relevance": "medium", "matched_queries": [ "function-calling" ], "arxiv_id": "2601.06937", "source": "arxiv", "source_id": "arxiv:2601.06937", "pdf_url": "https://arxiv.org/pdf/2601.06937", "primary_query": "function-calling" }, { "id": "2510.13558", "title": "Steer-MoE: Efficient Audio-Language Alignment with a Mixture-of-Experts Steering Module", "url": "https://arxiv.org/abs/2510.13558", "published": "2025-10-15", "updated": "2025-10-15", "authors": [ "Ruitao Feng", "Bixi Zhang", "Sheng Liang", "Zheng Yuan" ], "categories": [ "cs.SD" ], "topics": [ "agent-safety", "reasoning" ], "score": 8, "relevance": "medium", "matched_queries": [ "function-calling" ], "arxiv_id": "2510.13558", "source": "arxiv", "source_id": "arxiv:2510.13558", "pdf_url": "https://arxiv.org/pdf/2510.13558", "primary_query": "function-calling" }, { "id": "2509.26463", "title": "ErrorPrism: Reconstructing Error Propagation Paths in Cloud Service Systems", "url": "https://arxiv.org/abs/2509.26463", "published": "2025-09-30", "updated": "2025-09-30", "authors": [ "Junsong Pu", "Yichen Li", "Zhuangbin Chen", "Jinyang Liu", "Zhihan Jiang", "Jianjun Chen", "Rui Shi", "Zibin Zheng", "Tieying Zhang" ], "categories": [ "cs.SE" ], "topics": [ "agent-evaluation", "tool-use" ], "score": 8, "relevance": "medium", "matched_queries": [ "function-calling" ], "arxiv_id": "2509.26463", "source": "arxiv", "source_id": "arxiv:2509.26463", "pdf_url": "https://arxiv.org/pdf/2509.26463", "primary_query": "function-calling" }, { "id": "2509.24229", "title": "Model Fusion with Multi-LoRA Inference for Tool-Enhanced Game Dialogue Agents", "url": "https://arxiv.org/abs/2509.24229", "published": "2025-09-29", "updated": "2025-09-29", "authors": [ "Kangxu Wang", "Ze Chen", "Chengcheng Wei", "Jiewen Zheng", "Jiarong He", "Max Gao" ], "categories": [ "cs.CL" ], "topics": [ "tool-use" ], "score": 8, "relevance": "medium", "matched_queries": [ "function-calling" ], "arxiv_id": "2509.24229", "source": "arxiv", "source_id": "arxiv:2509.24229", "pdf_url": "https://arxiv.org/pdf/2509.24229", "primary_query": "function-calling" }, { "id": "2509.04518", "title": "Advancing SLM Tool-Use Capability using Reinforcement Learning", "url": "https://arxiv.org/abs/2509.04518", "published": "2025-09-03", "updated": "2025-09-08", "authors": [ "Dhruvi Paprunia", "Vansh Kharidia", "Pankti Doshi" ], "categories": [ "cs.CL" ], "topics": [ "tool-use" ], "score": 8, "relevance": "medium", "matched_queries": [ "function-calling" ], "arxiv_id": "2509.04518", "source": "arxiv", "source_id": "arxiv:2509.04518", "pdf_url": "https://arxiv.org/pdf/2509.04518", "primary_query": "function-calling" }, { "id": "2607.04103", "title": "Governing Generative AI Across Financial Institutions: An SR 26-2-Compatible Framework for Generative AI Risk Control", "url": "https://arxiv.org/abs/2607.04103", "published": "2026-07-05", "updated": "2026-07-05", "authors": [ "Yiqing Wang", "Yixin Kang", "Luyun Lin", "Siqi Mao" ], "categories": [ "q-fin.RM", "cs.LG" ], "topics": [ "agent-safety", "tool-use", "workflow-agent" ], "score": 7, "relevance": "medium", "matched_queries": [ "agentic-ai" ], "arxiv_id": "2607.04103", "source": "arxiv", "source_id": "arxiv:2607.04103", "pdf_url": "https://arxiv.org/pdf/2607.04103", "primary_query": "agentic-ai" }, { "id": "2607.03181", "title": "Teaming Up with AI: Coordination and Cooperation", "url": "https://arxiv.org/abs/2607.03181", "published": "2026-07-03", "updated": "2026-07-03", "authors": [ "Nicole Immorlica", "Inbal Talgam-Cohen" ], "categories": [ "cs.GT", "cs.AI" ], "topics": [ "agent-safety", "multi-agent", "tool-use" ], "score": 7, "relevance": "medium", "matched_queries": [ "ai-agent" ], "arxiv_id": "2607.03181", "source": "arxiv", "source_id": "arxiv:2607.03181", "pdf_url": "https://arxiv.org/pdf/2607.03181", "primary_query": "ai-agent" }, { "id": "2607.03100", "title": "Flow-A11y: Flow-Aware Accessibility Testing", "url": "https://arxiv.org/abs/2607.03100", "published": "2026-07-03", "updated": "2026-07-03", "authors": [ "Nasr Eddine Fliti", "Leisan Kokorina", "Florian Tambon", "Michael Papadakis" ], "categories": [ "cs.SE", "cs.AI" ], "topics": [ "agent-evaluation", "computer-use", "tool-use" ], "score": 7, "relevance": "medium", "matched_queries": [ "web-gui-agent" ], "arxiv_id": "2607.03100", "source": "arxiv", "source_id": "arxiv:2607.03100", "pdf_url": "https://arxiv.org/pdf/2607.03100", "primary_query": "web-gui-agent" }, { "id": "2607.00613", "title": "Ai2-Kit: Streamlining AI-Accelerated Ab Initio Workflows for Complex Chemical Systems", "url": "https://arxiv.org/abs/2607.00613", "published": "2026-07-01", "updated": "2026-07-01", "authors": [ "Sheng Bi", "Wei-Hong Xu", "Yong-Bin Zhuang", "Jia-Xin Zhu", "Jiang-Peng Qiu", "Yu-Hang Tang", "Xiang-Long Du", "Qi You", "Yun-Pei Liu", "Fu-Qiang Gong", "Yu-Xin Guo", "Yi-Ze Wang", "Cheng-Xuan Wang", "Zi-Heng Gong", "Zi-Qiang Chen", "Chang Liu", "Si-Yuan Han", "Jian Gu", "Jia-Xin Li", "Yi-Ming Chen", "Lin Huang", "Si-Jie Chen", "Bo-Ying Huang", "Jie-Zhen Xia", "Fan-Jie Xu", "Su-Yang Zhong", "Peng-Wei Xu", "Jun-Yi Wang", "Xing-Yun Xie", "Yu-Lei Gong", "Yan-Yi Su", "Yue Liu", "Rui-Hao Bi", "Lang Li", "Fei-Teng Wang", "Jing-Xiang Zou", "Mei Jia", "Jie-Qiong Li", "Min Lin", "Qi-Yuan Fan", "Juan-Juan Sun", "Jia-Bo Le", "Zixuan Wei", "Jin-Yuan Hu", "Meng-Lei Jia", "Yan Sun", "Xiao-Hui Yang", "Fujie Tang", "Feng Wang", "Jun Cheng" ], "categories": [ "physics.chem-ph" ], "topics": [ "rag", "tool-use", "workflow-agent", "world-model" ], "score": 7, "relevance": "medium", "matched_queries": [ "ai-agent" ], "arxiv_id": "2607.00613", "source": "arxiv", "source_id": "arxiv:2607.00613", "pdf_url": "https://arxiv.org/pdf/2607.00613", "primary_query": "ai-agent" }, { "id": "2606.31399", "title": "World-Model Collapse as a Phase Transition", "url": "https://arxiv.org/abs/2606.31399", "published": "2026-06-30", "updated": "2026-07-04", "authors": [ "Xinyuan Song", "Zekun Cai" ], "categories": [ "cs.AI" ], "topics": [ "planning", "tool-use", "world-model" ], "score": 7, "relevance": "medium", "matched_queries": [ "language-agent" ], "arxiv_id": "2606.31399", "source": "arxiv", "source_id": "arxiv:2606.31399", "pdf_url": "https://arxiv.org/pdf/2606.31399", "primary_query": "language-agent" }, { "id": "2607.00155", "title": "A Contextual-Bandit Oversight Game with Two-Sided Informational Asymmetry", "url": "https://arxiv.org/abs/2607.00155", "published": "2026-06-30", "updated": "2026-06-30", "authors": [ "Yunjin Tong" ], "categories": [ "cs.AI", "cs.GT" ], "topics": [ "agent-evaluation", "embodied-agent", "tool-use" ], "score": 7, "relevance": "medium", "matched_queries": [ "ai-agent" ], "arxiv_id": "2607.00155", "source": "arxiv", "source_id": "arxiv:2607.00155", "pdf_url": "https://arxiv.org/pdf/2607.00155", "primary_query": "ai-agent" }, { "id": "2606.31214", "title": "EasyScan_HEP 2: Agent-Ready Parameter Scans for High-Energy Physics", "url": "https://arxiv.org/abs/2606.31214", "published": "2026-06-30", "updated": "2026-06-30", "authors": [ "Yang Xiao", "Yuanfang Yue", "Yang Zhang" ], "categories": [ "hep-ph" ], "topics": [ "workflow-agent" ], "score": 7, "relevance": "medium", "matched_queries": [ "ai-agent" ], "arxiv_id": "2606.31214", "source": "arxiv", "source_id": "arxiv:2606.31214", "pdf_url": "https://arxiv.org/pdf/2606.31214", "primary_query": "ai-agent" }, { "id": "2606.29389", "title": "Exploring the Cryptographic Limits of Transformer Networks", "url": "https://arxiv.org/abs/2606.29389", "published": "2026-06-28", "updated": "2026-06-28", "authors": [ "Stefan Domunco", "Andis Draguns", "Philip Torr", "Isaac Robinson", "Christian Schroeder de Witt" ], "categories": [ "cs.CR", "cs.LG" ], "topics": [ "agent-evaluation", "agent-safety" ], "score": 7, "relevance": "medium", "matched_queries": [ "ai-agent" ], "arxiv_id": "2606.29389", "source": "arxiv", "source_id": "arxiv:2606.29389", "pdf_url": "https://arxiv.org/pdf/2606.29389", "primary_query": "ai-agent" }, { "id": "2606.29100", "title": "Toward Exascale AI for Science: A Scalable AI Skill for Autonomous Microkinetics Discovery", "url": "https://arxiv.org/abs/2606.29100", "published": "2026-06-27", "updated": "2026-07-03", "authors": [ "Ken-ichi Nomura", "William Dawson", "Nabankur Dasgupta", "Taufeq Mohammed Razakh", "Thomas Linker", "Kai Ito", "Aiichiro Nakano" ], "categories": [ "cs.CE" ], "topics": [ "agent-evaluation", "workflow-agent", "world-model" ], "score": 7, "relevance": "medium", "matched_queries": [ "agentic-ai" ], "arxiv_id": "2606.29100", "source": "arxiv", "source_id": "arxiv:2606.29100", "pdf_url": "https://arxiv.org/pdf/2606.29100", "primary_query": "agentic-ai" }, { "id": "2606.26448", "title": "Closing the Loop to Discover Psychological Theories with an Automated Cognitive Scientist", "url": "https://arxiv.org/abs/2606.26448", "published": "2026-06-24", "updated": "2026-06-24", "authors": [ "Akshay K. Jagadish", "Younes Strittmatter", "Nori Jacoby", "George Kachergis", "Eric Schulz", "Nathaniel Daw", "Suyog H. Chandramouli", "Thomas L. Griffiths" ], "categories": [ "q-bio.NC", "cs.AI" ], "topics": [ "agent" ], "score": 7, "relevance": "medium", "matched_queries": [ "agentic-ai", "autonomous-agent-llm" ], "arxiv_id": "2606.26448", "source": "arxiv", "source_id": "arxiv:2606.26448", "pdf_url": "https://arxiv.org/pdf/2606.26448", "primary_query": "agentic-ai" }, { "id": "2606.25525", "title": "The impact of artificial intelligence on enterprise software user roles", "url": "https://arxiv.org/abs/2606.25525", "published": "2026-06-24", "updated": "2026-06-24", "authors": [ "Isabel Unger", "Elizangela Valarini", "Martin Schrepp", "Nina Hollender", "Gabriela Rocha", "Erik Bertram" ], "categories": [ "cs.SE", "cs.AI" ], "topics": [ "multi-agent", "tool-use", "workflow-agent" ], "score": 7, "relevance": "medium", "matched_queries": [ "agentic-ai" ], "arxiv_id": "2606.25525", "source": "arxiv", "source_id": "arxiv:2606.25525", "pdf_url": "https://arxiv.org/pdf/2606.25525", "primary_query": "agentic-ai" }, { "id": "2606.25496", "title": "Recommendation as Generation: Unifying Personalized Video Generation and Recommendation at Industrial Scale", "url": "https://arxiv.org/abs/2606.25496", "published": "2026-06-24", "updated": "2026-06-24", "authors": [ "Yanhua Cheng", "Bo Wang", "Haotian Zhang", "Xinyuan Gao", "Zhihui Yin", "Ben Xue", "Yongzhi Li", "Jieting Xue", "Ye Ma", "Minquan Wang", "Jiahui Li", "Tianyu Xu", "Zhiqiang Liu", "Xiao Lin", "Shiyang Wen", "Changcheng Li", "Liu Liu", "Quan Chen", "Peng Jiang", "Kun Gai" ], "categories": [ "cs.IR" ], "topics": [ "agent-evaluation", "agent-safety", "planning", "rag" ], "score": 7, "relevance": "medium", "matched_queries": [ "rag-agent" ], "arxiv_id": "2606.25496", "source": "arxiv", "source_id": "arxiv:2606.25496", "pdf_url": "https://arxiv.org/pdf/2606.25496", "primary_query": "rag-agent" }, { "id": "2606.23348", "title": "Superhuman AI for Generals.io Using Self-Play Reinforcement Learning", "url": "https://arxiv.org/abs/2606.23348", "published": "2026-06-22", "updated": "2026-06-22", "authors": [ "Matej Straka", "Viliam Lisý", "Martin Schmid" ], "categories": [ "cs.LG" ], "topics": [ "agent-evaluation", "planning", "rag" ], "score": 7, "relevance": "medium", "matched_queries": [ "ai-agent" ], "arxiv_id": "2606.23348", "source": "arxiv", "source_id": "arxiv:2606.23348", "pdf_url": "https://arxiv.org/pdf/2606.23348", "primary_query": "ai-agent" }, { "id": "2606.21124", "title": "PulseCX: Breaking the Closed-World Assumption in Real-Time CX", "url": "https://arxiv.org/abs/2606.21124", "published": "2026-06-19", "updated": "2026-06-19", "authors": [ "Rajat Agarwal", "Suvidha Tripathi", "Shubham Sharma" ], "categories": [ "cs.AI", "cs.IR" ], "topics": [ "memory", "tool-use" ], "score": 7, "relevance": "medium", "matched_queries": [ "ai-agent" ], "arxiv_id": "2606.21124", "source": "arxiv", "source_id": "arxiv:2606.21124", "pdf_url": "https://arxiv.org/pdf/2606.21124", "primary_query": "ai-agent" }, { "id": "2606.16319", "title": "Architectural Wisdom: A Framework for Governing Optimization in AI Systems", "url": "https://arxiv.org/abs/2606.16319", "published": "2026-06-15", "updated": "2026-06-15", "authors": [ "Edward Y. Chang" ], "categories": [ "cs.AI" ], "topics": [ "rag", "tool-use" ], "score": 7, "relevance": "medium", "matched_queries": [ "tool-use" ], "arxiv_id": "2606.16319", "source": "arxiv", "source_id": "arxiv:2606.16319", "pdf_url": "https://arxiv.org/pdf/2606.16319", "primary_query": "tool-use" }, { "id": "2605.18991", "title": "Agent Security is a Systems Problem", "url": "https://arxiv.org/abs/2605.18991", "published": "2026-05-18", "updated": "2026-05-20", "authors": [ "Mihai Christodorescu", "Earlence Fernandes", "Ashish Hooda", "Somesh Jha", "Johann Rehberger", "Kamalika Chaudhuri", "Xiaohan Fu", "Khawaja Shams", "Guy Amir", "Jihye Choi", "Sarthak Choudhary", "Nils Palumbo", "Andrey Labunets", "Nishit V. Pandya" ], "categories": [ "cs.CR", "cs.AI" ], "topics": [ "agent-safety" ], "score": 7, "relevance": "medium", "matched_queries": [ "agent-safety" ], "arxiv_id": "2605.18991", "source": "arxiv", "source_id": "arxiv:2605.18991", "pdf_url": "https://arxiv.org/pdf/2605.18991", "primary_query": "agent-safety" }, { "id": "2605.02028", "title": "Language models fail at extended rule following", "url": "https://arxiv.org/abs/2605.02028", "published": "2026-05-03", "updated": "2026-05-16", "authors": [ "Tianxiang Dai", "Jonathan Fan" ], "categories": [ "cs.CL" ], "topics": [ "tool-use" ], "score": 7, "relevance": "medium", "matched_queries": [ "autonomous-agent-llm" ], "arxiv_id": "2605.02028", "source": "arxiv", "source_id": "arxiv:2605.02028", "pdf_url": "https://arxiv.org/pdf/2605.02028", "primary_query": "autonomous-agent-llm" }, { "id": "2604.25610", "title": "Optimizing ground state preparation protocols with autoresearch", "url": "https://arxiv.org/abs/2604.25610", "published": "2026-04-28", "updated": "2026-05-08", "authors": [ "Luis Mantilla Calderón", "Jérôme F. Gonthier", "Ignacio Gustin", "Varinia Bernales", "Alán Aspuru-Guzik" ], "categories": [ "quant-ph" ], "topics": [ "coding-agent", "computer-use", "world-model" ], "score": 7, "relevance": "medium", "matched_queries": [ "language-agent" ], "arxiv_id": "2604.25610", "source": "arxiv", "source_id": "arxiv:2604.25610", "pdf_url": "https://arxiv.org/pdf/2604.25610", "primary_query": "language-agent" }, { "id": "2603.25636", "title": "Designing Any Imaging System from Natural Language: Agent-Constrained Composition over a Finite Primitive Basis", "url": "https://arxiv.org/abs/2603.25636", "published": "2026-03-26", "updated": "2026-03-26", "authors": [ "Chengshuai Yang" ], "categories": [ "cs.CV" ], "topics": [ "planning", "tool-use" ], "score": 7, "relevance": "medium", "matched_queries": [ "language-agent" ], "arxiv_id": "2603.25636", "source": "arxiv", "source_id": "arxiv:2603.25636", "pdf_url": "https://arxiv.org/pdf/2603.25636", "primary_query": "language-agent" }, { "id": "2607.05835", "title": "Tangent classes of matroids and wonderful compactifications", "url": "https://arxiv.org/abs/2607.05835", "published": "2026-07-07", "updated": "2026-07-07", "authors": [ "Ronnie Cheng", "Shurui Liu", "Guoxiong Gao" ], "categories": [ "math.AG", "cs.AI", "math.CO" ], "topics": [ "computer-use", "reasoning" ], "score": 6, "relevance": "low", "matched_queries": [ "ai-agent" ], "arxiv_id": "2607.05835", "source": "arxiv", "source_id": "arxiv:2607.05835", "pdf_url": "https://arxiv.org/pdf/2607.05835", "primary_query": "ai-agent" }, { "id": "2607.05744", "title": "Unicode TAG-Block Concealment of Tool-Metadata Payloads in the Model Context Protocol: An Approval-View Fidelity Gap Across Three Independent Server Implementations", "url": "https://arxiv.org/abs/2607.05744", "published": "2026-07-07", "updated": "2026-07-07", "authors": [ "Mohammadreza Rashidi" ], "categories": [ "cs.CR", "cs.AI", "cs.SE" ], "topics": [ "coding-agent", "tool-use" ], "score": 6, "relevance": "low", "matched_queries": [ "coding-agent" ], "arxiv_id": "2607.05744", "source": "arxiv", "source_id": "arxiv:2607.05744", "pdf_url": "https://arxiv.org/pdf/2607.05744", "primary_query": "coding-agent" }, { "id": "2607.02905", "title": "Pre-Strings Lectures on Artificial Intelligence", "url": "https://arxiv.org/abs/2607.02905", "published": "2026-07-03", "updated": "2026-07-03", "authors": [ "James Halverson" ], "categories": [ "hep-th" ], "topics": [ "agent-evaluation", "workflow-agent" ], "score": 6, "relevance": "low", "matched_queries": [ "agentic-ai" ], "arxiv_id": "2607.02905", "source": "arxiv", "source_id": "arxiv:2607.02905", "pdf_url": "https://arxiv.org/pdf/2607.02905", "primary_query": "agentic-ai" }, { "id": "2607.02609", "title": "Knowledge-Centric Information Systems", "url": "https://arxiv.org/abs/2607.02609", "published": "2026-07-01", "updated": "2026-07-01", "authors": [ "Mariano Garralda-Barrio" ], "categories": [ "cs.SE", "cs.AI", "cs.DB" ], "topics": [ "workflow-agent" ], "score": 6, "relevance": "low", "matched_queries": [ "agentic-ai" ], "arxiv_id": "2607.02609", "source": "arxiv", "source_id": "arxiv:2607.02609", "pdf_url": "https://arxiv.org/pdf/2607.02609", "primary_query": "agentic-ai" }, { "id": "2607.01415", "title": "The Rollout Infrastructure Tax in Coding-Agent Reinforcement Learning", "url": "https://arxiv.org/abs/2607.01415", "published": "2026-07-01", "updated": "2026-07-01", "authors": [ "Daniel Thi Graviet", "Lovre Pesut", "Ivan Dagelic", "Vedran Jukic", "Ivan Burazin" ], "categories": [ "cs.LG", "cs.DC" ], "topics": [ "coding-agent" ], "score": 6, "relevance": "low", "matched_queries": [ "coding-agent" ], "arxiv_id": "2607.01415", "source": "arxiv", "source_id": "arxiv:2607.01415", "pdf_url": "https://arxiv.org/pdf/2607.01415", "primary_query": "coding-agent" }, { "id": "2607.01380", "title": "Lagrangian evaluation of polymeric stress in viscoelastic fluids", "url": "https://arxiv.org/abs/2607.01380", "published": "2026-07-01", "updated": "2026-07-01", "authors": [ "Mohammad Majidi", "Rishu Gandhi", "Louison Thorens", "Maliheh Teimouri", "Jeffrey S. Guasto", "Arezoo M. Ardekani" ], "categories": [ "physics.flu-dyn" ], "topics": [ "agent-evaluation", "world-model" ], "score": 6, "relevance": "low", "matched_queries": [ "web-gui-agent" ], "arxiv_id": "2607.01380", "source": "arxiv", "source_id": "arxiv:2607.01380", "pdf_url": "https://arxiv.org/pdf/2607.01380", "primary_query": "web-gui-agent" }, { "id": "2606.32014", "title": "Scalable Behaviour Cloning on Browser Using via Skill Distillation", "url": "https://arxiv.org/abs/2606.32014", "published": "2026-06-30", "updated": "2026-06-30", "authors": [ "Kaisen Yang", "Zheng Jiang", "Yuzhao Peng", "Houde Qian", "Boshi Zhang", "Youjie Zheng", "Shijin Hong", "Qingle Liu", "Ruoyu Han", "Bohan Lyu", "Bingxiang He", "Eren Cai", "Calvin Xiao", "Qinhuai Na" ], "categories": [ "cs.CL" ], "topics": [ "computer-use", "tool-use", "workflow-agent" ], "score": 6, "relevance": "low", "matched_queries": [ "web-gui-agent" ], "arxiv_id": "2606.32014", "source": "arxiv", "source_id": "arxiv:2606.32014", "pdf_url": "https://arxiv.org/pdf/2606.32014", "primary_query": "web-gui-agent" }, { "id": "2606.29532", "title": "SemJoin: Semantic Join Optimization", "url": "https://arxiv.org/abs/2606.29532", "published": "2026-06-28", "updated": "2026-06-28", "authors": [ "Christopher Gou", "Aditya Banerjee", "Jiaxuan Wang", "Chunwei Liu" ], "categories": [ "cs.DB", "cs.AI" ], "topics": [ "agent-evaluation" ], "score": 6, "relevance": "low", "matched_queries": [ "llm-agent" ], "arxiv_id": "2606.29532", "source": "arxiv", "source_id": "arxiv:2606.29532", "pdf_url": "https://arxiv.org/pdf/2606.29532", "primary_query": "llm-agent" }, { "id": "2606.27951", "title": "AI Persuasive Framing in Collective Dilemmas", "url": "https://arxiv.org/abs/2606.27951", "published": "2026-06-26", "updated": "2026-06-26", "authors": [ "Anders Giovanni Møller", "Alessia Galdeman", "Arianna Pera", "Luca Maria Aiello" ], "categories": [ "cs.CY", "cs.CL", "cs.HC", "physics.soc-ph" ], "topics": [ "agent-safety", "tool-use" ], "score": 6, "relevance": "low", "matched_queries": [ "ai-agent" ], "arxiv_id": "2606.27951", "source": "arxiv", "source_id": "arxiv:2606.27951", "pdf_url": "https://arxiv.org/pdf/2606.27951", "primary_query": "ai-agent" }, { "id": "2606.27291", "title": "Designing Reward Signals for Portable Query Generation: A Case Study in Industrial Semantic Job Search", "url": "https://arxiv.org/abs/2606.27291", "published": "2026-06-25", "updated": "2026-06-25", "authors": [ "Ping Liu", "Qianqi Shen", "Jianqiang Shen", "Wenqiong Liu", "Rajat Arora", "Yunxiang Ren", "Chunnan Yao", "Dan Xu", "Baofen Zheng", "Wanjun Jiang", "Andrii Soviak", "Kevin Kao", "Jingwei Wu", "Wenjing Zhang" ], "categories": [ "cs.LG" ], "topics": [ "agent-evaluation", "tool-use" ], "score": 6, "relevance": "low", "matched_queries": [ "ai-agent" ], "arxiv_id": "2606.27291", "source": "arxiv", "source_id": "arxiv:2606.27291", "pdf_url": "https://arxiv.org/pdf/2606.27291", "primary_query": "ai-agent" }, { "id": "2606.25244", "title": "Reading AI Model Compilation in MLIR Through the Lens of Formal Theories", "url": "https://arxiv.org/abs/2606.25244", "published": "2026-06-24", "updated": "2026-06-24", "authors": [ "Javed Absar" ], "categories": [ "cs.PL" ], "topics": [ "coding-agent", "tool-use" ], "score": 6, "relevance": "low", "matched_queries": [ "coding-agent" ], "arxiv_id": "2606.25244", "source": "arxiv", "source_id": "arxiv:2606.25244", "pdf_url": "https://arxiv.org/pdf/2606.25244", "primary_query": "coding-agent" }, { "id": "2606.23768", "title": "Cryptographic certificates of validity for trustworthy AI", "url": "https://arxiv.org/abs/2606.23768", "published": "2026-06-22", "updated": "2026-06-22", "authors": [ "Murdoch J. Gabbay" ], "categories": [ "cs.CR", "cs.AI", "cs.LO" ], "topics": [ "reasoning", "tool-use" ], "score": 6, "relevance": "low", "matched_queries": [ "agentic-ai" ], "arxiv_id": "2606.23768", "source": "arxiv", "source_id": "arxiv:2606.23768", "pdf_url": "https://arxiv.org/pdf/2606.23768", "primary_query": "agentic-ai" }, { "id": "2606.22447", "title": "A Differentiable Atari VCS:A Complex, Fully Known Ground Truth for Explainable AI", "url": "https://arxiv.org/abs/2606.22447", "published": "2026-06-21", "updated": "2026-06-21", "authors": [ "Andreas Maier", "Siming Bayer", "Patrick Krauss" ], "categories": [ "cs.AI", "cs.LG" ], "topics": [ "coding-agent", "planning" ], "score": 6, "relevance": "low", "matched_queries": [ "coding-agent" ], "arxiv_id": "2606.22447", "source": "arxiv", "source_id": "arxiv:2606.22447", "pdf_url": "https://arxiv.org/pdf/2606.22447", "primary_query": "coding-agent" }, { "id": "2606.19924", "title": "The Tao of Agency: Autotelic AI, Embedded Agency and Dissolution of the Self", "url": "https://arxiv.org/abs/2606.19924", "published": "2026-06-18", "updated": "2026-06-18", "authors": [ "Aritra Sarkar" ], "categories": [ "cs.AI" ], "topics": [ "agent" ], "score": 6, "relevance": "low", "matched_queries": [ "ai-agent" ], "arxiv_id": "2606.19924", "source": "arxiv", "source_id": "arxiv:2606.19924", "pdf_url": "https://arxiv.org/pdf/2606.19924", "primary_query": "ai-agent" }, { "id": "2606.19047", "title": "RODS: Reward-Driven Online Data Synthesis for Multi-Turn Tool-Use Agents", "url": "https://arxiv.org/abs/2606.19047", "published": "2026-06-17", "updated": "2026-06-17", "authors": [ "Ruishan Fang", "Siyuan Lu", "Chenyi Zhuang", "Tao Lin" ], "categories": [ "cs.AI" ], "topics": [ "tool-use" ], "score": 6, "relevance": "low", "matched_queries": [ "tool-use" ], "arxiv_id": "2606.19047", "source": "arxiv", "source_id": "arxiv:2606.19047", "pdf_url": "https://arxiv.org/pdf/2606.19047", "primary_query": "tool-use" }, { "id": "2606.19458", "title": "MonaVec: A Training-Free Embedded Vector Search Kernel for Edge and Offline AI Systems", "url": "https://arxiv.org/abs/2606.19458", "published": "2026-06-17", "updated": "2026-06-17", "authors": [ "Oğuzhan Yenen" ], "categories": [ "cs.IR" ], "topics": [ "agent-evaluation", "rag" ], "score": 6, "relevance": "low", "matched_queries": [ "function-calling", "rag-agent" ], "arxiv_id": "2606.19458", "source": "arxiv", "source_id": "arxiv:2606.19458", "pdf_url": "https://arxiv.org/pdf/2606.19458", "primary_query": "function-calling" }, { "id": "2606.17321", "title": "ProCUA-SFT Technical Report", "url": "https://arxiv.org/abs/2606.17321", "published": "2026-06-15", "updated": "2026-06-15", "authors": [ "Jaehun Jung", "Ximing Lu", "Brandon Cui", "Muhammad Khalifa", "Shaokun Zhang", "Hao Zhang", "Jin Xu", "Amala Sanjay Deshmukh", "Karan Sapra", "Andrew Tao", "Yejin Choi", "Jan Kautz", "Mingjie Liu", "Yi Dong" ], "categories": [ "cs.LG", "cs.CV" ], "topics": [ "computer-use", "planning", "tool-use" ], "score": 6, "relevance": "low", "matched_queries": [ "web-gui-agent" ], "arxiv_id": "2606.17321", "source": "arxiv", "source_id": "arxiv:2606.17321", "pdf_url": "https://arxiv.org/pdf/2606.17321", "primary_query": "web-gui-agent" }, { "id": "2606.15071", "title": "Quantum learning with a single-atom sensor", "url": "https://arxiv.org/abs/2606.15071", "published": "2026-06-13", "updated": "2026-06-13", "authors": [ "Yin Mo", "Emilio Bagan", "Giulio Chiribella" ], "categories": [ "quant-ph" ], "topics": [ "memory", "tool-use" ], "score": 6, "relevance": "low", "matched_queries": [ "agent-memory" ], "arxiv_id": "2606.15071", "source": "arxiv", "source_id": "arxiv:2606.15071", "pdf_url": "https://arxiv.org/pdf/2606.15071", "primary_query": "agent-memory" }, { "id": "2606.09416", "title": "Harness Engineering for Physical AI: Robot Middleware Is the Harness Layer", "url": "https://arxiv.org/abs/2606.09416", "published": "2026-06-08", "updated": "2026-06-08", "authors": [ "Sanghoon Lee", "Jiyeong Chae", "Kyung-Joon Park" ], "categories": [ "cs.RO", "cs.AI", "cs.SE" ], "topics": [ "embodied-agent", "planning", "tool-use" ], "score": 6, "relevance": "low", "matched_queries": [ "language-agent" ], "arxiv_id": "2606.09416", "source": "arxiv", "source_id": "arxiv:2606.09416", "pdf_url": "https://arxiv.org/pdf/2606.09416", "primary_query": "language-agent" }, { "id": "2606.02840", "title": "Self-Regulation through Communication in Evolved Neural Agents", "url": "https://arxiv.org/abs/2606.02840", "published": "2026-06-01", "updated": "2026-06-01", "authors": [ "Joshua Nunley" ], "categories": [ "q-bio.PE", "cs.MA", "cs.NE", "nlin.AO" ], "topics": [ "agent-safety" ], "score": 6, "relevance": "low", "matched_queries": [ "agent-safety" ], "arxiv_id": "2606.02840", "source": "arxiv", "source_id": "arxiv:2606.02840", "pdf_url": "https://arxiv.org/pdf/2606.02840", "primary_query": "agent-safety" }, { "id": "2606.02528", "title": "Auditing Asset-Specific Preferences in Financial Large Language Models: Evidence from Bitcoin Representations and Portfolio Allocation", "url": "https://arxiv.org/abs/2606.02528", "published": "2026-06-01", "updated": "2026-06-01", "authors": [ "Wenbin Wu" ], "categories": [ "q-fin.GN", "cs.CY", "cs.LG" ], "topics": [ "rag" ], "score": 6, "relevance": "low", "matched_queries": [ "autonomous-agent-llm" ], "arxiv_id": "2606.02528", "source": "arxiv", "source_id": "arxiv:2606.02528", "pdf_url": "https://arxiv.org/pdf/2606.02528", "primary_query": "autonomous-agent-llm" }, { "id": "2605.08378", "title": "Reinforcement Learning for Scalable and Trustworthy Intelligent Systems", "url": "https://arxiv.org/abs/2605.08378", "published": "2026-05-08", "updated": "2026-05-08", "authors": [ "Guangchen Lan" ], "categories": [ "cs.LG", "cs.AI", "cs.CL" ], "topics": [ "agent-safety" ], "score": 6, "relevance": "low", "matched_queries": [ "autonomous-agent-llm" ], "arxiv_id": "2605.08378", "source": "arxiv", "source_id": "arxiv:2605.08378", "pdf_url": "https://arxiv.org/pdf/2605.08378", "primary_query": "autonomous-agent-llm" }, { "id": "2605.05795", "title": "Reward Shaping and Action Masking for Compositional Tasks using Behavior Trees and LLMs", "url": "https://arxiv.org/abs/2605.05795", "published": "2026-05-07", "updated": "2026-05-23", "authors": [ "Nicholas Potteiger", "Ankita Samaddar", "Taylor T. Johnson", "Xenofon Koutsoukos" ], "categories": [ "cs.LG" ], "topics": [ "tool-use" ], "score": 6, "relevance": "low", "matched_queries": [ "autonomous-agent-llm" ], "arxiv_id": "2605.05795", "source": "arxiv", "source_id": "arxiv:2605.05795", "pdf_url": "https://arxiv.org/pdf/2605.05795", "primary_query": "autonomous-agent-llm" }, { "id": "2509.25378", "title": "Detecting and Fixing API Misuses of Data Science Libraries Using Large Language Models", "url": "https://arxiv.org/abs/2509.25378", "published": "2025-09-29", "updated": "2025-09-29", "authors": [ "Akalanka Galappaththi", "Francisco Ribeiro", "Sarah Nadi" ], "categories": [ "cs.SE" ], "topics": [ "tool-use" ], "score": 6, "relevance": "low", "matched_queries": [ "function-calling" ], "arxiv_id": "2509.25378", "source": "arxiv", "source_id": "arxiv:2509.25378", "pdf_url": "https://arxiv.org/pdf/2509.25378", "primary_query": "function-calling" }, { "id": "2509.06736", "title": "VehicleWorld: A Highly Integrated Multi-Device Environment for Intelligent Vehicle Interaction", "url": "https://arxiv.org/abs/2509.06736", "published": "2025-09-08", "updated": "2025-09-08", "authors": [ "Jie Yang", "Jiajun Chen", "Zhangyue Yin", "Shuo Chen", "Yuxin Wang", "Yiran Guo", "Yuan Li", "Yining Zheng", "Xuanjing Huang", "Xipeng Qiu" ], "categories": [ "cs.AI", "cs.CL", "cs.RO" ], "topics": [ "agent-evaluation", "tool-use" ], "score": 6, "relevance": "low", "matched_queries": [ "function-calling" ], "arxiv_id": "2509.06736", "source": "arxiv", "source_id": "arxiv:2509.06736", "pdf_url": "https://arxiv.org/pdf/2509.06736", "primary_query": "function-calling" }, { "id": "2607.01188", "title": "Optimal Resource Utilization for Autonomous Laboratory Orchestrators", "url": "https://arxiv.org/abs/2607.01188", "published": "2026-07-01", "updated": "2026-07-01", "authors": [ "Austin McDannald", "Julia Tisaranni", "Howie Joress" ], "categories": [ "cs.AI", "cond-mat.mtrl-sci" ], "topics": [ "planning" ], "score": 5, "relevance": "low", "matched_queries": [ "ai-agent" ], "arxiv_id": "2607.01188", "source": "arxiv", "source_id": "arxiv:2607.01188", "pdf_url": "https://arxiv.org/pdf/2607.01188", "primary_query": "ai-agent" }, { "id": "2606.31132", "title": "ELASTIC: Efficiently Learning to Adaptively Scale Test-Time Compute for Generative Control Policies", "url": "https://arxiv.org/abs/2606.31132", "published": "2026-06-30", "updated": "2026-06-30", "authors": [ "Andrew Zou Li", "Gokul Swamy", "Yonatan Bisk", "Andrea Bajcsy" ], "categories": [ "cs.RO" ], "topics": [ "agent-evaluation", "embodied-agent", "tool-use" ], "score": 5, "relevance": "low", "matched_queries": [ "web-gui-agent" ], "arxiv_id": "2606.31132", "source": "arxiv", "source_id": "arxiv:2606.31132", "pdf_url": "https://arxiv.org/pdf/2606.31132", "primary_query": "web-gui-agent" }, { "id": "2606.30543", "title": "TRACE: Temporal Relationship-Aware Conversational Entrainment Detection in Dyadic Speech", "url": "https://arxiv.org/abs/2606.30543", "published": "2026-06-29", "updated": "2026-07-03", "authors": [ "Sathvik Manikantan Napa Ugandhar", "Hao Zhang", "Alison Gunzler", "Yuzhe Wang", "Thomas Thebaud", "Georgi Tinchev", "Venkatesh Ravichandran", "Laureano Moro-Velázquez" ], "categories": [ "cs.CL", "cs.AI" ], "topics": [ "tool-use" ], "score": 5, "relevance": "low", "matched_queries": [ "ai-agent" ], "arxiv_id": "2606.30543", "source": "arxiv", "source_id": "arxiv:2606.30543", "pdf_url": "https://arxiv.org/pdf/2606.30543", "primary_query": "ai-agent" }, { "id": "2606.27045", "title": "The Spec Growth Engine: Spec-Anchored, Code-Coupled, Drift-Enforced Architecture for AI-Assisted Software Development", "url": "https://arxiv.org/abs/2606.27045", "published": "2026-06-25", "updated": "2026-06-25", "authors": [ "Hartwig Grabowski" ], "categories": [ "cs.SE", "cs.AI" ], "topics": [ "coding-agent" ], "score": 5, "relevance": "low", "matched_queries": [ "coding-agent" ], "arxiv_id": "2606.27045", "source": "arxiv", "source_id": "arxiv:2606.27045", "pdf_url": "https://arxiv.org/pdf/2606.27045", "primary_query": "coding-agent" }, { "id": "2606.27365", "title": "3D Imaging of Complex Skyrmion and Hopf Topologies in an Extended Sample", "url": "https://arxiv.org/abs/2606.27365", "published": "2026-06-25", "updated": "2026-06-25", "authors": [ "I. Binnie", "H. Fang", "B. Shearer", "A. Grafov", "N. Jenkins", "Y. Shao", "C. O'Leary", "Y. Liao", "T. Feggeler", "A. Oh", "S. Yazdi", "J. Zou", "B. Wang", "E-E. Cating", "S. A. Montoya", "D. Shapiro", "J. Miao", "H. C. Kapteyn", "M. M. Murnane" ], "categories": [ "cond-mat.mtrl-sci", "cond-mat.mes-hall", "physics.app-ph" ], "topics": [ "memory", "rag", "tool-use" ], "score": 5, "relevance": "low", "matched_queries": [ "web-gui-agent" ], "arxiv_id": "2606.27365", "source": "arxiv", "source_id": "arxiv:2606.27365", "pdf_url": "https://arxiv.org/pdf/2606.27365", "primary_query": "web-gui-agent" }, { "id": "2606.25337", "title": "AI Coaching for Accelerating Human Skill Development with Reinforcement Learning", "url": "https://arxiv.org/abs/2606.25337", "published": "2026-06-24", "updated": "2026-06-24", "authors": [ "Wei Wang", "Enlin Gu", "Antonio Loquercio", "Haimin Hu", "Rahul Mangharam" ], "categories": [ "cs.RO", "cs.AI", "cs.HC" ], "topics": [ "embodied-agent" ], "score": 5, "relevance": "low", "matched_queries": [ "ai-agent" ], "arxiv_id": "2606.25337", "source": "arxiv", "source_id": "arxiv:2606.25337", "pdf_url": "https://arxiv.org/pdf/2606.25337", "primary_query": "ai-agent" }, { "id": "2606.23315", "title": "Test-Driven, AI-Assisted Learning: Replacing Lectures with Weekly Closed-Book Tests", "url": "https://arxiv.org/abs/2606.23315", "published": "2026-06-22", "updated": "2026-06-22", "authors": [ "Jin-Guo Liu", "Shang-Qi Lu", "Xin-Ran Shi", "Long-Li Zheng", "Wei Wang" ], "categories": [ "cs.CY" ], "topics": [ "workflow-agent" ], "score": 5, "relevance": "low", "matched_queries": [ "ai-agent" ], "arxiv_id": "2606.23315", "source": "arxiv", "source_id": "arxiv:2606.23315", "pdf_url": "https://arxiv.org/pdf/2606.23315", "primary_query": "ai-agent" }, { "id": "2606.22568", "title": "SeFi-Image: A Text-to-Image Foundation Model with Semantic-First Diffusion", "url": "https://arxiv.org/abs/2606.22568", "published": "2026-06-21", "updated": "2026-06-26", "authors": [ "SeFi-Team" ], "categories": [ "cs.CV" ], "topics": [ "agent-evaluation", "computer-use", "rag" ], "score": 5, "relevance": "low", "matched_queries": [ "web-gui-agent" ], "arxiv_id": "2606.22568", "source": "arxiv", "source_id": "arxiv:2606.22568", "pdf_url": "https://arxiv.org/pdf/2606.22568", "primary_query": "web-gui-agent" }, { "id": "2606.22226", "title": "Quantifying Theoretical AI Alignment Guarantees: Receiver-Utility Bounds in Bayesian Persuasion", "url": "https://arxiv.org/abs/2606.22226", "published": "2026-06-20", "updated": "2026-06-20", "authors": [ "Eric Yachbes", "Eva Tardos" ], "categories": [ "cs.GT", "cs.AI", "cs.IT" ], "topics": [ "agent-safety" ], "score": 5, "relevance": "low", "matched_queries": [ "ai-agent" ], "arxiv_id": "2606.22226", "source": "arxiv", "source_id": "arxiv:2606.22226", "pdf_url": "https://arxiv.org/pdf/2606.22226", "primary_query": "ai-agent" }, { "id": "2606.19683", "title": "Exit-and-Join Dynamics for Decentralized Coalition Formation", "url": "https://arxiv.org/abs/2606.19683", "published": "2026-06-18", "updated": "2026-06-18", "authors": [ "Quanyan Zhu" ], "categories": [ "cs.AI", "cs.MA", "eess.SY" ], "topics": [ "agent-evaluation" ], "score": 5, "relevance": "low", "matched_queries": [ "agent-evaluation" ], "arxiv_id": "2606.19683", "source": "arxiv", "source_id": "arxiv:2606.19683", "pdf_url": "https://arxiv.org/pdf/2606.19683", "primary_query": "agent-evaluation" }, { "id": "2606.17790", "title": "Distributed Experimental Design: Bayes-optimal Fusion of Local Designs", "url": "https://arxiv.org/abs/2606.17790", "published": "2026-06-16", "updated": "2026-06-16", "authors": [ "Nagananda K G", "Lav R. Varshney", "Pramod K. Varshney" ], "categories": [ "stat.AP", "cs.IT" ], "topics": [ "agent-evaluation", "planning" ], "score": 5, "relevance": "low", "matched_queries": [ "agent-evaluation" ], "arxiv_id": "2606.17790", "source": "arxiv", "source_id": "arxiv:2606.17790", "pdf_url": "https://arxiv.org/pdf/2606.17790", "primary_query": "agent-evaluation" }, { "id": "2606.17388", "title": "Agent Utilities over Generalized Voronoi Regions and their Gradients", "url": "https://arxiv.org/abs/2606.17388", "published": "2026-06-16", "updated": "2026-06-16", "authors": [ "Andre N. Costa", "Petter Ögren", "Carlos H. C. Ribeiro" ], "categories": [ "cs.RO", "cs.CG", "eess.SY" ], "topics": [ "agent" ], "score": 5, "relevance": "low", "matched_queries": [ "web-gui-agent" ], "arxiv_id": "2606.17388", "source": "arxiv", "source_id": "arxiv:2606.17388", "pdf_url": "https://arxiv.org/pdf/2606.17388", "primary_query": "web-gui-agent" }, { "id": "2606.12235", "title": "BenDi: An Energy-Efficient Quasi-Stochastic Systolic Architecture for Edge Bioelectronics", "url": "https://arxiv.org/abs/2606.12235", "published": "2026-06-10", "updated": "2026-06-10", "authors": [ "Bochen Ye", "Yihan Pan", "Shady Agwa", "Themis Prodromakis" ], "categories": [ "cs.AR" ], "topics": [ "agent-evaluation", "memory", "rag" ], "score": 5, "relevance": "low", "matched_queries": [ "web-gui-agent" ], "arxiv_id": "2606.12235", "source": "arxiv", "source_id": "arxiv:2606.12235", "pdf_url": "https://arxiv.org/pdf/2606.12235", "primary_query": "web-gui-agent" }, { "id": "2606.04131", "title": "Network node immunization: improving Netshield algorithm through random rooted forests", "url": "https://arxiv.org/abs/2606.04131", "published": "2026-06-02", "updated": "2026-06-02", "authors": [ "Luca Avena", "Alexandre Gaudillière", "Irina Gurewitsch", "Adoré Randriamandroso", "Alessio Troiani" ], "categories": [ "cs.SI", "math.PR" ], "topics": [ "agent-evaluation" ], "score": 5, "relevance": "low", "matched_queries": [ "function-calling" ], "arxiv_id": "2606.04131", "source": "arxiv", "source_id": "arxiv:2606.04131", "pdf_url": "https://arxiv.org/pdf/2606.04131", "primary_query": "function-calling" }, { "id": "2605.21792", "title": "Residual Skill Optimization for Text-to-SQL Ensembles", "url": "https://arxiv.org/abs/2605.21792", "published": "2026-05-20", "updated": "2026-05-20", "authors": [ "Jiongli Zhu", "Haoquan Guan", "Parjanya Prajakta Prashant", "Nikki Lijing Kuang", "Seyedeh Baharan Khatami", "Canwen Xu", "Xiaodong Yu", "Yingyu Lin", "Zhewei Yao", "Yuxiong He", "Babak Salimi" ], "categories": [ "cs.CL", "cs.AI", "cs.DB", "cs.LG" ], "topics": [ "coding-agent" ], "score": 5, "relevance": "low", "matched_queries": [ "function-calling" ], "arxiv_id": "2605.21792", "source": "arxiv", "source_id": "arxiv:2605.21792", "pdf_url": "https://arxiv.org/pdf/2605.21792", "primary_query": "function-calling" }, { "id": "2602.23397", "title": "Lifecycle-Integrated Security for AI-Cloud Convergence in Cyber-Physical Infrastructure", "url": "https://arxiv.org/abs/2602.23397", "published": "2026-02-26", "updated": "2026-02-26", "authors": [ "S M Zia Ur Rashid", "Deepa Gurung", "Sonam Raj Gupta", "Suman Rath" ], "categories": [ "cs.CR", "eess.SY" ], "topics": [ "agent-safety" ], "score": 5, "relevance": "low", "matched_queries": [ "agent-safety" ], "arxiv_id": "2602.23397", "source": "arxiv", "source_id": "arxiv:2602.23397", "pdf_url": "https://arxiv.org/pdf/2602.23397", "primary_query": "agent-safety" }, { "id": "2607.05498", "title": "Non-spherical Cows: Introducing the Asphericity Parameter as a Measure of Accretion Geometry", "url": "https://arxiv.org/abs/2607.05498", "published": "2026-07-06", "updated": "2026-07-06", "authors": [ "Benjamin A. Seidel", "Rhea-Silvia Remus", "Lucas C. Kimmig", "Lucas M. Valenzuela", "Klaus Dolag" ], "categories": [ "astro-ph.GA", "astro-ph.CO" ], "topics": [ "agent-evaluation", "tool-use", "world-model" ], "score": 4, "relevance": "low", "matched_queries": [ "web-gui-agent" ], "arxiv_id": "2607.05498", "source": "arxiv", "source_id": "arxiv:2607.05498", "pdf_url": "https://arxiv.org/pdf/2607.05498", "primary_query": "web-gui-agent" }, { "id": "2607.04222", "title": "Unsupervised Features Mining via Activation Geometry", "url": "https://arxiv.org/abs/2607.04222", "published": "2026-07-05", "updated": "2026-07-05", "authors": [ "Amit LeVi", "Elad David", "Max Fomin" ], "categories": [ "cs.AI" ], "topics": [ "reasoning" ], "score": 4, "relevance": "low", "matched_queries": [ "agentic-ai" ], "arxiv_id": "2607.04222", "source": "arxiv", "source_id": "arxiv:2607.04222", "pdf_url": "https://arxiv.org/pdf/2607.04222", "primary_query": "agentic-ai" }, { "id": "2607.03955", "title": "Strategy-Proof Probabilistic Social Choice Correspondences under Conditional Expected Utility", "url": "https://arxiv.org/abs/2607.03955", "published": "2026-07-04", "updated": "2026-07-04", "authors": [ "Madhuparna Karmokar", "Ujjwal Kumar", "Soumyarup Sadhukhan" ], "categories": [ "econ.TH" ], "topics": [ "agent-evaluation" ], "score": 4, "relevance": "low", "matched_queries": [ "agent-evaluation" ], "arxiv_id": "2607.03955", "source": "arxiv", "source_id": "arxiv:2607.03955", "pdf_url": "https://arxiv.org/pdf/2607.03955", "primary_query": "agent-evaluation" }, { "id": "2607.03949", "title": "TESSERA v2: Scaling Pixel-wise Earth Foundation Models", "url": "https://arxiv.org/abs/2607.03949", "published": "2026-07-04", "updated": "2026-07-04", "authors": [ "Zhengpeng Feng", "Sadiq Jaffer", "Ira Shokar", "Jovana Knezevic", "Mark Elvers", "Clement Atzberger", "Robin Young", "Aneesh Naik", "Niall Robinson", "Andrew Blake", "David Coomes", "Anil Madhavapeddy", "Srinivasan Keshav" ], "categories": [ "cs.CV", "cs.LG" ], "topics": [ "agent-evaluation", "planning", "rag" ], "score": 4, "relevance": "low", "matched_queries": [ "web-gui-agent" ], "arxiv_id": "2607.03949", "source": "arxiv", "source_id": "arxiv:2607.03949", "pdf_url": "https://arxiv.org/pdf/2607.03949", "primary_query": "web-gui-agent" }, { "id": "2607.05439", "title": "Design-CP: Context Parallelism for Design of Protein Nanoparticles", "url": "https://arxiv.org/abs/2607.05439", "published": "2026-07-03", "updated": "2026-07-03", "authors": [ "Lorenzo Tarricone", "Helen E. Eisenach", "Aiko Muraishi", "Charlotte M. Deane" ], "categories": [ "cs.LG", "cs.DC", "q-bio.QM" ], "topics": [ "agent-evaluation", "memory" ], "score": 4, "relevance": "low", "matched_queries": [ "agentic-ai" ], "arxiv_id": "2607.05439", "source": "arxiv", "source_id": "arxiv:2607.05439", "pdf_url": "https://arxiv.org/pdf/2607.05439", "primary_query": "agentic-ai" }, { "id": "2606.28801", "title": "Cross-channel Specific Emitter Identification and Verification via Signal Envelope", "url": "https://arxiv.org/abs/2606.28801", "published": "2026-06-27", "updated": "2026-06-27", "authors": [ "Yuhao Chen", "Boxiang He", "Shilian Wang", "Jing Lei" ], "categories": [ "eess.SP" ], "topics": [ "agent-evaluation", "computer-use", "reasoning" ], "score": 4, "relevance": "low", "matched_queries": [ "web-gui-agent" ], "arxiv_id": "2606.28801", "source": "arxiv", "source_id": "arxiv:2606.28801", "pdf_url": "https://arxiv.org/pdf/2606.28801", "primary_query": "web-gui-agent" }, { "id": "2606.22997", "title": "A Greatest Common Divisor Criterion of Certain Binomial Coefficients", "url": "https://arxiv.org/abs/2606.22997", "published": "2026-06-22", "updated": "2026-06-22", "authors": [ "Dakai Guo", "Ruichen Qiu", "Yichuan Cao", "Ruyong Feng", "Xiao-Shan Gao" ], "categories": [ "math.NT", "cs.LO" ], "topics": [ "agent" ], "score": 4, "relevance": "low", "matched_queries": [ "ai-agent" ], "arxiv_id": "2606.22997", "source": "arxiv", "source_id": "arxiv:2606.22997", "pdf_url": "https://arxiv.org/pdf/2606.22997", "primary_query": "ai-agent" }, { "id": "2606.20208", "title": "Beyond Accuracy: Measuring Logical Compliance of Predictive Models", "url": "https://arxiv.org/abs/2606.20208", "published": "2026-06-18", "updated": "2026-06-18", "authors": [ "Guillaume Olivier Delplanque", "Pierre Genevès", "Nabil Layaïda", "Zephirin Faure" ], "categories": [ "cs.AI", "cs.DB", "cs.NE" ], "topics": [ "agent-evaluation" ], "score": 4, "relevance": "low", "matched_queries": [ "web-gui-agent" ], "arxiv_id": "2606.20208", "source": "arxiv", "source_id": "arxiv:2606.20208", "pdf_url": "https://arxiv.org/pdf/2606.20208", "primary_query": "web-gui-agent" }, { "id": "2606.23719", "title": "A Hybrid Quantum-Classical Approach for Melt Pool Prediction in Laser Powder Bed Fusion", "url": "https://arxiv.org/abs/2606.23719", "published": "2026-06-17", "updated": "2026-06-17", "authors": [ "Matthew M. Sato", "Kincho H. Law" ], "categories": [ "quant-ph", "cs.LG" ], "topics": [ "coding-agent", "rag", "tool-use" ], "score": 4, "relevance": "low", "matched_queries": [ "web-gui-agent" ], "arxiv_id": "2606.23719", "source": "arxiv", "source_id": "arxiv:2606.23719", "pdf_url": "https://arxiv.org/pdf/2606.23719", "primary_query": "web-gui-agent" }, { "id": "2606.17860", "title": "An Epistemic Analysis of Random Coordinated Attack", "url": "https://arxiv.org/abs/2606.17860", "published": "2026-06-16", "updated": "2026-06-16", "authors": [ "Sophia Knight", "David Lehnherr", "Sergio Rajsbaum" ], "categories": [ "cs.DC", "cs.LO" ], "topics": [ "computer-use", "reasoning", "tool-use" ], "score": 4, "relevance": "low", "matched_queries": [ "web-gui-agent" ], "arxiv_id": "2606.17860", "source": "arxiv", "source_id": "arxiv:2606.17860", "pdf_url": "https://arxiv.org/pdf/2606.17860", "primary_query": "web-gui-agent" }, { "id": "2606.13225", "title": "The QR Factorization for Banded-Plus-Semiseparable Matrices Is Computable in Linear Complexity", "url": "https://arxiv.org/abs/2606.13225", "published": "2026-06-11", "updated": "2026-06-12", "authors": [ "Tao Chen", "Sheehan Olver" ], "categories": [ "math.NA" ], "topics": [ "agent-evaluation", "rag", "reasoning" ], "score": 4, "relevance": "low", "matched_queries": [ "web-gui-agent" ], "arxiv_id": "2606.13225", "source": "arxiv", "source_id": "arxiv:2606.13225", "pdf_url": "https://arxiv.org/pdf/2606.13225", "primary_query": "web-gui-agent" }, { "id": "2606.07463", "title": "Amortized Neural Optimization for Pre-Layout Signal Integrity Design Space Exploration using Differentiable Surrogates", "url": "https://arxiv.org/abs/2606.07463", "published": "2026-06-05", "updated": "2026-06-05", "authors": [ "Julian Withöft", "Werner John", "Emre Ecik", "Ralf Brüning", "Jürgen Götze" ], "categories": [ "eess.SP", "cs.CE", "cs.LG" ], "topics": [ "agent-evaluation", "workflow-agent", "world-model" ], "score": 4, "relevance": "low", "matched_queries": [ "web-gui-agent" ], "arxiv_id": "2606.07463", "source": "arxiv", "source_id": "arxiv:2606.07463", "pdf_url": "https://arxiv.org/pdf/2606.07463", "primary_query": "web-gui-agent" }, { "id": "2606.31582", "title": "Generalized Laura-Andoyer equations and the enumeration of some symmetrical classes of Dziobek configurations", "url": "https://arxiv.org/abs/2606.31582", "published": "2026-06-30", "updated": "2026-06-30", "authors": [ "Thiago Dias", "Ya-Lun Tsai" ], "categories": [ "math.DS" ], "topics": [ "agent-evaluation", "tool-use" ], "score": 3, "relevance": "low", "matched_queries": [ "web-gui-agent" ], "arxiv_id": "2606.31582", "source": "arxiv", "source_id": "arxiv:2606.31582", "pdf_url": "https://arxiv.org/pdf/2606.31582", "primary_query": "web-gui-agent" }, { "id": "2606.30536", "title": "Evaluating the Fourier Approximation in Pulsar Timing Array Analysis", "url": "https://arxiv.org/abs/2606.30536", "published": "2026-06-29", "updated": "2026-06-29", "authors": [ "Yongqi Zhang", "Hayden Scholz", "Ken D. Olum", "Lucas Steinberger", "Gabriella Agazie", "Akash Anumarlapudi", "Anne M. Archibald", "Zaven Arzoumanian", "Paul T. Baker", "Paul R. Brook", "H. Thankful Cromartie", "Kathryn Crowter", "Megan E. DeCesar", "Paul B. Demorest", "Timothy Dolch", "Justin A. Ellis", "Elizabeth C. Ferrara", "William Fiore", "Emmanuel Fonseca", "Gabriel E. Freedman", "Nate Garver-Daniels", "Peter A. Gentile", "Joseph Glaser", "Deborah C. Good", "Jeffrey S. Hazboun", "Ross J. Jennings", "Megan L. Jones", "David L. Kaplan", "Matthew Kerr", "Michael T. Lam", "Duncan R. Lorimer", "Jing Luo", "Ryan S. Lynch", "Alexander McEwen", "Maura A. McLaughlin", "Natasha McMann", "Bradley W. Meyers", "Cherry Ng", "David J. Nice", "Timothy T. Pennucci", "Benetge B. P. Perera", "Nihan S. Pol", "Henri A. Radovan", "Scott M. Ransom", "Paul S. Ray", "Ann Schmiedekamp", "Carl Schmiedekamp", "Brent J. Shapiro-Albert", "Ingrid H. Stairs", "Kevin Stovall", "Abhimanyu Susobhanan", "Joseph K. Swiggum", "Stephen R. Taylor", "Michele Vallisneri", "Rutger van Haasteren", "Haley M. Wahl" ], "categories": [ "gr-qc", "astro-ph.HE" ], "topics": [ "agent-evaluation", "rag" ], "score": 3, "relevance": "low", "matched_queries": [ "web-gui-agent" ], "arxiv_id": "2606.30536", "source": "arxiv", "source_id": "arxiv:2606.30536", "pdf_url": "https://arxiv.org/pdf/2606.30536", "primary_query": "web-gui-agent" }, { "id": "2606.23478", "title": "ffortissimo: A Freeform Forward-Modeling Pipeline for High-Contrast Images of Circumstellar Disks Based on Automatic Differentiation", "url": "https://arxiv.org/abs/2606.23478", "published": "2026-06-22", "updated": "2026-06-22", "authors": [ "Jay K. Kueny", "Joseph D. Long", "Jared R. Males", "Alycia J. Weinberger", "Laird M. Close", "Joshua Liberman", "Sebastiaan Haffert", "Eden McEwen", "Maggie Y. Kautz", "Olivier Guyon", "Logan Pearce", "Parker T. Johnson", "Katie Twitchell", "Jialin Li", "Alex Hedglen", "Avalon Gower", "Warren Foster", "Jhen Lumbres", "Lauren Schatz" ], "categories": [ "astro-ph.IM", "astro-ph.SR" ], "topics": [ "agent-evaluation", "tool-use" ], "score": 3, "relevance": "low", "matched_queries": [ "web-gui-agent" ], "arxiv_id": "2606.23478", "source": "arxiv", "source_id": "arxiv:2606.23478", "pdf_url": "https://arxiv.org/pdf/2606.23478", "primary_query": "web-gui-agent" }, { "id": "2606.20257", "title": "Measurements of charged-particle pseudorapidity and transverse momentum distributions in O+O and Ne+Ne collisions at $\\sqrt{s_{_\\text{NN}}} = 5.36$ TeV with the ATLAS detector", "url": "https://arxiv.org/abs/2606.20257", "published": "2026-06-18", "updated": "2026-06-18", "authors": [ "ATLAS Collaboration" ], "categories": [ "nucl-ex", "hep-ex" ], "topics": [ "agent-evaluation", "tool-use" ], "score": 3, "relevance": "low", "matched_queries": [ "web-gui-agent" ], "arxiv_id": "2606.20257", "source": "arxiv", "source_id": "arxiv:2606.20257", "pdf_url": "https://arxiv.org/pdf/2606.20257", "primary_query": "web-gui-agent" }, { "id": "2606.18975", "title": "On the robustness of the angular homogeneity scale $θ_H$: a comparative analysis of computational approaches", "url": "https://arxiv.org/abs/2606.18975", "published": "2026-06-17", "updated": "2026-06-17", "authors": [ "Pedro Fanha", "António da Silva", "José Fonseca", "José Pedro Mimoso", "Ziad Sakr" ], "categories": [ "astro-ph.CO" ], "topics": [ "agent-evaluation", "world-model" ], "score": 3, "relevance": "low", "matched_queries": [ "web-gui-agent" ], "arxiv_id": "2606.18975", "source": "arxiv", "source_id": "arxiv:2606.18975", "pdf_url": "https://arxiv.org/pdf/2606.18975", "primary_query": "web-gui-agent" }, { "id": "2606.15776", "title": "Spin-dependent electron transfer through a ring-wire coupled junction: Role of in-plane electric field", "url": "https://arxiv.org/abs/2606.15776", "published": "2026-06-14", "updated": "2026-06-14", "authors": [ "Prabhab Patra", "Santanu K. Maiti" ], "categories": [ "cond-mat.mes-hall" ], "topics": [ "computer-use", "planning" ], "score": 3, "relevance": "low", "matched_queries": [ "web-gui-agent" ], "arxiv_id": "2606.15776", "source": "arxiv", "source_id": "arxiv:2606.15776", "pdf_url": "https://arxiv.org/pdf/2606.15776", "primary_query": "web-gui-agent" }, { "id": "2606.15336", "title": "Nonlocal Orbital-Free Kinetic Energy Functional from the Jellium-with-Gap Model for Finite Systems", "url": "https://arxiv.org/abs/2606.15336", "published": "2026-06-13", "updated": "2026-06-13", "authors": [ "Abhishek Bhattacharjee", "Subrata Jana", "Szymon Smiga", "Prasanjit Samal" ], "categories": [ "cond-mat.mtrl-sci" ], "topics": [ "agent-evaluation" ], "score": 3, "relevance": "low", "matched_queries": [ "web-gui-agent" ], "arxiv_id": "2606.15336", "source": "arxiv", "source_id": "arxiv:2606.15336", "pdf_url": "https://arxiv.org/pdf/2606.15336", "primary_query": "web-gui-agent" }, { "id": "2606.14131", "title": "G-computation for causal effect estimation from observational hierarchical data with unmeasured cluster context", "url": "https://arxiv.org/abs/2606.14131", "published": "2026-06-12", "updated": "2026-06-12", "authors": [ "Shafayet Khan Shafee", "Bishal Sarker", "Md. Niamul Islam Sium" ], "categories": [ "stat.ME" ], "topics": [ "agent-evaluation", "world-model" ], "score": 3, "relevance": "low", "matched_queries": [ "web-gui-agent" ], "arxiv_id": "2606.14131", "source": "arxiv", "source_id": "arxiv:2606.14131", "pdf_url": "https://arxiv.org/pdf/2606.14131", "primary_query": "web-gui-agent" }, { "id": "2606.13980", "title": "On Cutting Cakes and Crossing Curves", "url": "https://arxiv.org/abs/2606.13980", "published": "2026-06-11", "updated": "2026-06-11", "authors": [ "Alexandros Hollender", "Gilbert Maystre", "Kilian Risse" ], "categories": [ "cs.GT", "cs.CC" ], "topics": [ "agent" ], "score": 3, "relevance": "low", "matched_queries": [ "web-gui-agent" ], "arxiv_id": "2606.13980", "source": "arxiv", "source_id": "arxiv:2606.13980", "pdf_url": "https://arxiv.org/pdf/2606.13980", "primary_query": "web-gui-agent" }, { "id": "2606.08070", "title": "Earth-Density Stratification and Quantum Gravity Corrections in Long-Baseline Neutrino Oscillation Experiments", "url": "https://arxiv.org/abs/2606.08070", "published": "2026-06-06", "updated": "2026-06-06", "authors": [ "Bipin Singh Koranga", "Vivek Kumar Nautiya" ], "categories": [ "hep-ph" ], "topics": [ "agent-evaluation", "planning" ], "score": 3, "relevance": "low", "matched_queries": [ "web-gui-agent" ], "arxiv_id": "2606.08070", "source": "arxiv", "source_id": "arxiv:2606.08070", "pdf_url": "https://arxiv.org/pdf/2606.08070", "primary_query": "web-gui-agent" }, { "id": "2607.01374", "title": "Algebraic conditions for second-moment stability boundaries of linear, time-invariant stochastic delay-differential equations", "url": "https://arxiv.org/abs/2607.01374", "published": "2026-07-01", "updated": "2026-07-01", "authors": [ "Zsolt Iklodi", "Harry Dankowicz" ], "categories": [ "math.DS" ], "topics": [ "world-model" ], "score": 2, "relevance": "low", "matched_queries": [ "web-gui-agent" ], "arxiv_id": "2607.01374", "source": "arxiv", "source_id": "arxiv:2607.01374", "pdf_url": "https://arxiv.org/pdf/2607.01374", "primary_query": "web-gui-agent" }, { "id": "2606.30540", "title": "Synthesizability and Mechanical Properties of High-Entropy Borides: First-Principles and Machine Learning Studies", "url": "https://arxiv.org/abs/2606.30540", "published": "2026-06-29", "updated": "2026-06-29", "authors": [ "Luke Moore", "Ethan Fox", "Bria Storr", "Jayden R. Palomino", "Shane A. Catledge", "Yogesh K. Vohra", "Cheng-Chien Chen" ], "categories": [ "cond-mat.mtrl-sci" ], "topics": [ "agent-evaluation" ], "score": 2, "relevance": "low", "matched_queries": [ "web-gui-agent" ], "arxiv_id": "2606.30540", "source": "arxiv", "source_id": "arxiv:2606.30540", "pdf_url": "https://arxiv.org/pdf/2606.30540", "primary_query": "web-gui-agent" }, { "id": "2606.29989", "title": "Rendering Coherent Scattering via Quantum Collision Models", "url": "https://arxiv.org/abs/2606.29989", "published": "2026-06-29", "updated": "2026-06-29", "authors": [ "João S. Ferreira", "Spencer S. Topel", "Pierre Fromholz", "James R. Wootton" ], "categories": [ "cs.GR", "physics.pop-ph", "quant-ph" ], "topics": [ "tool-use" ], "score": 2, "relevance": "low", "matched_queries": [ "web-gui-agent" ], "arxiv_id": "2606.29989", "source": "arxiv", "source_id": "arxiv:2606.29989", "pdf_url": "https://arxiv.org/pdf/2606.29989", "primary_query": "web-gui-agent" }, { "id": "2606.28146", "title": "A statistically robust framework for detecting and classifying hysteresis patterns in astrophysical spectral evolution", "url": "https://arxiv.org/abs/2606.28146", "published": "2026-06-26", "updated": "2026-06-26", "authors": [ "Tomislav Terzić" ], "categories": [ "astro-ph.HE", "astro-ph.IM" ], "topics": [ "planning" ], "score": 2, "relevance": "low", "matched_queries": [ "web-gui-agent" ], "arxiv_id": "2606.28146", "source": "arxiv", "source_id": "arxiv:2606.28146", "pdf_url": "https://arxiv.org/pdf/2606.28146", "primary_query": "web-gui-agent" }, { "id": "2606.20975", "title": "Solving Einstein Field Equations on a Digital Quantum Computer", "url": "https://arxiv.org/abs/2606.20975", "published": "2026-06-18", "updated": "2026-06-18", "authors": [ "Clelia Altomonte", "Malcolm Fairbairn" ], "categories": [ "gr-qc" ], "topics": [ "world-model" ], "score": 2, "relevance": "low", "matched_queries": [ "web-gui-agent" ], "arxiv_id": "2606.20975", "source": "arxiv", "source_id": "arxiv:2606.20975", "pdf_url": "https://arxiv.org/pdf/2606.20975", "primary_query": "web-gui-agent" }, { "id": "2606.20796", "title": "$N=1$ Supersymmetry, Weil-Petersson Volume Recursion, and a Spectral Curve", "url": "https://arxiv.org/abs/2606.20796", "published": "2026-06-18", "updated": "2026-06-18", "authors": [ "Clifford V. Johnson" ], "categories": [ "hep-th", "math-ph" ], "topics": [ "agent-evaluation" ], "score": 2, "relevance": "low", "matched_queries": [ "web-gui-agent" ], "arxiv_id": "2606.20796", "source": "arxiv", "source_id": "arxiv:2606.20796", "pdf_url": "https://arxiv.org/pdf/2606.20796", "primary_query": "web-gui-agent" }, { "id": "2606.18196", "title": "Receiver-Aware Analysis and Verification of the Spectral Separation Coefficient Under Interference-Induced Degradation", "url": "https://arxiv.org/abs/2606.18196", "published": "2026-06-16", "updated": "2026-06-18", "authors": [ "Lucas Heublein", "Fabian Benschuh", "Alexander Rügamer", "Felix Ott" ], "categories": [ "eess.SP" ], "topics": [ "agent-evaluation" ], "score": 2, "relevance": "low", "matched_queries": [ "web-gui-agent" ], "arxiv_id": "2606.18196", "source": "arxiv", "source_id": "arxiv:2606.18196", "pdf_url": "https://arxiv.org/pdf/2606.18196", "primary_query": "web-gui-agent" }, { "id": "2606.12698", "title": "Higher Dimensional Loop Quantum Black hole in de Sitter Spacetime: Quasinormal Modes and Shadow Signatures", "url": "https://arxiv.org/abs/2606.12698", "published": "2026-06-10", "updated": "2026-06-10", "authors": [ "Kourosh Nozari", "Sara Saghafi", "Ali Mohammadpour" ], "categories": [ "gr-qc", "hep-ph", "hep-th" ], "topics": [ "tool-use" ], "score": 2, "relevance": "low", "matched_queries": [ "web-gui-agent" ], "arxiv_id": "2606.12698", "source": "arxiv", "source_id": "arxiv:2606.12698", "pdf_url": "https://arxiv.org/pdf/2606.12698", "primary_query": "web-gui-agent" }, { "id": "2606.10341", "title": "Global polarization of $Λ$, $Ξ^{-}$, and $Ω^{-}$ hyperons in Au+Au collisions at RHIC BES-II energies", "url": "https://arxiv.org/abs/2606.10341", "published": "2026-06-09", "updated": "2026-06-09", "authors": [ "Gen-Hui Li", "Cong Yi", "Xiang-Yu Wu", "Shi Pu", "Guang-You Qin" ], "categories": [ "nucl-th" ], "topics": [ "tool-use" ], "score": 2, "relevance": "low", "matched_queries": [ "web-gui-agent" ], "arxiv_id": "2606.10341", "source": "arxiv", "source_id": "arxiv:2606.10341", "pdf_url": "https://arxiv.org/pdf/2606.10341", "primary_query": "web-gui-agent" }, { "id": "2607.02160", "title": "Algorithms for hyperelliptic Mumford Curves $p$-adic Uniformization, $p$-adic integrals and $p$-adic heights", "url": "https://arxiv.org/abs/2607.02160", "published": "2026-07-02", "updated": "2026-07-02", "authors": [ "Enis Kaya", "Marc Masdeu", "J. Steffen Müller", "Marius van der Put" ], "categories": [ "math.NT", "math.AG" ], "topics": [], "score": 1, "relevance": "low", "matched_queries": [ "web-gui-agent" ], "arxiv_id": "2607.02160", "source": "arxiv", "source_id": "arxiv:2607.02160", "pdf_url": "https://arxiv.org/pdf/2607.02160", "primary_query": "web-gui-agent" }, { "id": "2607.01650", "title": "Computed emissivity of carbon dioxide, water vapor, and their mixtures for a wide range of temperatures and pressure-pathlengths", "url": "https://arxiv.org/abs/2607.01650", "published": "2026-07-02", "updated": "2026-07-02", "authors": [ "Osama A. Marzouk" ], "categories": [ "physics.gen-ph" ], "topics": [], "score": 1, "relevance": "low", "matched_queries": [ "web-gui-agent" ], "arxiv_id": "2607.01650", "source": "arxiv", "source_id": "arxiv:2607.01650", "pdf_url": "https://arxiv.org/pdf/2607.01650", "primary_query": "web-gui-agent" }, { "id": "2607.00476", "title": "Complexity of Low-Degree Skew Polynomial Multiplication over Finite Fields", "url": "https://arxiv.org/abs/2607.00476", "published": "2026-07-01", "updated": "2026-07-01", "authors": [ "Ke Ye", "Yichuan Cao", "Ruichen Qiu" ], "categories": [ "cs.SC" ], "topics": [], "score": 1, "relevance": "low", "matched_queries": [ "web-gui-agent" ], "arxiv_id": "2607.00476", "source": "arxiv", "source_id": "arxiv:2607.00476", "pdf_url": "https://arxiv.org/pdf/2607.00476", "primary_query": "web-gui-agent" }, { "id": "2606.29294", "title": "Quantum models of the Riemann zeta function, lattice spin models and algebraic models of entanglement", "url": "https://arxiv.org/abs/2606.29294", "published": "2026-06-28", "updated": "2026-06-28", "authors": [ "Nikolaj M. Glazunov" ], "categories": [ "math.NT" ], "topics": [], "score": 1, "relevance": "low", "matched_queries": [ "web-gui-agent" ], "arxiv_id": "2606.29294", "source": "arxiv", "source_id": "arxiv:2606.29294", "pdf_url": "https://arxiv.org/pdf/2606.29294", "primary_query": "web-gui-agent" }, { "id": "2606.27415", "title": "Elimination of Flux Trapping in Superconducting Circuits in Ambient Magnetic Fields", "url": "https://arxiv.org/abs/2606.27415", "published": "2026-06-25", "updated": "2026-06-25", "authors": [ "Rohan T. Kapur", "Alex Wynn", "Sergey K. Tolpygo", "Neel Parmar", "Anil Mankame", "Adam A. Libson", "Rabindra Das", "Michele Kelley", "Pauli Kehayias", "Nathaniel J. O'Connor", "Collin N. Muniz", "Justin L. Mallek", "Jennifer M. Schloss" ], "categories": [ "cond-mat.supr-con", "cond-mat.mes-hall", "physics.app-ph", "quant-ph" ], "topics": [], "score": 1, "relevance": "low", "matched_queries": [ "web-gui-agent" ], "arxiv_id": "2606.27415", "source": "arxiv", "source_id": "arxiv:2606.27415", "pdf_url": "https://arxiv.org/pdf/2606.27415", "primary_query": "web-gui-agent" }, { "id": "2606.26255", "title": "Hochschild (co)homology and cyclic homology via a graded Euler characteristic with applications to higher preprojective algebras", "url": "https://arxiv.org/abs/2606.26255", "published": "2026-06-24", "updated": "2026-06-24", "authors": [ "Jon Wallem Anundsen", "Mads Hustad Sandøy" ], "categories": [ "math.RT" ], "topics": [], "score": 1, "relevance": "low", "matched_queries": [ "web-gui-agent" ], "arxiv_id": "2606.26255", "source": "arxiv", "source_id": "arxiv:2606.26255", "pdf_url": "https://arxiv.org/pdf/2606.26255", "primary_query": "web-gui-agent" }, { "id": "2606.21829", "title": "Elementary solutions of ordinary tropical differential equations, and vanishing orders of solutions of algebraic differential equations", "url": "https://arxiv.org/abs/2606.21829", "published": "2026-06-20", "updated": "2026-06-20", "authors": [ "Cristhian Garay-López", "Johana Luviano-Flores", "Carla Valencia-Negrete" ], "categories": [ "math.AG", "math.CO" ], "topics": [], "score": 1, "relevance": "low", "matched_queries": [ "web-gui-agent" ], "arxiv_id": "2606.21829", "source": "arxiv", "source_id": "arxiv:2606.21829", "pdf_url": "https://arxiv.org/pdf/2606.21829", "primary_query": "web-gui-agent" }, { "id": "2606.19503", "title": "Kernel transformations and bounds for smeared spectral functions", "url": "https://arxiv.org/abs/2606.19503", "published": "2026-06-17", "updated": "2026-06-17", "authors": [ "William I. Jay", "Matteo Saccardi" ], "categories": [ "hep-lat" ], "topics": [], "score": 1, "relevance": "low", "matched_queries": [ "web-gui-agent" ], "arxiv_id": "2606.19503", "source": "arxiv", "source_id": "arxiv:2606.19503", "pdf_url": "https://arxiv.org/pdf/2606.19503", "primary_query": "web-gui-agent" }, { "id": "2606.16510", "title": "Petrov-Galerkin Variational Physics-Informed Neural Network Framework for Two-Dimensional Singularly Perturbed Problems", "url": "https://arxiv.org/abs/2606.16510", "published": "2026-06-15", "updated": "2026-06-15", "authors": [ "Vijay Kumar", "Gautam Singh" ], "categories": [ "math.NA", "cs.LG" ], "topics": [], "score": 1, "relevance": "low", "matched_queries": [ "web-gui-agent" ], "arxiv_id": "2606.16510", "source": "arxiv", "source_id": "arxiv:2606.16510", "pdf_url": "https://arxiv.org/pdf/2606.16510", "primary_query": "web-gui-agent" }, { "id": "2606.12998", "title": "Efficient emulation of nuclear ground states with neural-network variational Monte Carlo and eigenvector continuation", "url": "https://arxiv.org/abs/2606.12998", "published": "2026-06-11", "updated": "2026-06-12", "authors": [ "Mao Li", "Yilong Yang", "Pengwei Zhao" ], "categories": [ "nucl-th" ], "topics": [], "score": 1, "relevance": "low", "matched_queries": [ "web-gui-agent" ], "arxiv_id": "2606.12998", "source": "arxiv", "source_id": "arxiv:2606.12998", "pdf_url": "https://arxiv.org/pdf/2606.12998", "primary_query": "web-gui-agent" } ]