[ { "collection": "jobs", "path": "jobs/items/2026-07-08-baidu-aidu-agent-algorithm-engineer-beijing.md", "title": "2027AIDU-智能体算法工程师", "type": "job", "meta": { "type": "job", "company": "百度", "role": "2027AIDU-智能体算法工程师", "location": "北京市", "source_url": "https://talent.baidu.com/jobs/list?projectType=3&recruitType=GRADUATE", "source_name": "百度校园招聘", "source_quality": "official", "collected_at": "2026-07-08", "posted_at": "2026-05-12", "first_seen_at": "2026-07-08", "last_checked_at": "2026-07-08", "snapshot_type": "new", "job_code": "J99969", "previous_snapshot": [], "status": "active", "level": "graduate", "team": "AIDU项目", "business_area": "search / conversation / office / data platform / entertainment", "employment_type": "campus", "salary": [], "skills": [ "agent", "planning", "tool-use", "memory", "multi-agent", "rag", "agent-evaluation" ], "topics": [ "agent-architecture", "evaluation", "memory" ], "models": [], "related_papers": [ "2025-vijayvargiya-openagentsafety", "2026-memory-agent-survey" ], "related_experiments": [], "related_projects": [], "relevance": "high" } }, { "collection": "jobs", "path": "jobs/items/2026-07-08-baidu-aidu-agent-fullstack-engineer-beijing.md", "title": "2027AIDU-Agent应用全栈工程师", "type": "job", "meta": { "type": "job", "company": "百度", "role": "2027AIDU-Agent应用全栈工程师", "location": "北京市", "source_url": "https://talent.baidu.com/jobs/list?projectType=3&recruitType=GRADUATE", "source_name": "百度校园招聘", "source_quality": "official", "collected_at": "2026-07-08", "posted_at": "2026-05-12", "first_seen_at": "2026-07-08", "last_checked_at": "2026-07-08", "snapshot_type": "new", "job_code": "J99974", "previous_snapshot": [], "status": "active", "level": "graduate", "team": "AIDU项目", "business_area": "search / healthcare / enterprise service / data analysis / office automation", "employment_type": "campus", "salary": [], "skills": [ "agent", "planning", "tool-use", "function-calling", "memory", "reasoning", "state-management", "multi-agent", "rag", "agent-evaluation" ], "topics": [ "workflow-agent", "agent-architecture", "evaluation" ], "models": [], "related_papers": [ "2026-dialogue-swebench", "2026-swe-evo" ], "related_experiments": [], "related_projects": [], "relevance": "high" } }, { "collection": "jobs", "path": "jobs/items/2026-07-08-baidu-aidu-llm-infra-engineer-beijing.md", "title": "2027AIDU-大模型Infra工程师", "type": "job", "meta": { "type": "job", "company": "百度", "role": "2027AIDU-大模型Infra工程师", "location": "北京市", "source_url": "https://talent.baidu.com/jobs/list?projectType=3&recruitType=GRADUATE", "source_name": "百度校园招聘", "source_quality": "official", "collected_at": "2026-07-08", "posted_at": "2026-05-12", "first_seen_at": "2026-07-08", "last_checked_at": "2026-07-08", "snapshot_type": "new", "job_code": "J99967", "previous_snapshot": [], "status": "active", "level": "graduate", "team": "AIDU项目", "business_area": "model infrastructure", "employment_type": "campus", "salary": [], "skills": [ "inference", "serving", "distributed-training", "gpu", "model-compression", "cloud" ], "topics": [ "llm-infra", "inference" ], "models": [], "related_papers": [], "related_experiments": [], "related_projects": [], "relevance": "medium" } }, { "collection": "jobs", "path": "jobs/items/2026-07-08-bytedance-seed-llm-agent-research-engineer.md", "title": "大模型Agent研究工程师-Seed", "type": "job", "meta": { "type": "job", "company": "字节 Seed", "role": "大模型Agent研究工程师-Seed", "location": "unknown", "source_url": "https://jobs.bytedance.com/experienced/position/7628902314323806469/detail", "source_name": "字节跳动招聘", "source_quality": "official", "collected_at": "2026-07-08", "posted_at": [], "first_seen_at": "2026-07-08", "last_checked_at": "2026-07-08", "snapshot_type": "new", "job_code": "7628902314323806469", "previous_snapshot": [], "status": "unknown", "level": "experienced", "team": "Seed", "business_area": "agent research and engineering", "employment_type": "full-time", "salary": [], "skills": [ "agent", "memory", "context-engineering", "planning", "multi-agent", "llm-application" ], "topics": [ "harness", "memory", "context-compression" ], "models": [ "Seed" ], "related_papers": [ "2026-memory-agent-survey", "2026-evomembench" ], "related_experiments": [], "related_projects": [], "relevance": "high" } }, { "collection": "jobs", "path": "jobs/items/2026-07-08-deepseek-agent-hiring-wave-beijing-hangzhou.md", "title": "Agent 方向招聘组合", "type": "job", "meta": { "type": "job", "company": "DeepSeek", "role": "Agent 方向招聘组合", "location": "北京市 / 杭州市", "source_url": "https://hub.baai.ac.cn/view/53416", "source_name": "智源社区转载量子位", "source_quality": "secondary", "collected_at": "2026-07-08", "posted_at": "2026-03-27", "first_seen_at": "2026-07-08", "last_checked_at": "2026-07-08", "snapshot_type": "new", "job_code": [], "previous_snapshot": [], "status": "unknown", "level": "mixed", "team": "Agent / data evaluation / infra / product", "business_area": "search / creation / multimodal / personal assistant / workflow", "employment_type": "full-time / internship", "salary": [], "skills": [ "agent", "rl", "agent-evaluation", "tool-use", "function-calling", "memory", "multi-agent", "coding-agent", "data-quality", "inference" ], "topics": [ "agent-productization", "evaluation", "agent-infra" ], "models": [ "DeepSeek" ], "related_papers": [ "2025-vijayvargiya-openagentsafety", "2026-agentrx" ], "related_experiments": [], "related_projects": [], "relevance": "high" } }, { "collection": "jobs", "path": "jobs/items/2026-07-08-tencent-cloud-ai-agent-test-engineer.md", "title": "腾讯云-AI Agent测试工程师", "type": "job", "meta": { "type": "job", "company": "腾讯", "role": "腾讯云-AI Agent测试工程师", "location": "深圳市", "source_url": "https://careers.tencent.com/jobdesc.html?postId=2055555661177204736", "source_name": "腾讯招聘", "source_quality": "official", "collected_at": "2026-07-08", "posted_at": "2026-06-03", "first_seen_at": "2026-07-08", "last_checked_at": "2026-07-08", "snapshot_type": "new", "job_code": "2055555661177204736", "previous_snapshot": [], "status": "active", "level": "experienced", "team": "CSIG", "business_area": "cloud agent testing", "employment_type": "full-time", "salary": [], "skills": [ "agent", "agent-evaluation", "observability", "cloud", "kubernetes", "system-design" ], "topics": [ "agent-testing", "harness-engineering" ], "models": [], "related_papers": [ "2026-agentrx", "2025-vijayvargiya-openagentsafety" ], "related_experiments": [], "related_projects": [], "relevance": "medium" } }, { "collection": "papers", "path": "papers/items/2025-2507-20395-mazeeval-a-benchmark-for-testing-sequential-decision-making-in-language-models.md", "title": "\"MazeEval: A Benchmark for Testing Sequential Decision-Making in Language Models\"", "type": "paper", "meta": { "type": "paper", "title": "\"MazeEval: A Benchmark for Testing Sequential Decision-Making in Language Models\"", "authors": "Hafsteinn Einarsson", "year": "2025", "venue": "arXiv", "url": "https://arxiv.org/abs/2507.20395", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2025-07-27", "updated_at": "2025-07-27", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "computer-use", "embodied-agent", "reasoning" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "13", "collection_queries": "function-calling" } }, { "collection": "papers", "path": "papers/items/2025-2507-20666-mimii-agent-leveraging-llms-with-function-calling-for-relative-evaluation-of-ano.md", "title": "\"MIMII-Agent: Leveraging LLMs with Function Calling for Relative Evaluation of Anomalous Sound Detection\"", "type": "paper", "meta": { "type": "paper", "title": "\"MIMII-Agent: Leveraging LLMs with Function Calling for Relative Evaluation of Anomalous Sound Detection\"", "authors": "Harsh Purohit, Tomoya Nishida, Kota Dohi, Takashi Endo, Yohei Kawaguchi", "year": "2025", "venue": "arXiv", "url": "https://arxiv.org/abs/2507.20666", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2025-07-28", "updated_at": "2025-07-28", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "rag", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "eess.AS", "cs.AI", "cs.LG", "cs.SD" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "13", "collection_queries": "function-calling" } }, { "collection": "papers", "path": "papers/items/2025-2508-07575-mcptoolbench-a-large-scale-ai-agent-model-context-protocol-mcp-tool-use-benchmar.md", "title": "\"MCPToolBench++: A Large Scale AI Agent Model Context Protocol MCP Tool Use Benchmark\"", "type": "paper", "meta": { "type": "paper", "title": "\"MCPToolBench++: A Large Scale AI Agent Model Context Protocol MCP Tool Use Benchmark\"", "authors": "Shiqing Fan, Xichen Ding, Liang Zhang, Linjian Mo", "year": "2025", "venue": "arXiv", "url": "https://arxiv.org/abs/2508.07575", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2025-08-11", "updated_at": "2025-08-11", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "computer-use", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "21", "collection_queries": "function-calling" } }, { "collection": "papers", "path": "papers/items/2025-2508-11027-hell-or-high-water-evaluating-agentic-recovery-from-external-failures.md", "title": "\"Hell or High Water: Evaluating Agentic Recovery from External Failures\"", "type": "paper", "meta": { "type": "paper", "title": "\"Hell or High Water: Evaluating Agentic Recovery from External Failures\"", "authors": "Andrew Wang, Sophia Hager, Adi Asija, Daniel Khashabi, Nicholas Andrews", "year": "2025", "venue": "arXiv", "url": "https://arxiv.org/abs/2508.11027", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2025-08-14", "updated_at": "2025-08-14", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "planning", "tool-use", "workflow-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CL" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "16", "collection_queries": "function-calling" } }, { "collection": "papers", "path": "papers/items/2025-2508-12685-toolace-mt-non-autoregressive-generation-for-agentic-multi-turn-interaction.md", "title": "\"ToolACE-MT: Non-Autoregressive Generation for Agentic Multi-Turn Interaction\"", "type": "paper", "meta": { "type": "paper", "title": "\"ToolACE-MT: Non-Autoregressive Generation for Agentic Multi-Turn Interaction\"", "authors": "Xingshan Zeng, Weiwen Liu, Lingzhi Wang, Liangyou Li, Fei Mi, Yasheng Wang, Lifeng Shang, Xin Jiang, et al.", "year": "2025", "venue": "arXiv", "url": "https://arxiv.org/abs/2508.12685", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2025-08-18", "updated_at": "2026-02-13", "status": "queued", "relevance": "high", "topics": [ "tool-use", "world-model" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CL", "cs.AI", "cs.LG" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "13", "collection_queries": "function-calling" } }, { "collection": "papers", "path": "papers/items/2025-2508-17094-powerchain-a-verifiable-agentic-ai-system-for-automating-distribution-grid-analy.md", "title": "\"PowerChain: A Verifiable Agentic AI System for Automating Distribution Grid Analyses\"", "type": "paper", "meta": { "type": "paper", "title": "\"PowerChain: A Verifiable Agentic AI System for Automating Distribution Grid Analyses\"", "authors": "Emmanuel O. Badmus, Peng Sang, Dimitrios Stamoulis, Amritanshu Pandey", "year": "2025", "venue": "arXiv", "url": "https://arxiv.org/abs/2508.17094", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2025-08-23", "updated_at": "2025-10-21", "status": "queued", "relevance": "high", "topics": [ "planning", "rag", "reasoning", "tool-use", "workflow-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI", "eess.SY" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "13", "collection_queries": "function-calling" } }, { "collection": "papers", "path": "papers/items/2025-2509-02444-appcopilot-toward-general-accurate-long-horizon-and-efficient-mobile-agent.md", "title": "\"AppCopilot: Toward General, Accurate, Long-Horizon, and Efficient Mobile Agent\"", "type": "paper", "meta": { "type": "paper", "title": "\"AppCopilot: Toward General, Accurate, Long-Horizon, and Efficient Mobile Agent\"", "authors": "Jingru Fan, Yufan Dang, Jingyao Wu, Huatao Li, Runde Yang, Xiyuan Yang, Yuheng Wang, Chen Qian", "year": "2025", "venue": "arXiv", "url": "https://arxiv.org/abs/2509.02444", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2025-09-02", "updated_at": "2025-10-17", "status": "queued", "relevance": "high", "topics": [ "computer-use", "memory", "multi-agent", "planning", "reasoning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI", "cs.CL", "cs.CV", "cs.HC" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "15", "collection_queries": "function-calling" } }, { "collection": "papers", "path": "papers/items/2025-2509-02494-gridmind-llms-powered-agents-for-power-system-analysis-and-operations.md", "title": "\"GridMind: LLMs-Powered Agents for Power System Analysis and Operations\"", "type": "paper", "meta": { "type": "paper", "title": "\"GridMind: LLMs-Powered Agents for Power System Analysis and Operations\"", "authors": "Hongwei Jin, Kibaek Kim, Jonghwan Kwon", "year": "2025", "venue": "arXiv", "url": "https://arxiv.org/abs/2509.02494", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2025-09-02", "updated_at": "2025-09-02", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "multi-agent", "workflow-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "13", "collection_queries": "function-calling" } }, { "collection": "papers", "path": "papers/items/2025-2509-08863-geojson-agents-a-multi-agent-llm-architecture-for-geospatial-analysis-function-c.md", "title": "\"GeoJSON Agents:A Multi-Agent LLM Architecture for Geospatial Analysis-Function Calling vs Code Generation\"", "type": "paper", "meta": { "type": "paper", "title": "\"GeoJSON Agents:A Multi-Agent LLM Architecture for Geospatial Analysis-Function Calling vs Code Generation\"", "authors": "Qianqian Luo, Qingming Lin, Liuchang Xu, Sensen Wu, Ruichen Mao, Chao Wang, Hailin Feng, Bo Huang, et al.", "year": "2025", "venue": "arXiv", "url": "https://arxiv.org/abs/2509.08863", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2025-09-10", "updated_at": "2025-12-03", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "multi-agent", "planning", "tool-use", "workflow-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.SE" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "16", "collection_queries": "function-calling" } }, { "collection": "papers", "path": "papers/items/2025-2509-10769-agentarch-a-comprehensive-benchmark-to-evaluate-agent-architectures-in-enterpris.md", "title": "\"AgentArch: A Comprehensive Benchmark to Evaluate Agent Architectures in Enterprise\"", "type": "paper", "meta": { "type": "paper", "title": "\"AgentArch: A Comprehensive Benchmark to Evaluate Agent Architectures in Enterprise\"", "authors": "Tara Bogavelli, Roshnee Sharma, Hari Subramani", "year": "2025", "venue": "arXiv", "url": "https://arxiv.org/abs/2509.10769", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2025-09-13", "updated_at": "2026-01-06", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "memory", "multi-agent", "tool-use", "workflow-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI", "cs.CL", "cs.MA" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "18", "collection_queries": "function-calling" } }, { "collection": "papers", "path": "papers/items/2025-2509-13311-towards-general-agentic-intelligence-via-environment-scaling.md", "title": "Towards General Agentic Intelligence via Environment Scaling", "type": "paper", "meta": { "type": "paper", "title": "Towards General Agentic Intelligence via Environment Scaling", "authors": "Runnan Fang, Shihao Cai, Baixuan Li, Jialong Wu, Guangyu Li, Wenbiao Yin, Xinyu Wang, Xiaobin Wang, et al.", "year": "2025", "venue": "arXiv", "url": "https://arxiv.org/abs/2509.13311", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2025-09-16", "updated_at": "2025-09-16", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CL" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "13", "collection_queries": "function-calling" } }, { "collection": "papers", "path": "papers/items/2025-2509-14477-ticket-bench-a-kickoff-for-multilingual-and-regionalized-agent-evaluation.md", "title": "\"Ticket-Bench: A Kickoff for Multilingual and Regionalized Agent Evaluation\"", "type": "paper", "meta": { "type": "paper", "title": "\"Ticket-Bench: A Kickoff for Multilingual and Regionalized Agent Evaluation\"", "authors": "Thales Sales Almeida, João Guilherme Alves Santos, Thiago Laitz, Giovana Kerche Bonás", "year": "2025", "venue": "arXiv", "url": "https://arxiv.org/abs/2509.14477", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2025-09-17", "updated_at": "2025-09-17", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "computer-use", "reasoning" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CL" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "15", "collection_queries": "function-calling" } }, { "collection": "papers", "path": "papers/items/2025-2509-20998-core-full-path-evaluation-of-llm-agents-beyond-final-state.md", "title": "\"CORE: Full-Path Evaluation of LLM Agents Beyond Final State\"", "type": "paper", "meta": { "type": "paper", "title": "\"CORE: Full-Path Evaluation of LLM Agents Beyond Final State\"", "authors": "Panagiotis Michelakis, Yiannis Hadjiyiannis, Dimitrios Stamoulis", "year": "2025", "venue": "arXiv", "url": "https://arxiv.org/abs/2509.20998", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2025-09-25", "updated_at": "2025-09-25", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "tool-use", "world-model" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "20", "collection_queries": "function-calling" } }, { "collection": "papers", "path": "papers/items/2025-2509-26553-towards-reliable-benchmarking-a-contamination-free-controllable-evaluation-frame.md", "title": "\"Towards Reliable Benchmarking: A Contamination Free, Controllable Evaluation Framework for Multi-step LLM Function Calling\"", "type": "paper", "meta": { "type": "paper", "title": "\"Towards Reliable Benchmarking: A Contamination Free, Controllable Evaluation Framework for Multi-step LLM Function Calling\"", "authors": "Seiji Maekawa, Jackson Hassell, Pouya Pezeshkpour, Tom Mitchell, Estevam Hruschka", "year": "2025", "venue": "arXiv", "url": "https://arxiv.org/abs/2509.26553", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2025-09-30", "updated_at": "2026-02-06", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "reasoning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CL", "cs.PL" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "15", "collection_queries": "function-calling" } }, { "collection": "papers", "path": "papers/items/2025-2510-03847-small-language-models-for-agentic-systems-a-survey-of-architectures-capabilities.md", "title": "\"Small Language Models for Agentic Systems: A Survey of Architectures, Capabilities, and Deployment Trade offs\"", "type": "paper", "meta": { "type": "paper", "title": "\"Small Language Models for Agentic Systems: A Survey of Architectures, Capabilities, and Deployment Trade offs\"", "authors": "Raghav Sharma, Manan Mehta", "year": "2025", "venue": "arXiv", "url": "https://arxiv.org/abs/2510.03847", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2025-10-04", "updated_at": "2025-10-04", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "coding-agent", "computer-use", "planning", "rag", "reasoning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI", "cs.LG" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "19", "collection_queries": "function-calling" } }, { "collection": "papers", "path": "papers/items/2025-2510-04206-agentrl-scaling-agentic-reinforcement-learning-with-a-multi-turn-multi-task-fram.md", "title": "\"AgentRL: Scaling Agentic Reinforcement Learning with a Multi-Turn, Multi-Task Framework\"", "type": "paper", "meta": { "type": "paper", "title": "\"AgentRL: Scaling Agentic Reinforcement Learning with a Multi-Turn, Multi-Task Framework\"", "authors": "Hanchen Zhang, Xiao Liu, Bowen Lv, Xueqiao Sun, Bohao Jing, Iat Long Iong, Zhenyu Hou, Zehan Qi, et al.", "year": "2025", "venue": "arXiv", "url": "https://arxiv.org/abs/2510.04206", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2025-10-05", "updated_at": "2025-10-05", "status": "queued", "relevance": "high", "topics": [ "rag", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "13", "collection_queries": "function-calling" } }, { "collection": "papers", "path": "papers/items/2025-2510-14548-llm-agents-beyond-utility-an-open-ended-perspective.md", "title": "\"LLM Agents Beyond Utility: An Open-Ended Perspective\"", "type": "paper", "meta": { "type": "paper", "title": "\"LLM Agents Beyond Utility: An Open-Ended Perspective\"", "authors": "Asen Nachkov, Xi Wang, Luc Van Gool", "year": "2025", "venue": "arXiv", "url": "https://arxiv.org/abs/2510.14548", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2025-10-16", "updated_at": "2025-10-16", "status": "queued", "relevance": "high", "topics": [ "memory", "planning", "reasoning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "15", "collection_queries": "function-calling" } }, { "collection": "papers", "path": "papers/items/2025-2510-18586-tokencake-a-kv-cache-centric-serving-framework-for-llm-based-multi-agent-applica.md", "title": "\"TokenCake: A KV-Cache-centric Serving Framework for LLM-based Multi-Agent Applications\"", "type": "paper", "meta": { "type": "paper", "title": "\"TokenCake: A KV-Cache-centric Serving Framework for LLM-based Multi-Agent Applications\"", "authors": "Zhuohang Bian, Feiyang Wu, Zhuoran Li, Teng Ma, Youwei Zhuo", "year": "2025", "venue": "arXiv", "url": "https://arxiv.org/abs/2510.18586", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2025-10-21", "updated_at": "2026-05-20", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "computer-use", "memory", "multi-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.DC" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "17", "collection_queries": "function-calling" } }, { "collection": "papers", "path": "papers/items/2025-2510-21524-eu-agent-bench-measuring-illegal-behavior-of-llm-agents-under-eu-law.md", "title": "\"EU-Agent-Bench: Measuring Illegal Behavior of LLM Agents Under EU Law\"", "type": "paper", "meta": { "type": "paper", "title": "\"EU-Agent-Bench: Measuring Illegal Behavior of LLM Agents Under EU Law\"", "authors": "Ilija Lichkovski, Alexander Müller, Mariam Ibrahim, Tiwai Mhundwa", "year": "2025", "venue": "arXiv", "url": "https://arxiv.org/abs/2510.21524", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2025-10-24", "updated_at": "2025-10-24", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "rag", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "16", "collection_queries": "function-calling" } }, { "collection": "papers", "path": "papers/items/2025-2510-22768-seeing-is-believing-evaluating-vision-language-model-susceptibility-in-agent-to-.md", "title": "Seeing is Believing? Evaluating Vision-Language Model Susceptibility in Agent-to-Agent Multimodal Persuasion", "type": "paper", "meta": { "type": "paper", "title": "Seeing is Believing? Evaluating Vision-Language Model Susceptibility in Agent-to-Agent Multimodal Persuasion", "authors": "Haoyi Qiu, Yilun Zhou, Pranav Narayanan Venkit, Kung-Hsiang Huang, Jiaxin Zhang, Nanyun Peng, Chien-Sheng Wu", "year": "2025", "venue": "arXiv", "url": "https://arxiv.org/abs/2510.22768", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2025-10-26", "updated_at": "2026-06-02", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "multi-agent", "rag", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CL" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "14", "collection_queries": "function-calling" } }, { "collection": "papers", "path": "papers/items/2025-2510-24645-funreason-mt-technical-report-advanced-data-synthesis-solution-for-real-world-mu.md", "title": "\"FunReason-MT Technical Report: Advanced Data Synthesis Solution for Real-world Multi-Turn Tool-use\"", "type": "paper", "meta": { "type": "paper", "title": "\"FunReason-MT Technical Report: Advanced Data Synthesis Solution for Real-world Multi-Turn Tool-use\"", "authors": "Zengzhuang Xu, Bingguang Hao, Zechuan Wang, Yuntao Wen, Xinyi Xu, Yang Liu, Long Chen, Dong Wang, et al.", "year": "2025", "venue": "arXiv", "url": "https://arxiv.org/abs/2510.24645", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2025-10-28", "updated_at": "2025-11-16", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "computer-use", "multi-agent", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "13", "collection_queries": "function-calling" } }, { "collection": "papers", "path": "papers/items/2025-2510-26167-toolrm-towards-agentic-tool-use-reward-modeling.md", "title": "\"ToolRM: Towards Agentic Tool-Use Reward Modeling\"", "type": "paper", "meta": { "type": "paper", "title": "\"ToolRM: Towards Agentic Tool-Use Reward Modeling\"", "authors": "Renhao Li, Jianhong Tu, Yang Su, Yantao Liu, Fei Huang, Hamid Alinejad-Rokny, Derek F. Wong, Junyang Lin, et al.", "year": "2025", "venue": "arXiv", "url": "https://arxiv.org/abs/2510.26167", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2025-10-30", "updated_at": "2026-01-13", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI", "cs.CL" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "14", "collection_queries": "function-calling" } }, { "collection": "papers", "path": "papers/items/2025-2511-04847-test-time-adaptation-for-llm-agents-via-environment-interaction.md", "title": "Test-Time Adaptation for LLM Agents via Environment Interaction", "type": "paper", "meta": { "type": "paper", "title": "Test-Time Adaptation for LLM Agents via Environment Interaction", "authors": "Arthur Chen, Zuxin Liu, Jianguo Zhang, Akshara Prabhakar, Zhiwei Liu, Shelby Heinecke, Silvio Savarese, Victor Zhong, et al.", "year": "2025", "venue": "arXiv", "url": "https://arxiv.org/abs/2511.04847", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2025-11-06", "updated_at": "2026-02-22", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "embodied-agent", "rag", "tool-use", "world-model" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.LG" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "18", "collection_queries": "function-calling" } }, { "collection": "papers", "path": "papers/items/2025-2511-11169-refine-and-align-confidence-calibration-through-multi-agent-interaction-in-vqa.md", "title": "\"Refine and Align: Confidence Calibration through Multi-Agent Interaction in VQA\"", "type": "paper", "meta": { "type": "paper", "title": "\"Refine and Align: Confidence Calibration through Multi-Agent Interaction in VQA\"", "authors": "Ayush Pandey, Jai Bardhan, Ishita Jain, Ramya S Hebbalaguppe, Rohan Raju Dhanakshirur, Lovekesh Vig", "year": "2025", "venue": "arXiv", "url": "https://arxiv.org/abs/2511.11169", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2025-11-14", "updated_at": "2025-11-14", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "embodied-agent", "multi-agent", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CV", "cs.AI", "cs.LG" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "16", "collection_queries": "function-calling" } }, { "collection": "papers", "path": "papers/items/2025-2511-15203-taxonomy-evaluation-and-exploitation-of-ipi-centric-llm-agent-defense-frameworks.md", "title": "Taxonomy, Evaluation and Exploitation of IPI-Centric LLM Agent Defense Frameworks", "type": "paper", "meta": { "type": "paper", "title": "Taxonomy, Evaluation and Exploitation of IPI-Centric LLM Agent Defense Frameworks", "authors": "Zimo Ji, Xunguang Wang, Zongjie Li, Pingchuan Ma, Yudong Gao, Daoyuan Wu, Xincheng Yan, Tian Tian, et al.", "year": "2025", "venue": "arXiv", "url": "https://arxiv.org/abs/2511.15203", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2025-11-19", "updated_at": "2025-11-19", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "rag", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CR", "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "17", "collection_queries": "function-calling" } }, { "collection": "papers", "path": "papers/items/2025-2511-22138-tinyllm-evaluation-and-optimization-of-small-language-models-for-agentic-tasks-o.md", "title": "\"TinyLLM: Evaluation and Optimization of Small Language Models for Agentic Tasks on Edge Devices\"", "type": "paper", "meta": { "type": "paper", "title": "\"TinyLLM: Evaluation and Optimization of Small Language Models for Agentic Tasks on Edge Devices\"", "authors": "Mohd Ariful Haque, Fahad Rahman, Kishor Datta Gupta, Khalil Shujaee, Roy George", "year": "2025", "venue": "arXiv", "url": "https://arxiv.org/abs/2511.22138", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2025-11-27", "updated_at": "2025-11-27", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.LG" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "16", "collection_queries": "function-calling" } }, { "collection": "papers", "path": "papers/items/2025-2512-02605-iact-a-self-organizing-recursive-model-for-general-ai-agents-a-technical-white-p.md", "title": "\"IACT: A Self-Organizing Recursive Model for General AI Agents: A Technical White Paper on the Architecture Behind kragent.ai\"", "type": "paper", "meta": { "type": "paper", "title": "\"IACT: A Self-Organizing Recursive Model for General AI Agents: A Technical White Paper on the Architecture Behind kragent.ai\"", "authors": "Pengju Lu", "year": "2025", "venue": "arXiv", "url": "https://arxiv.org/abs/2512.02605", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2025-12-02", "updated_at": "2025-12-02", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "computer-use", "memory", "rag", "tool-use", "workflow-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI", "cs.MA", "cs.SE" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "15", "collection_queries": "function-calling" } }, { "collection": "papers", "path": "papers/items/2025-2512-11682-medai-evaluating-txagent-s-therapeutic-agentic-reasoning-in-the-neurips-cure-ben.md", "title": "\"MedAI: Evaluating TxAgent's Therapeutic Agentic Reasoning in the NeurIPS CURE-Bench Competition\"", "type": "paper", "meta": { "type": "paper", "title": "\"MedAI: Evaluating TxAgent's Therapeutic Agentic Reasoning in the NeurIPS CURE-Bench Competition\"", "authors": "Tim Cofala, Christian Kalfar, Jingge Xiao, Johanna Schrader, Michelle Tang, Wolfgang Nejdl", "year": "2025", "venue": "arXiv", "url": "https://arxiv.org/abs/2512.11682", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2025-12-12", "updated_at": "2026-06-15", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "computer-use", "planning", "rag", "reasoning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI", "cs.LG" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "17", "collection_queries": "function-calling" } }, { "collection": "papers", "path": "papers/items/2025-2512-23611-close-the-loop-synthesizing-infinite-tool-use-data-via-multi-agent-role-playing.md", "title": "\"Close the Loop: Synthesizing Infinite Tool-Use Data via Multi-Agent Role-Playing\"", "type": "paper", "meta": { "type": "paper", "title": "\"Close the Loop: Synthesizing Infinite Tool-Use Data via Multi-Agent Role-Playing\"", "authors": "Yuwen Li, Wei Zhang, Zelong Huang, Mason Yang, Jiajun Wu, Shawn Guo, Huahao Hu, Lingyi Sun, et al.", "year": "2025", "venue": "arXiv", "url": "https://arxiv.org/abs/2512.23611", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2025-12-29", "updated_at": "2025-12-29", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "multi-agent", "rag", "tool-use", "workflow-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CL" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "15", "collection_queries": "function-calling" } }, { "collection": "papers", "path": "papers/items/2025-2512-23647-nested-browser-use-learning-for-agentic-information-seeking.md", "title": "Nested Browser-Use Learning for Agentic Information Seeking", "type": "paper", "meta": { "type": "paper", "title": "Nested Browser-Use Learning for Agentic Information Seeking", "authors": "Baixuan Li, Jialong Wu, Wenbiao Yin, Kuan Li, Zhongwang Zhang, Huifeng Yin, Zhengwei Tao, Liwen Zhang, et al.", "year": "2025", "venue": "arXiv", "url": "https://arxiv.org/abs/2512.23647", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2025-12-29", "updated_at": "2025-12-29", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "computer-use", "rag", "reasoning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CL", "cs.AI", "cs.IR", "cs.MA" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "15", "collection_queries": "function-calling" } }, { "collection": "papers", "path": "papers/items/2025-2512-23747-state-of-the-art-small-language-coder-model-mify-coder.md", "title": "\"State-of-the-art Small Language Coder Model: Mify-Coder\"", "type": "paper", "meta": { "type": "paper", "title": "\"State-of-the-art Small Language Coder Model: Mify-Coder\"", "authors": "Abhinav Parmar, Abhisek Panigrahi, Abhishek Kumar Dwivedi, Abhishek Bhattacharya, Adarsh Ramachandra, Aditya Choudhary, Aditya Garg, Aditya Raj, et al.", "year": "2025", "venue": "arXiv", "url": "https://arxiv.org/abs/2512.23747", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2025-12-26", "updated_at": "2025-12-26", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "coding-agent", "computer-use", "workflow-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.SE", "cs.AI", "cs.CL" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "13", "collection_queries": "function-calling" } }, { "collection": "papers", "path": "papers/items/2025-vijayvargiya-openagentsafety.md", "title": "OpenAgentSafety: A Comprehensive Framework for Evaluating Real-World AI Agent Safety", "type": "paper", "meta": { "type": "paper", "title": "OpenAgentSafety: A Comprehensive Framework for Evaluating Real-World AI Agent Safety", "authors": "Sanidhya Vijayvargiya, Aditya Bharat Soni, Xuhui Zhou, Zora Zhiruo Wang, Nouha Dziri, Graham Neubig, Maarten Sap", "year": "2025", "venue": "ICLR 2026 / IASEAI 2026", "url": "https://arxiv.org/abs/2507.06134", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "status": "skimmed", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "tool-use" ], "methods": [ "multi-turn-agent-evaluation", "rule-based-analysis", "llm-as-judge" ], "benchmarks": [ "OpenAgentSafety" ], "models": [], "datasets": [], "related_concepts": [ "guardrail", "human-in-the-loop" ], "related_jobs": [ "2026-07-08-baidu-aidu-agent-algorithm-engineer-beijing", "2026-07-08-tencent-cloud-ai-agent-test-engineer" ], "related_experiments": [], "related_projects": [] } }, { "collection": "papers", "path": "papers/items/2026-2601-00268-beyond-perfect-apis-a-comprehensive-evaluation-of-llm-agents-under-real-world-ap.md", "title": "\"Beyond Perfect APIs: A Comprehensive Evaluation of LLM Agents Under Real-World API Complexity\"", "type": "paper", "meta": { "type": "paper", "title": "\"Beyond Perfect APIs: A Comprehensive Evaluation of LLM Agents Under Real-World API Complexity\"", "authors": "Doyoung Kim, Zhiwei Ren, Jie Hao, Zhongkai Sun, Lichao Wang, Xiyao Ma, Zack Ye, Xu Han, et al.", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2601.00268", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-01-01", "updated_at": "2026-01-01", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CL", "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "16", "collection_queries": "function-calling" } }, { "collection": "papers", "path": "papers/items/2026-2601-05467-stelp-secure-transpilation-and-execution-of-llm-generated-programs.md", "title": "\"STELP: Secure Transpilation and Execution of LLM-Generated Programs\"", "type": "paper", "meta": { "type": "paper", "title": "\"STELP: Secure Transpilation and Execution of LLM-Generated Programs\"", "authors": "Swapnil Shinde, Sahil Wadhwa, Andy Luo, Akshay Gupta, Mohammad Shahed Sorower", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2601.05467", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-01-09", "updated_at": "2026-01-15", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "multi-agent", "planning", "reasoning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.SE", "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "14", "collection_queries": "function-calling" } }, { "collection": "papers", "path": "papers/items/2026-2601-06007-don-t-break-the-cache-an-evaluation-of-prompt-caching-for-long-horizon-agentic-t.md", "title": "\"Don't Break the Cache: An Evaluation of Prompt Caching for Long-Horizon Agentic Tasks\"", "type": "paper", "meta": { "type": "paper", "title": "\"Don't Break the Cache: An Evaluation of Prompt Caching for Long-Horizon Agentic Tasks\"", "authors": "Elias Lumer, Faheem Nizar, Akshaya Jangiti, Kevin Frank, Anmol Gulati, Mandar Phadate, Vamse Kumar Subbiah", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2601.06007", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-01-09", "updated_at": "2026-01-31", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "computer-use", "planning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CL" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "18", "collection_queries": "function-calling" } }, { "collection": "papers", "path": "papers/items/2026-2601-06606-cedar-context-engineering-for-agentic-data-science.md", "title": "\"CEDAR: Context Engineering for Agentic Data Science\"", "type": "paper", "meta": { "type": "paper", "title": "\"CEDAR: Context Engineering for Agentic Data Science\"", "authors": "Rishiraj Saha Roy, Chris Hinze, Luzian Hahn, Fabian Kuech", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2601.06606", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-01-10", "updated_at": "2026-04-22", "status": "queued", "relevance": "high", "topics": [ "planning", "workflow-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.LG", "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "13", "collection_queries": "function-calling" } }, { "collection": "papers", "path": "papers/items/2026-2601-12988-paperguide-making-small-language-model-paper-reading-agents-more-efficient.md", "title": "\"PaperGuide: Making Small Language-Model Paper-Reading Agents More Efficient\"", "type": "paper", "meta": { "type": "paper", "title": "\"PaperGuide: Making Small Language-Model Paper-Reading Agents More Efficient\"", "authors": "Zijian Wang, Tiancheng Huang, Hanqi Li, Da Ma, Lu Chen, Kai Yu", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2601.12988", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-01-19", "updated_at": "2026-01-19", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "computer-use", "planning", "reasoning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.LG" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "13", "collection_queries": "function-calling" } }, { "collection": "papers", "path": "papers/items/2026-2601-14652-mas-orchestra-understanding-and-improving-multi-agent-reasoning-through-holistic.md", "title": "\"MAS-Orchestra: Understanding and Improving Multi-Agent Reasoning Through Holistic Orchestration and Controlled Benchmarks\"", "type": "paper", "meta": { "type": "paper", "title": "\"MAS-Orchestra: Understanding and Improving Multi-Agent Reasoning Through Holistic Orchestration and Controlled Benchmarks\"", "authors": "Zixuan Ke, Yifei Ming, Austin Xu, Ryan Chin, Xuan-Phi Nguyen, Prathyusha Jwalapuram, Jiayu Wang, Semih Yavuz, et al.", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2601.14652", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-01-21", "updated_at": "2026-05-21", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "computer-use", "multi-agent", "reasoning" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI", "cs.CL", "cs.MA" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "15", "collection_queries": "function-calling" } }, { "collection": "papers", "path": "papers/items/2026-2602-03117-agentdyn-are-your-agent-security-defenses-deployable-in-real-world-dynamic-envir.md", "title": "\"AgentDyn: Are Your Agent Security Defenses Deployable in Real-World Dynamic Environments?\"", "type": "paper", "meta": { "type": "paper", "title": "\"AgentDyn: Are Your Agent Security Defenses Deployable in Real-World Dynamic Environments?\"", "authors": "Hao Li, Ruoyao Wen, Shanghao Shi, Ning Zhang, Yevgeniy Vorobeychik, Chaowei Xiao", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2602.03117", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-02-03", "updated_at": "2026-05-07", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "planning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CR" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "13", "collection_queries": "agent-safety" } }, { "collection": "papers", "path": "papers/items/2026-2602-03224-tame-a-trustworthy-test-time-evolution-of-agent-memory-with-systematic-benchmark.md", "title": "\"TAME: A Trustworthy Test-Time Evolution of Agent Memory with Systematic Benchmarking\"", "type": "paper", "meta": { "type": "paper", "title": "\"TAME: A Trustworthy Test-Time Evolution of Agent Memory with Systematic Benchmarking\"", "authors": "Yu Cheng, Yongkang Hu, Jiuan Zhou, Yushuo Zhang, Yihang Chen, Huichi Zhou, Mingang Chen, Zhizhong Zhang, et al.", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2602.03224", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-02-03", "updated_at": "2026-06-06", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "computer-use", "memory", "reasoning" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI", "cs.LG" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "17", "collection_queries": "agent-safety" } }, { "collection": "papers", "path": "papers/items/2026-2602-03786-aorchestra-automating-sub-agent-creation-for-agentic-orchestration.md", "title": "\"AOrchestra: Automating Sub-Agent Creation for Agentic Orchestration\"", "type": "paper", "meta": { "type": "paper", "title": "\"AOrchestra: Automating Sub-Agent Creation for Agentic Orchestration\"", "authors": "Jianhao Ruan, Zhihao Xu, Yiran Peng, Fashen Ren, Zhaoyang Yu, Xinbing Liang, Jinyu Xiang, Yongru Chen, et al.", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2602.03786", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-02-03", "updated_at": "2026-02-07", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "coding-agent", "planning", "tool-use", "workflow-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI", "cs.CL" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "16", "collection_queries": "language-agent" } }, { "collection": "papers", "path": "papers/items/2026-2602-05115-socialveil-probing-social-intelligence-of-language-agents-under-communication-ba.md", "title": "\"SocialVeil: Probing Social Intelligence of Language Agents under Communication Barriers\"", "type": "paper", "meta": { "type": "paper", "title": "\"SocialVeil: Probing Social Intelligence of Language Agents under Communication Barriers\"", "authors": "Keyang Xuan, Pengda Wang, Chongrui Ye, Haofei Yu, Tal August, Jiaxuan You", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2602.05115", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-02-04", "updated_at": "2026-02-04", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "rag", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI", "cs.CL" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "16", "collection_queries": "language-agent" } }, { "collection": "papers", "path": "papers/items/2026-2602-05302-piearena-ranking-and-profiling-language-agents-in-realistic-negotiation-scenario.md", "title": "\"PieArena: Ranking and Profiling Language Agents in Realistic Negotiation Scenarios\"", "type": "paper", "meta": { "type": "paper", "title": "\"PieArena: Ranking and Profiling Language Agents in Realistic Negotiation Scenarios\"", "authors": "Chris Zhu, Sasha Cui, Will Sanok Dufallo, Runzhi Jin, Zhen Xu, Linjun Zhang, Daylian Cain", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2602.05302", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-02-05", "updated_at": "2026-06-01", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "multi-agent", "reasoning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "18", "collection_queries": "language-agent" } }, { "collection": "papers", "path": "papers/items/2026-2602-05386-spider-sense-intrinsic-risk-sensing-for-efficient-agent-defense-with-hierarchica.md", "title": "\"Spider-Sense: Intrinsic Risk Sensing for Efficient Agent Defense with Hierarchical Adaptive Screening\"", "type": "paper", "meta": { "type": "paper", "title": "\"Spider-Sense: Intrinsic Risk Sensing for Efficient Agent Defense with Hierarchical Adaptive Screening\"", "authors": "Zhenxiong Yu, Zhi Yang, Zhiheng Jin, Shuhe Wang, Heng Zhang, Yanlin Fei, Lingfeng Zeng, Fangqi Lou, et al.", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2602.05386", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-02-05", "updated_at": "2026-02-06", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "reasoning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CR", "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "13", "collection_queries": "agent-safety" } }, { "collection": "papers", "path": "papers/items/2026-2602-07391-naamse-framework-for-evolutionary-security-evaluation-of-agents.md", "title": "\"NAAMSE: Framework for Evolutionary Security Evaluation of Agents\"", "type": "paper", "meta": { "type": "paper", "title": "\"NAAMSE: Framework for Evolutionary Security Evaluation of Agents\"", "authors": "Kunal Pai, Parth Shah, Harshil Patel", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2602.07391", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-02-07", "updated_at": "2026-03-08", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI", "cs.MA" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "14", "collection_queries": "agent-safety" } }, { "collection": "papers", "path": "papers/items/2026-2602-07652-agent-fence-mapping-security-vulnerabilities-across-deep-research-agents.md", "title": "\"Agent-Fence: Mapping Security Vulnerabilities Across Deep Research Agents\"", "type": "paper", "meta": { "type": "paper", "title": "\"Agent-Fence: Mapping Security Vulnerabilities Across Deep Research Agents\"", "authors": "Sai Puppala, Ismail Hossain, Md Jahangir Alam, Yoonpyo Lee, Jay Yoo, Tanzim Ahad, Syed Bahauddin Alam, Sajedul Talukder", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2602.07652", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-02-07", "updated_at": "2026-02-07", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "memory", "planning", "rag", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CR", "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "16", "collection_queries": "agent-safety" } }, { "collection": "papers", "path": "papers/items/2026-2602-07962-loca-bench-benchmarking-language-agents-under-controllable-and-extreme-context-g.md", "title": "\"LOCA-bench: Benchmarking Language Agents Under Controllable and Extreme Context Growth\"", "type": "paper", "meta": { "type": "paper", "title": "\"LOCA-bench: Benchmarking Language Agents Under Controllable and Extreme Context Growth\"", "authors": "Weihao Zeng, Yuzhen Huang, Junxian He", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2602.07962", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-02-08", "updated_at": "2026-02-08", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "planning", "rag", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "18", "collection_queries": "language-agent" } }, { "collection": "papers", "path": "papers/items/2026-2602-08082-spectral-guardrails-for-agents-in-the-wild-detecting-tool-use-hallucinations-via.md", "title": "\"Spectral Guardrails for Agents in the Wild: Detecting Tool Use Hallucinations via Attention Topology\"", "type": "paper", "meta": { "type": "paper", "title": "\"Spectral Guardrails for Agents in the Wild: Detecting Tool Use Hallucinations via Attention Topology\"", "authors": "Valentin Noël", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2602.08082", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-02-08", "updated_at": "2026-02-08", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.LG", "cs.AI", "eess.SP" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "13", "collection_queries": "agent-safety" } }, { "collection": "papers", "path": "papers/items/2026-2602-08412-from-assistant-to-double-agent-formalizing-and-benchmarking-attacks-on-openclaw-.md", "title": "\"From Assistant to Double Agent: Formalizing and Benchmarking Attacks on OpenClaw for Personalized Local AI Agent\"", "type": "paper", "meta": { "type": "paper", "title": "\"From Assistant to Double Agent: Formalizing and Benchmarking Attacks on OpenClaw for Personalized Local AI Agent\"", "authors": "Yuhang Wang, Feiming Xu, Zheng Lin, Guangyu He, Yuzhe Huang, Haichang Gao, Zhenxing Niu, Shiguo Lian, et al.", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2602.08412", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-02-09", "updated_at": "2026-02-11", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "memory", "planning", "rag", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "21", "collection_queries": "agent-safety" } }, { "collection": "papers", "path": "papers/items/2026-2602-10133-agenttrace-a-structured-logging-framework-for-agent-system-observability.md", "title": "\"AgentTrace: A Structured Logging Framework for Agent System Observability\"", "type": "paper", "meta": { "type": "paper", "title": "\"AgentTrace: A Structured Logging Framework for Agent System Observability\"", "authors": "Adam AlSayyad, Kelvin Yuxiang Huang, Richik Pal", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2602.10133", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-02-07", "updated_at": "2026-02-07", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "reasoning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.SE", "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "13", "collection_queries": "agent-safety" } }, { "collection": "papers", "path": "papers/items/2026-2602-11749-air-improving-agent-safety-through-incident-response.md", "title": "\"AIR: Improving Agent Safety through Incident Response\"", "type": "paper", "meta": { "type": "paper", "title": "\"AIR: Improving Agent Safety through Incident Response\"", "authors": "Zibo Xiao, Jun Sun, Junjie Chen", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2602.11749", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-02-12", "updated_at": "2026-06-20", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "computer-use", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "15", "collection_queries": "agent-safety" } }, { "collection": "papers", "path": "papers/items/2026-2602-13379-unsafer-in-many-turns-benchmarking-and-defending-multi-turn-safety-risks-in-tool.md", "title": "\"Unsafer in Many Turns: Benchmarking and Defending Multi-Turn Safety Risks in Tool-Using Agents\"", "type": "paper", "meta": { "type": "paper", "title": "\"Unsafer in Many Turns: Benchmarking and Defending Multi-Turn Safety Risks in Tool-Using Agents\"", "authors": "Xu Li, Simon Yu, Minzhou Pan, Yiyou Sun, Bo Li, Dawn Song, Xue Lin, Weiyan Shi", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2602.13379", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-02-13", "updated_at": "2026-06-10", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "rag", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CR", "cs.AI", "cs.CL", "cs.LG", "cs.SE" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "15", "collection_queries": "agent-safety" } }, { "collection": "papers", "path": "papers/items/2026-2602-13530-remem-reasoning-with-episodic-memory-in-language-agent.md", "title": "\"REMem: Reasoning with Episodic Memory in Language Agent\"", "type": "paper", "meta": { "type": "paper", "title": "\"REMem: Reasoning with Episodic Memory in Language Agent\"", "authors": "Yiheng Shu, Saisri Padmaja Jonnalagedda, Xiang Gao, Bernal Jiménez Gutiérrez, Weijian Qi, Kamalika Das, Huan Sun, Yu Su", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2602.13530", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-02-13", "updated_at": "2026-02-28", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "memory", "rag", "reasoning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "19", "collection_queries": "language-agent" } }, { "collection": "papers", "path": "papers/items/2026-2602-13665-hyfunc-accelerating-llm-based-function-calls-for-agentic-ai-through-hybrid-model.md", "title": "\"HyFunc: Accelerating LLM-based Function Calls for Agentic AI through Hybrid-Model Cascade and Dynamic Templating\"", "type": "paper", "meta": { "type": "paper", "title": "\"HyFunc: Accelerating LLM-based Function Calls for Agentic AI through Hybrid-Model Cascade and Dynamic Templating\"", "authors": "Weibin Liao, Jian-guang Lou, Haoyi Xiong", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2602.13665", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-02-14", "updated_at": "2026-02-14", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "computer-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "13", "collection_queries": "function-calling" } }, { "collection": "papers", "path": "papers/items/2026-2602-14234-redsearcher-a-scalable-and-cost-efficient-framework-for-long-horizon-search-agen.md", "title": "\"REDSearcher: A Scalable and Cost-Efficient Framework for Long-Horizon Search Agents\"", "type": "paper", "meta": { "type": "paper", "title": "\"REDSearcher: A Scalable and Cost-Efficient Framework for Long-Horizon Search Agents\"", "authors": "Zheng Chu, Xiao Wang, Jack Hong, Huiming Fan, Yuqi Huang, Yue Yang, Guohai Xu, Chenxiao Zhao, et al.", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2602.14234", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-02-15", "updated_at": "2026-02-15", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "planning", "rag", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI", "cs.CL" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "13", "collection_queries": "function-calling" } }, { "collection": "papers", "path": "papers/items/2026-2602-14281-mcpshield-a-security-cognition-layer-for-adaptive-trust-calibration-in-model-con.md", "title": "\"MCPShield: A Security Cognition Layer for Adaptive Trust Calibration in Model Context Protocol Agents\"", "type": "paper", "meta": { "type": "paper", "title": "\"MCPShield: A Security Cognition Layer for Adaptive Trust Calibration in Model Context Protocol Agents\"", "authors": "Zhenhong Zhou, Yuanhe Zhang, Hongwei Cai, Moayad Aloqaily, Ouns Bouachir, Linsey Pang, Prakhar Mehrotra, Kun Wang, et al.", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2602.14281", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-02-15", "updated_at": "2026-02-24", "status": "queued", "relevance": "high", "topics": [ "agent-safety", "computer-use", "reasoning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CR", "cs.CL" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "13", "collection_queries": "agent-safety" } }, { "collection": "papers", "path": "papers/items/2026-2602-16931-narrow-fine-tuning-erodes-safety-alignment-in-vision-language-agents.md", "title": "Narrow Fine-Tuning Erodes Safety Alignment in Vision-Language Agents", "type": "paper", "meta": { "type": "paper", "title": "Narrow Fine-Tuning Erodes Safety Alignment in Vision-Language Agents", "authors": "Idhant Gulati, Shivam Raval", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2602.16931", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-02-18", "updated_at": "2026-03-15", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "14", "collection_queries": "language-agent" } }, { "collection": "papers", "path": "papers/items/2026-2602-18456-beyond-single-channel-agentic-benchmarking.md", "title": "Beyond single-channel agentic benchmarking", "type": "paper", "meta": { "type": "paper", "title": "Beyond single-channel agentic benchmarking", "authors": "Nelu D. Radpour", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2602.18456", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-02-05", "updated_at": "2026-02-05", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CY", "cs.AI", "cs.HC" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "15", "collection_queries": "agent-safety" } }, { "collection": "papers", "path": "papers/items/2026-2602-19008-capable-but-unreliable-canonical-path-deviation-as-a-causal-mechanism-of-agent-f.md", "title": "\"Capable but Unreliable: Canonical Path Deviation as a Causal Mechanism of Agent Failure in Long-Horizon Tasks\"", "type": "paper", "meta": { "type": "paper", "title": "\"Capable but Unreliable: Canonical Path Deviation as a Causal Mechanism of Agent Failure in Long-Horizon Tasks\"", "authors": "Wilson Y. Lee", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2602.19008", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-02-22", "updated_at": "2026-02-22", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "computer-use", "planning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CL", "cs.LG" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "13", "collection_queries": "language-agent" } }, { "collection": "papers", "path": "papers/items/2026-2602-21127-are-you-sure-an-empirical-study-of-human-perception-vulnerability-in-llm-driven-.md", "title": "\"\\\"Are You Sure?\\\": An Empirical Study of Human Perception Vulnerability in LLM-Driven Agentic Systems\"", "type": "paper", "meta": { "type": "paper", "title": "\"\\\"Are You Sure?\\\": An Empirical Study of Human Perception Vulnerability in LLM-Driven Agentic Systems\"", "authors": "Xinfeng Li, Shenyu Dai, Kelong Zheng, Yue Xiao, Gelei Deng, Wei Dong, Xiaofeng Wang", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2602.21127", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-02-24", "updated_at": "2026-02-24", "status": "queued", "relevance": "high", "topics": [ "agent-safety", "tool-use", "workflow-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.HC", "cs.AI", "cs.CR", "cs.SI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "13", "collection_queries": "agent-safety" } }, { "collection": "papers", "path": "papers/items/2026-2602-23320-parammem-augmenting-language-agents-with-parametric-reflective-memory.md", "title": "\"ParamMem: Augmenting Language Agents with Parametric Reflective Memory\"", "type": "paper", "meta": { "type": "paper", "title": "\"ParamMem: Augmenting Language Agents with Parametric Reflective Memory\"", "authors": "Tianjun Yao, Yongqiang Chen, Yujia Zheng, Pan Li, Zhiqiang Shen, Kun Zhang", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2602.23320", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-02-26", "updated_at": "2026-02-27", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "memory", "reasoning" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.LG", "cs.MA" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "14", "collection_queries": "language-agent" } }, { "collection": "papers", "path": "papers/items/2026-2603-00131-thought-virus-viral-misalignment-via-subliminal-prompting-in-multi-agent-systems.md", "title": "\"Thought Virus: Viral Misalignment via Subliminal Prompting in Multi-Agent Systems\"", "type": "paper", "meta": { "type": "paper", "title": "\"Thought Virus: Viral Misalignment via Subliminal Prompting in Multi-Agent Systems\"", "authors": "Moritz Weckbecker, Jonas Müller, Ben Hagag, Michael Mulet", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2603.00131", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-02-23", "updated_at": "2026-02-23", "status": "queued", "relevance": "high", "topics": [ "agent-safety", "multi-agent", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.MA", "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "13", "collection_queries": "agent-safety" } }, { "collection": "papers", "path": "papers/items/2026-2603-00623-tracesir-a-multi-agent-framework-for-structured-analysis-and-reporting-of-agenti.md", "title": "\"TraceSIR: A Multi-Agent Framework for Structured Analysis and Reporting of Agentic Execution Traces\"", "type": "paper", "meta": { "type": "paper", "title": "\"TraceSIR: A Multi-Agent Framework for Structured Analysis and Reporting of Agentic Execution Traces\"", "authors": "Shu-Xun Yang, Cunxiang Wang, Haoke Zhang, Wenbo Yu, Lindong Wu, Jiayi Gui, Dayong Yang, Yukuo Cen, et al.", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2603.00623", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-02-28", "updated_at": "2026-02-28", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "coding-agent", "multi-agent", "reasoning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI", "cs.CL" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "19", "collection_queries": "function-calling" } }, { "collection": "papers", "path": "papers/items/2026-2603-00801-the-synthetic-web-adversarially-curated-mini-internets-for-diagnosing-epistemic-.md", "title": "\"The Synthetic Web: Adversarially-Curated Mini-Internets for Diagnosing Epistemic Weaknesses of Language Agents\"", "type": "paper", "meta": { "type": "paper", "title": "\"The Synthetic Web: Adversarially-Curated Mini-Internets for Diagnosing Epistemic Weaknesses of Language Agents\"", "authors": "Shrey Shah, Levent Ozgur", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2603.00801", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-02-28", "updated_at": "2026-02-28", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "embodied-agent", "rag", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI", "cs.IR" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "13", "collection_queries": "language-agent" } }, { "collection": "papers", "path": "papers/items/2026-2603-01438-enhancing-persona-following-at-decoding-time-via-dynamic-importance-estimation-f.md", "title": "Enhancing Persona Following at Decoding Time via Dynamic Importance Estimation for Role-Playing Agents", "type": "paper", "meta": { "type": "paper", "title": "Enhancing Persona Following at Decoding Time via Dynamic Importance Estimation for Role-Playing Agents", "authors": "Yuxin Liu, Mingye Zhu, Siyuan Liu, Bo Hu, Lei Zhang", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2603.01438", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-03-02", "updated_at": "2026-03-02", "status": "queued", "relevance": "high", "topics": [ "agent-safety", "coding-agent", "computer-use", "planning", "rag", "world-model" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CL", "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "14", "collection_queries": "language-agent" } }, { "collection": "papers", "path": "papers/items/2026-2603-01712-ft-dojo-towards-autonomous-llm-fine-tuning-with-language-agents.md", "title": "\"FT-Dojo: Towards Autonomous LLM Fine-Tuning with Language Agents\"", "type": "paper", "meta": { "type": "paper", "title": "\"FT-Dojo: Towards Autonomous LLM Fine-Tuning with Language Agents\"", "authors": "Qizheng Li, Yifei Zhang, Xiao Yang, Xu Yang, Zhuo Wang, Weiqing Liu, Jiang Bian", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2603.01712", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-03-02", "updated_at": "2026-05-20", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "coding-agent", "planning" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI", "cs.LG" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "15", "collection_queries": "language-agent" } }, { "collection": "papers", "path": "papers/items/2026-2603-02711-a-natural-language-agentic-approach-to-study-affective-polarization.md", "title": "A Natural Language Agentic Approach to Study Affective Polarization", "type": "paper", "meta": { "type": "paper", "title": "A Natural Language Agentic Approach to Study Affective Polarization", "authors": "Stephanie Anneris Malvicini, Ewelina Gajewska, Arda Derbent, Katarzyna Budzynska, Jarosław A. Chudziak, Maria Vanina Martinez", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2603.02711", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-03-03", "updated_at": "2026-03-03", "status": "queued", "relevance": "high", "topics": [ "multi-agent", "rag", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "17", "collection_queries": "language-agent" } }, { "collection": "papers", "path": "papers/items/2026-2603-03515-the-controllability-trap-a-governance-framework-for-military-ai-agents.md", "title": "\"The Controllability Trap: A Governance Framework for Military AI Agents\"", "type": "paper", "meta": { "type": "paper", "title": "\"The Controllability Trap: A Governance Framework for Military AI Agents\"", "authors": "Subramanyam Sahoo", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2603.03515", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-03-03", "updated_at": "2026-03-03", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "planning", "tool-use", "world-model" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CY", "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "17", "collection_queries": "agent-safety" } }, { "collection": "papers", "path": "papers/items/2026-2603-03680-mage-meta-reinforcement-learning-for-language-agents-toward-strategic-exploratio.md", "title": "\"MAGE: Meta-Reinforcement Learning for Language Agents toward Strategic Exploration and Exploitation\"", "type": "paper", "meta": { "type": "paper", "title": "\"MAGE: Meta-Reinforcement Learning for Language Agents toward Strategic Exploration and Exploitation\"", "authors": "Lu Yang, Zelai Xu, Minyang Xie, Jiaxuan Gao, Zhao Shok, Yu Wang, Yi Wu", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2603.03680", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-03-04", "updated_at": "2026-03-04", "status": "queued", "relevance": "high", "topics": [ "memory", "multi-agent", "reasoning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "17", "collection_queries": "language-agent" } }, { "collection": "papers", "path": "papers/items/2026-2603-05553-eigendata-a-self-evolving-multi-agent-platform-for-function-calling-data-synthes.md", "title": "\"EigenData: A Self-Evolving Multi-Agent Platform for Function-Calling Data Synthesis, Auditing, and Repair\"", "type": "paper", "meta": { "type": "paper", "title": "\"EigenData: A Self-Evolving Multi-Agent Platform for Function-Calling Data Synthesis, Auditing, and Repair\"", "authors": "Jiaao Chen, Jingyuan Qi, Mingye Gao, Wei-Chen Wang, Hanrui Wang, Di Jin", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2603.05553", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-03-05", "updated_at": "2026-03-05", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "coding-agent", "multi-agent", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.SE", "cs.AI", "cs.CL" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "16", "collection_queries": "function-calling" } }, { "collection": "papers", "path": "papers/items/2026-2603-05578-tool-genesis-a-task-driven-tool-creation-benchmark-for-self-evolving-language-ag.md", "title": "\"Tool-Genesis: A Task-Driven Tool Creation Benchmark for Self-Evolving Language Agent\"", "type": "paper", "meta": { "type": "paper", "title": "\"Tool-Genesis: A Task-Driven Tool Creation Benchmark for Self-Evolving Language Agent\"", "authors": "Bowei Xia, Mengkang Hu, Shijian Wang, Jiarui Jin, Wenxiang Jiao, Yuan Lu, Kexin Li, Ping Luo", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2603.05578", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-03-05", "updated_at": "2026-03-05", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "computer-use", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.SE", "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "15", "collection_queries": "language-agent" } }, { "collection": "papers", "path": "papers/items/2026-2603-07496-from-thinker-to-society-security-in-hierarchical-autonomy-evolution-of-ai-agents.md", "title": "\"From Thinker to Society: Security in Hierarchical Autonomy Evolution of AI Agents\"", "type": "paper", "meta": { "type": "paper", "title": "\"From Thinker to Society: Security in Hierarchical Autonomy Evolution of AI Agents\"", "authors": "Xiaolei Zhang, Lu Zhou, Xiaogang Xu, Jiafei Wu, Tianyu Du, Heqing Huang, Hao Peng, Zhe Liu", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2603.07496", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-03-08", "updated_at": "2026-03-21", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "computer-use", "multi-agent", "reasoning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CR", "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "17", "collection_queries": "agent-safety" } }, { "collection": "papers", "path": "papers/items/2026-2603-07557-agentraft-automated-detection-of-data-over-exposure-in-llm-agents.md", "title": "\"AgentRaft: Automated Detection of Data Over-Exposure in LLM Agents\"", "type": "paper", "meta": { "type": "paper", "title": "\"AgentRaft: Automated Detection of Data Over-Exposure in LLM Agents\"", "authors": "Yixi Lin, Jiangrong Wu, Yuhong Nan, Xueqiang Wang, Xinyuan Zhang, Zibin Zheng", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2603.07557", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-03-08", "updated_at": "2026-03-08", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "rag", "reasoning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.SE" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "15", "collection_queries": "function-calling" } }, { "collection": "papers", "path": "papers/items/2026-2603-07980-onemillion-bench-how-far-are-language-agents-from-human-experts.md", "title": "\"\\\\$OneMillion-Bench: How Far are Language Agents from Human Experts?\"", "type": "paper", "meta": { "type": "paper", "title": "\"\\\\$OneMillion-Bench: How Far are Language Agents from Human Experts?\"", "authors": "Qianyu Yang, Yang Liu, Jiaqi Li, Jun Bai, Hao Chen, Kaiyuan Chen, Tiliang Duan, Jiayun Dong, et al.", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2603.07980", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-03-09", "updated_at": "2026-03-09", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "planning", "reasoning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.LG", "cs.AI", "cs.CL" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "18", "collection_queries": "language-agent" } }, { "collection": "papers", "path": "papers/items/2026-2603-08721-kernelcraft-benchmarking-for-agentic-close-to-metal-kernel-generation-on-emergin.md", "title": "\"KernelCraft: Benchmarking for Agentic Close-to-Metal Kernel Generation on Emerging Hardware\"", "type": "paper", "meta": { "type": "paper", "title": "\"KernelCraft: Benchmarking for Agentic Close-to-Metal Kernel Generation on Emerging Hardware\"", "authors": "Jiayi Nie, Haoran Wu, Yao Lai, Zeyu Cao, Cheng Zhang, Binglei Lou, Erwei Wang, Jianyi Cheng, et al.", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2603.08721", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-02-10", "updated_at": "2026-05-29", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "reasoning", "workflow-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AR", "cs.LG", "cs.SE" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "17", "collection_queries": "function-calling" } }, { "collection": "papers", "path": "papers/items/2026-2603-09002-security-considerations-for-multi-agent-systems.md", "title": "Security Considerations for Multi-agent Systems", "type": "paper", "meta": { "type": "paper", "title": "Security Considerations for Multi-agent Systems", "authors": "Tam Nguyen, Moses Ndebugre, Dheeraj Arremsetty", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2603.09002", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-03-09", "updated_at": "2026-04-26", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "computer-use", "memory", "multi-agent", "planning", "rag", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CR", "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "18", "collection_queries": "agent-safety" } }, { "collection": "papers", "path": "papers/items/2026-2603-10492-human-ai-co-reasoning-for-clinical-diagnosis-with-evidence-integrated-language-a.md", "title": "Human-AI Co-reasoning for Clinical Diagnosis with Evidence-Integrated Language Agent", "type": "paper", "meta": { "type": "paper", "title": "Human-AI Co-reasoning for Clinical Diagnosis with Evidence-Integrated Language Agent", "authors": "Zhongzhen Huang, Yan Ling, Hong Chen, Ye Feng, Li Wu, Linjie Mu, Shaoting Zhang, Xiaofan Zhang, et al.", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2603.10492", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-03-11", "updated_at": "2026-03-18", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "multi-agent", "rag", "reasoning", "workflow-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CL" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "17", "collection_queries": "language-agent" } }, { "collection": "papers", "path": "papers/items/2026-2603-11088-the-attack-and-defense-landscape-of-agentic-ai-a-comprehensive-survey.md", "title": "\"The Attack and Defense Landscape of Agentic AI: A Comprehensive Survey\"", "type": "paper", "meta": { "type": "paper", "title": "\"The Attack and Defense Landscape of Agentic AI: A Comprehensive Survey\"", "authors": "Juhee Kim, Xiaoyuan Liu, Zhun Wang, Shi Qiu, Bo Li, Wenbo Guo, Dawn Song", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2603.11088", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-03-11", "updated_at": "2026-03-11", "status": "queued", "relevance": "high", "topics": [ "agent-safety", "tool-use", "workflow-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CR", "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "14", "collection_queries": "agent-safety" } }, { "collection": "papers", "path": "papers/items/2026-2603-11890-quare-quality-aware-requirements-analysis-through-multi-agent-dialectical-negoti.md", "title": "\"QUARE: Quality-Aware Requirements Analysis through Multi-Agent Dialectical Negotiation\"", "type": "paper", "meta": { "type": "paper", "title": "\"QUARE: Quality-Aware Requirements Analysis through Multi-Agent Dialectical Negotiation\"", "authors": "Haowei Cheng, Milhan Kim, Foutse Khomh, Teeradaj Racharak, Nobukazu Yoshioka, Naoyasu Ubayashi, Hironori Washizaki", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2603.11890", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-03-12", "updated_at": "2026-06-05", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "multi-agent", "rag" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.SE" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "15", "collection_queries": "agent-safety" } }, { "collection": "papers", "path": "papers/items/2026-2603-15309-cctu-a-benchmark-for-tool-use-under-complex-constraints.md", "title": "\"CCTU: A Benchmark for Tool Use under Complex Constraints\"", "type": "paper", "meta": { "type": "paper", "title": "\"CCTU: A Benchmark for Tool Use under Complex Constraints\"", "authors": "Junjie Ye, Guoqiang Zhang, Wenjie Fu, Tao Gui, Qi Zhang, Xuanjing Huang", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2603.15309", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-03-16", "updated_at": "2026-03-16", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "rag", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CL", "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "15", "collection_queries": "function-calling" } }, { "collection": "papers", "path": "papers/items/2026-2603-15666-compiled-memory-not-more-information-but-more-precise-instructions-for-language-.md", "title": "\"Compiled Memory: Not More Information, but More Precise Instructions for Language Agents\"", "type": "paper", "meta": { "type": "paper", "title": "\"Compiled Memory: Not More Information, but More Precise Instructions for Language Agents\"", "authors": "James Rhodes, George Kang", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2603.15666", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-03-12", "updated_at": "2026-03-12", "status": "skimmed", "relevance": "high", "topics": [ "memory", "rag" ], "methods": [ "verified-experience-distillation", "instruction-rewriting", "promotion-gate" ], "benchmarks": [ "CUAD", "HotpotQA" ], "models": [], "datasets": [ "cs.AI" ], "related_concepts": [ "procedural-memory", "prompt-compilation" ], "related_jobs": [], "related_experiments": [ "KC-001-agent-memory-pilot" ], "related_projects": [ "learning/agent-memory" ], "collection_score": "13", "collection_queries": "language-agent" } }, { "collection": "papers", "path": "papers/items/2026-2603-16734-differential-harm-propensity-in-personalized-llm-agents-the-curious-case-of-ment.md", "title": "\"Differential Harm Propensity in Personalized LLM Agents: The Curious Case of Mental Health Disclosure\"", "type": "paper", "meta": { "type": "paper", "title": "\"Differential Harm Propensity in Personalized LLM Agents: The Curious Case of Mental Health Disclosure\"", "authors": "Caglar Yildirim", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2603.16734", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-03-17", "updated_at": "2026-03-17", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "memory", "rag", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "20", "collection_queries": "agent-safety" } }, { "collection": "papers", "path": "papers/items/2026-2603-17392-agentic-cognitive-profiling-realigning-automated-alzheimer-s-disease-detection-w.md", "title": "\"Agentic Cognitive Profiling: Realigning Automated Alzheimer's Disease Detection with Clinical Construct Validity\"", "type": "paper", "meta": { "type": "paper", "title": "\"Agentic Cognitive Profiling: Realigning Automated Alzheimer's Disease Detection with Clinical Construct Validity\"", "authors": "Jiawen Kang, Kun Li, Dongrui Han, Jinchao Li, Junan Li, Lingwei Meng, Xixin Wu, Helen Meng", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2603.17392", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-03-18", "updated_at": "2026-03-18", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.MA", "cs.IR", "q-bio.NC" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "13", "collection_queries": "function-calling" } }, { "collection": "papers", "path": "papers/items/2026-2603-18245-who-tests-the-testers-systematic-enumeration-and-coverage-audit-of-llm-agent-too.md", "title": "Who Tests the Testers? Systematic Enumeration and Coverage Audit of LLM Agent Tool Call Safety", "type": "paper", "meta": { "type": "paper", "title": "Who Tests the Testers? Systematic Enumeration and Coverage Audit of LLM Agent Tool Call Safety", "authors": "Xuan Chen, Lu Yan, Ruqi Zhang, Xiangyu Zhang", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2603.18245", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-03-18", "updated_at": "2026-03-18", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "rag", "tool-use", "workflow-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.SE", "cs.CR" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "18", "collection_queries": "agent-safety" } }, { "collection": "papers", "path": "papers/items/2026-2603-19469-a-framework-for-formalizing-llm-agent-security.md", "title": "A Framework for Formalizing LLM Agent Security", "type": "paper", "meta": { "type": "paper", "title": "A Framework for Formalizing LLM Agent Security", "authors": "Vincent Siu, Jingxuan He, Kyle Montgomery, Zhun Wang, Neil Gong, Chenguang Wang, Dawn Song", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2603.19469", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-03-19", "updated_at": "2026-03-19", "status": "queued", "relevance": "high", "topics": [ "agent-safety", "memory", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CR", "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "14", "collection_queries": "agent-safety" } }, { "collection": "papers", "path": "papers/items/2026-2603-19684-tsegagent-zero-shot-tooth-segmentation-via-geometry-aware-vision-language-agents.md", "title": "\"TSegAgent: Zero-Shot Tooth Segmentation via Geometry-Aware Vision-Language Agents\"", "type": "paper", "meta": { "type": "paper", "title": "\"TSegAgent: Zero-Shot Tooth Segmentation via Geometry-Aware Vision-Language Agents\"", "authors": "Shaojie Zhuang, Lu Yin, Guangshun Wei, Yunpeng Li, Xilu Wang, Yuanfeng Zhou", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2603.19684", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-03-20", "updated_at": "2026-06-23", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "coding-agent", "rag", "reasoning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CV" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "13", "collection_queries": "language-agent" } }, { "collection": "papers", "path": "papers/items/2026-2603-21357-agenther-hindsight-experience-replay-for-llm-agent-trajectory-relabeling.md", "title": "\"AgentHER: Hindsight Experience Replay for LLM Agent Trajectory Relabeling\"", "type": "paper", "meta": { "type": "paper", "title": "\"AgentHER: Hindsight Experience Replay for LLM Agent Trajectory Relabeling\"", "authors": "Liang Ding", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2603.21357", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-03-22", "updated_at": "2026-05-10", "status": "queued", "relevance": "high", "topics": [ "computer-use", "memory", "tool-use", "workflow-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI", "cs.CL" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "16", "collection_queries": "language-agent" } }, { "collection": "papers", "path": "papers/items/2026-2603-21564-toward-a-theory-of-hierarchical-memory-for-language-agents.md", "title": "Toward a Theory of Hierarchical Memory for Language Agents", "type": "paper", "meta": { "type": "paper", "title": "Toward a Theory of Hierarchical Memory for Language Agents", "authors": "Yashar Talebirad, Ali Parsaee, Csongor Y. Szepesvari, Amirhossein Nadiri, Osmar Zaiane", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2603.21564", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-03-23", "updated_at": "2026-03-23", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "memory", "rag", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.IR", "cs.AI", "cs.IT", "cs.SI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "16", "collection_queries": "language-agent" } }, { "collection": "papers", "path": "papers/items/2026-2603-24257-memory-augmented-vision-language-agents-for-persistent-and-semantically-consiste.md", "title": "Memory-Augmented Vision-Language Agents for Persistent and Semantically Consistent Object Captioning", "type": "paper", "meta": { "type": "paper", "title": "Memory-Augmented Vision-Language Agents for Persistent and Semantically Consistent Object Captioning", "authors": "Tommaso Galliena, Stefano Rosa, Tommaso Apicella, Pietro Morerio, Alessio Del Bue, Lorenzo Natale", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2603.24257", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-03-25", "updated_at": "2026-03-30", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "embodied-agent", "memory" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CV" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "17", "collection_queries": "language-agent" } }, { "collection": "papers", "path": "papers/items/2026-2603-25353-safeguard-asf-sr-agentic-humanoid-robot-system-for-autonomous-industrial-safety.md", "title": "\"SafeGuard ASF: SR Agentic Humanoid Robot System for Autonomous Industrial Safety\"", "type": "paper", "meta": { "type": "paper", "title": "\"SafeGuard ASF: SR Agentic Humanoid Robot System for Autonomous Industrial Safety\"", "authors": "Thanh Nguyen Canh, Thang Tran Viet, Thanh Tuan Tran, Ben Wei Lim", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2603.25353", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-03-26", "updated_at": "2026-03-26", "status": "queued", "relevance": "high", "topics": [ "agent-safety", "embodied-agent", "reasoning", "tool-use", "world-model" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.RO" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "13", "collection_queries": "agent-safety" } }, { "collection": "papers", "path": "papers/items/2026-2603-27148-safetydrift-predicting-when-ai-agents-cross-the-line-before-they-actually-do.md", "title": "\"SafetyDrift: Predicting When AI Agents Cross the Line Before They Actually Do\"", "type": "paper", "meta": { "type": "paper", "title": "\"SafetyDrift: Predicting When AI Agents Cross the Line Before They Actually Do\"", "authors": "Aditya Dhodapkar, Farhaan Pishori", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2603.27148", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-03-28", "updated_at": "2026-03-28", "status": "queued", "relevance": "high", "topics": [ "agent-safety", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CR", "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "14", "collection_queries": "agent-safety" } }, { "collection": "papers", "path": "papers/items/2026-2603-28166-evaluating-privilege-usage-of-agents-with-real-world-tools.md", "title": "Evaluating Privilege Usage of Agents with Real-World Tools", "type": "paper", "meta": { "type": "paper", "title": "Evaluating Privilege Usage of Agents with Real-World Tools", "authors": "Quan Zhang, Lianhang Fu, Lvsi Lian, Gwihwan Go, Yujue Wang, Chijin Zhou, Yu Jiang, Geguang Pu", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2603.28166", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-03-30", "updated_at": "2026-04-20", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "rag", "tool-use", "workflow-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CR", "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "16", "collection_queries": "agent-safety" } }, { "collection": "papers", "path": "papers/items/2026-2603-28428-synergy-a-next-generation-general-purpose-agent-for-open-agentic-web.md", "title": "\"Synergy: A Next-Generation General-Purpose Agent for Open Agentic Web\"", "type": "paper", "meta": { "type": "paper", "title": "\"Synergy: A Next-Generation General-Purpose Agent for Open Agentic Web\"", "authors": "Xiaohang Nie, Zihan Guo, Kezhuo Yang, Zhichong Zheng, Bochen Ge, Shuai Pan, Zeyi Chen, Youling Xiang, et al.", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2603.28428", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-03-30", "updated_at": "2026-03-30", "status": "queued", "relevance": "high", "topics": [ "coding-agent", "embodied-agent", "memory", "multi-agent", "rag", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CY", "cs.MA" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "16", "collection_queries": "function-calling" } }, { "collection": "papers", "path": "papers/items/2026-2603-28900-robust-multi-agent-reinforcement-learning-for-small-uas-separation-assurance-und.md", "title": "Robust Multi-Agent Reinforcement Learning for Small UAS Separation Assurance under GPS Degradation and Spoofing", "type": "paper", "meta": { "type": "paper", "title": "Robust Multi-Agent Reinforcement Learning for Small UAS Separation Assurance under GPS Degradation and Spoofing", "authors": "Alex Zongo, Filippos Fotiadis, Ufuk Topcu, Peng Wei", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2603.28900", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-03-30", "updated_at": "2026-03-30", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "multi-agent", "world-model" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.RO", "cs.AI", "cs.LG", "eess.SY" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "13", "collection_queries": "agent-safety" } }, { "collection": "papers", "path": "papers/items/2026-2604-02022-atbench-a-diverse-and-realistic-agent-trajectory-benchmark-for-safety-evaluation.md", "title": "\"ATBench: A Diverse and Realistic Agent Trajectory Benchmark for Safety Evaluation and Diagnosis\"", "type": "paper", "meta": { "type": "paper", "title": "\"ATBench: A Diverse and Realistic Agent Trajectory Benchmark for Safety Evaluation and Diagnosis\"", "authors": "Yu Li, Haoyu Luo, Yuejin Xie, Yuqian Fu, Zhonghao Yang, Shuai Shao, Qihan Ren, Wanying Qu, et al.", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2604.02022", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-04-02", "updated_at": "2026-05-13", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "planning", "rag", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "20", "collection_queries": "agent-safety" } }, { "collection": "papers", "path": "papers/items/2026-2604-02155-brief-is-better-non-monotonic-chain-of-thought-budget-effects-in-function-callin.md", "title": "\"Brief Is Better: Non-Monotonic Chain-of-Thought Budget Effects in Function-Calling Language Agents\"", "type": "paper", "meta": { "type": "paper", "title": "\"Brief Is Better: Non-Monotonic Chain-of-Thought Budget Effects in Function-Calling Language Agents\"", "authors": "Xuan Qi", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2604.02155", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-04-02", "updated_at": "2026-04-02", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "rag", "reasoning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CL" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "14", "collection_queries": "function-calling, language-agent" } }, { "collection": "papers", "path": "papers/items/2026-2604-03098-co-evolution-of-policy-and-internal-reward-for-language-agents.md", "title": "Co-Evolution of Policy and Internal Reward for Language Agents", "type": "paper", "meta": { "type": "paper", "title": "Co-Evolution of Policy and Internal Reward for Language Agents", "authors": "Xinyu Wang, Hanwei Wu, Jingwei Song, Shuyuan Zhang, Jiayi Zhang, Fanqi Kong, Tung Sum Thomas Kwok, Xiao-Wen Chang, et al.", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2604.03098", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-04-03", "updated_at": "2026-04-03", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "computer-use", "planning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.LG", "cs.AI", "cs.CL" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "15", "collection_queries": "language-agent" } }, { "collection": "papers", "path": "papers/items/2026-2604-03242-draft-task-decoupled-latent-reasoning-for-agent-safety.md", "title": "\"DRAFT: Task Decoupled Latent Reasoning for Agent Safety\"", "type": "paper", "meta": { "type": "paper", "title": "\"DRAFT: Task Decoupled Latent Reasoning for Agent Safety\"", "authors": "Lin Wang, Junfeng Fang, Dan Zhang, Fei Shen, Xiang Wang, Tat-Seng Chua", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2604.03242", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-02-11", "updated_at": "2026-02-11", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "rag", "reasoning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.LG" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "17", "collection_queries": "agent-safety" } }, { "collection": "papers", "path": "papers/items/2026-2604-04131-profile-then-reason-bounded-semantic-complexity-for-tool-augmented-language-agen.md", "title": "\"Profile-Then-Reason: Bounded Semantic Complexity for Tool-Augmented Language Agents\"", "type": "paper", "meta": { "type": "paper", "title": "\"Profile-Then-Reason: Bounded Semantic Complexity for Tool-Augmented Language Agents\"", "authors": "Paulo Akira F. Enabe", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2604.04131", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-04-05", "updated_at": "2026-04-05", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "rag", "reasoning", "tool-use", "workflow-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "16", "collection_queries": "language-agent" } }, { "collection": "papers", "path": "papers/items/2026-2604-04426-shieldnet-network-level-guardrails-against-emerging-supply-chain-injections-in-a.md", "title": "\"ShieldNet: Network-Level Guardrails against Emerging Supply-Chain Injections in Agentic Systems\"", "type": "paper", "meta": { "type": "paper", "title": "\"ShieldNet: Network-Level Guardrails against Emerging Supply-Chain Injections in Agentic Systems\"", "authors": "Zhuowen Yuan, Zhaorun Chen, Zhen Xiang, Nathaniel D. Bastian, Seyyed Hadi Hashemi, Chaowei Xiao, Wenbo Guo, Bo Li", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2604.04426", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-04-06", "updated_at": "2026-04-06", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "15", "collection_queries": "agent-safety" } }, { "collection": "papers", "path": "papers/items/2026-2604-06762-arulecon-agentic-security-rule-conversion.md", "title": "\"ARuleCon: Agentic Security Rule Conversion\"", "type": "paper", "meta": { "type": "paper", "title": "\"ARuleCon: Agentic Security Rule Conversion\"", "authors": "Ming Xu, Hongtai Wang, Yanpei Guo, Zhengmin Yu, Weili Han, Hoon Wei Lim, Jin Song Dong, Jiaheng Zhang", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2604.06762", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-04-08", "updated_at": "2026-04-08", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "rag" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CR" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "14", "collection_queries": "agent-safety" } }, { "collection": "papers", "path": "papers/items/2026-2604-06972-differentiable-environment-trajectory-co-optimization-for-safe-multi-agent-navig.md", "title": "Differentiable Environment-Trajectory Co-Optimization for Safe Multi-Agent Navigation", "type": "paper", "meta": { "type": "paper", "title": "Differentiable Environment-Trajectory Co-Optimization for Safe Multi-Agent Navigation", "authors": "Zhan Gao, Gabriele Fadini, Stelian Coros, Amanda Prorok", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2604.06972", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-04-08", "updated_at": "2026-04-08", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "computer-use", "embodied-agent", "multi-agent", "rag", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.RO", "cs.MA" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "15", "collection_queries": "agent-safety" } }, { "collection": "papers", "path": "papers/items/2026-2604-08388-awakening-the-sleeping-agent-lean-specific-agentic-data-reactivates-general-tool.md", "title": "\"Awakening the Sleeping Agent: Lean-Specific Agentic Data Reactivates General Tool Use in Goedel Prover\"", "type": "paper", "meta": { "type": "paper", "title": "\"Awakening the Sleeping Agent: Lean-Specific Agentic Data Reactivates General Tool Use in Goedel Prover\"", "authors": "Jui-Hui Chung, Hongzhou Lin, Lai Jiang, Shange Tang, Chi Jin", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2604.08388", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-04-09", "updated_at": "2026-04-09", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "14", "collection_queries": "function-calling" } }, { "collection": "papers", "path": "papers/items/2026-2604-10577-the-blind-spot-of-agent-safety-how-benign-user-instructions-expose-critical-vuln.md", "title": "\"The Blind Spot of Agent Safety: How Benign User Instructions Expose Critical Vulnerabilities in Computer-Use Agents\"", "type": "paper", "meta": { "type": "paper", "title": "\"The Blind Spot of Agent Safety: How Benign User Instructions Expose Critical Vulnerabilities in Computer-Use Agents\"", "authors": "Xuwei Ding, Skylar Zhai, Linxin Song, Jiate Li, Taiwei Shi, Nicholas Meade, Siva Reddy, Jian Kang, et al.", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2604.10577", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-04-12", "updated_at": "2026-04-17", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "multi-agent", "rag", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CR", "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "17", "collection_queries": "agent-safety" } }, { "collection": "papers", "path": "papers/items/2026-2604-11557-unitoolcall-unifying-tool-use-representation-data-and-evaluation-for-llm-agents.md", "title": "\"UniToolCall: Unifying Tool-Use Representation, Data, and Evaluation for LLM Agents\"", "type": "paper", "meta": { "type": "paper", "title": "\"UniToolCall: Unifying Tool-Use Representation, Data, and Evaluation for LLM Agents\"", "authors": "Yijuan Liang, Xinghao Chen, Yifan Ge, Ziyi Wu, Hao Wu, Changyu Zeng, Wei Xing, Xiaoyu Shen", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2604.11557", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-04-13", "updated_at": "2026-05-25", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "reasoning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "17", "collection_queries": "function-calling" } }, { "collection": "papers", "path": "papers/items/2026-2604-12986-parallax-why-ai-agents-that-think-must-never-act.md", "title": "\"Parallax: Why AI Agents That Think Must Never Act\"", "type": "paper", "meta": { "type": "paper", "title": "\"Parallax: Why AI Agents That Think Must Never Act\"", "authors": "Joel Fokou", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2604.12986", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-04-14", "updated_at": "2026-04-14", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "reasoning", "tool-use", "workflow-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CR", "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "15", "collection_queries": "agent-safety" } }, { "collection": "papers", "path": "papers/items/2026-2604-13298-can-agents-secure-hardware-evaluating-agentic-llm-driven-obfuscation-for-ip-prot.md", "title": "Can Agents Secure Hardware? Evaluating Agentic LLM-Driven Obfuscation for IP Protection", "type": "paper", "meta": { "type": "paper", "title": "Can Agents Secure Hardware? Evaluating Agentic LLM-Driven Obfuscation for IP Protection", "authors": "Sujan Ghimire, Parsa Mirfasihi, Muhtasim Alam Chowdhury, Veeramani Pugazhenthi, Harish Kumar Dharavath, Farshad Firouzi, Rozhin Yasaei, Pratik Satam, et al.", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2604.13298", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-04-14", "updated_at": "2026-04-14", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "planning", "rag" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CR" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "16", "collection_queries": "agent-safety" } }, { "collection": "papers", "path": "papers/items/2026-2604-13536-don-t-let-ai-agents-yolo-your-files-shifting-information-and-control-to-filesyst.md", "title": "\"Don't Let AI Agents YOLO Your Files: Shifting Information and Control to Filesystems for Agent Safety and Autonomy\"", "type": "paper", "meta": { "type": "paper", "title": "\"Don't Let AI Agents YOLO Your Files: Shifting Information and Control to Filesystems for Agent Safety and Autonomy\"", "authors": "Shawn Wanxiang Zhong, Junxuan Liao, Jing Liu, Mai Zheng, Andrea C. Arpaci-Dusseau, Remzi H. Arpaci-Dusseau", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2604.13536", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-04-15", "updated_at": "2026-04-16", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "coding-agent", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.OS" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "16", "collection_queries": "agent-safety" } }, { "collection": "papers", "path": "papers/items/2026-2604-13954-hintbench-horizon-agent-intrinsic-non-attack-trajectory-benchmark.md", "title": "\"HINTBench: Horizon-agent Intrinsic Non-attack Trajectory Benchmark\"", "type": "paper", "meta": { "type": "paper", "title": "\"HINTBench: Horizon-agent Intrinsic Non-attack Trajectory Benchmark\"", "authors": "Jiacheng Wang, Jinchang Hou, Fabian Wang, Ping Jian, Chenfu Bao, Zhonghou Lv", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2604.13954", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-04-15", "updated_at": "2026-04-15", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "planning", "rag" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.LG", "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "16", "collection_queries": "agent-safety" } }, { "collection": "papers", "path": "papers/items/2026-2604-14399-spacemind-a-modular-and-self-evolving-embodied-vision-language-agent-framework-f.md", "title": "\"SpaceMind: A Modular and Self-Evolving Embodied Vision-Language Agent Framework for Autonomous On-orbit Servicing\"", "type": "paper", "meta": { "type": "paper", "title": "\"SpaceMind: A Modular and Self-Evolving Embodied Vision-Language Agent Framework for Autonomous On-orbit Servicing\"", "authors": "Aodi Wu, Haodong Han, Xubo Luo, Ruisuo Wang, Shan He, Xue Wan", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2604.14399", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-04-15", "updated_at": "2026-04-15", "status": "queued", "relevance": "high", "topics": [ "embodied-agent", "reasoning", "tool-use", "world-model" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.RO", "cs.AI", "eess.SY" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "14", "collection_queries": "language-agent" } }, { "collection": "papers", "path": "papers/items/2026-2604-15415-harmfulskillbench-how-do-harmful-skills-weaponize-your-agents.md", "title": "\"HarmfulSkillBench: How Do Harmful Skills Weaponize Your Agents?\"", "type": "paper", "meta": { "type": "paper", "title": "\"HarmfulSkillBench: How Do Harmful Skills Weaponize Your Agents?\"", "authors": "Yukun Jiang, Yage Zhang, Michael Backes, Xinyue Shen, Yang Zhang", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2604.15415", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-04-16", "updated_at": "2026-04-16", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "rag", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CR", "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "14", "collection_queries": "agent-safety" } }, { "collection": "papers", "path": "papers/items/2026-2604-15579-don-t-make-models-guess-security-and-safety-symbolic-guardrails-for-domain-speci.md", "title": "\"Don't Make Models Guess Security and Safety: Symbolic Guardrails for Domain-Specific AI Agents\"", "type": "paper", "meta": { "type": "paper", "title": "\"Don't Make Models Guess Security and Safety: Symbolic Guardrails for Domain-Specific AI Agents\"", "authors": "Yining Hong, Yining She, Eunsuk Kang, Christopher S. Timperley, Christian Kästner", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2604.15579", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-04-16", "updated_at": "2026-07-05", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "coding-agent", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.SE", "cs.AI", "cs.CR" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "14", "collection_queries": "agent-safety" } }, { "collection": "papers", "path": "papers/items/2026-2604-16706-evaluating-tool-using-language-agents-judge-reliability-propagation-cascades-and.md", "title": "\"Evaluating Tool-Using Language Agents: Judge Reliability, Propagation Cascades, and Runtime Mitigation in AgentProp-Bench\"", "type": "paper", "meta": { "type": "paper", "title": "\"Evaluating Tool-Using Language Agents: Judge Reliability, Propagation Cascades, and Runtime Mitigation in AgentProp-Bench\"", "authors": "Bhaskar Gurram", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2604.16706", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-04-17", "updated_at": "2026-04-17", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI", "cs.CL", "cs.MA" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "14", "collection_queries": "language-agent" } }, { "collection": "papers", "path": "papers/items/2026-2604-17562-safeagent-a-runtime-protection-architecture-for-agentic-systems.md", "title": "\"SafeAgent: A Runtime Protection Architecture for Agentic Systems\"", "type": "paper", "meta": { "type": "paper", "title": "\"SafeAgent: A Runtime Protection Architecture for Agentic Systems\"", "authors": "Hailin Liu, Eugene Ilyushin, Jie Ni, Min Zhu", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2604.17562", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-04-19", "updated_at": "2026-04-19", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "coding-agent", "memory", "reasoning", "tool-use", "workflow-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI", "cs.MA" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "19", "collection_queries": "agent-safety" } }, { "collection": "papers", "path": "papers/items/2026-2604-18658-owner-harm-a-missing-threat-model-for-ai-agent-safety.md", "title": "\"Owner-Harm: A Missing Threat Model for AI Agent Safety\"", "type": "paper", "meta": { "type": "paper", "title": "\"Owner-Harm: A Missing Threat Model for AI Agent Safety\"", "authors": "Dongcheng Zhang, Yiqing Jiang", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2604.18658", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-04-20", "updated_at": "2026-04-20", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "rag", "reasoning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CR", "cs.AI", "cs.CL" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "19", "collection_queries": "agent-safety" } }, { "collection": "papers", "path": "papers/items/2026-2604-18718-towards-optimal-agentic-architectures-for-offensive-security-tasks.md", "title": "Towards Optimal Agentic Architectures for Offensive Security Tasks", "type": "paper", "meta": { "type": "paper", "title": "Towards Optimal Agentic Architectures for Offensive Security Tasks", "authors": "Isaac David, Arthur Gervais", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2604.18718", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-04-20", "updated_at": "2026-04-20", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "rag", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CR", "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "15", "collection_queries": "agent-safety" } }, { "collection": "papers", "path": "papers/items/2026-2604-18847-human-guided-harm-recovery-for-computer-use-agents.md", "title": "Human-Guided Harm Recovery for Computer Use Agents", "type": "paper", "meta": { "type": "paper", "title": "Human-Guided Harm Recovery for Computer Use Agents", "authors": "Christy Li, Sky CH-Wang, Andi Peng, Andreea Bobu", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2604.18847", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-04-20", "updated_at": "2026-05-28", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "computer-use", "memory", "planning", "rag", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI", "cs.CL" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "18", "collection_queries": "agent-safety" } }, { "collection": "papers", "path": "papers/items/2026-2604-19821-jtpro-a-joint-tool-prompt-reflective-optimization-framework-for-language-agents.md", "title": "\"JTPRO: A Joint Tool-Prompt Reflective Optimization Framework for Language Agents\"", "type": "paper", "meta": { "type": "paper", "title": "\"JTPRO: A Joint Tool-Prompt Reflective Optimization Framework for Language Agents\"", "authors": "Sandip Ghoshal, Anshul Mittal, Jyotika Singh, Miguel Ballesteros, Weiyi Sun, Fang Tu, Shailender Singh, Yassine Benajiba, et al.", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2604.19821", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-04-20", "updated_at": "2026-04-20", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "computer-use", "reasoning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI", "cs.SE" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "15", "collection_queries": "language-agent" } }, { "collection": "papers", "path": "papers/items/2026-2604-19844-if-you-re-waiting-for-a-sign-that-might-not-be-it-mitigating-trust-boundary-conf.md", "title": "\"If you're waiting for a sign... that might not be it! Mitigating Trust Boundary Confusion from Visual Injections on Vision-Language Agentic Systems\"", "type": "paper", "meta": { "type": "paper", "title": "\"If you're waiting for a sign... that might not be it! Mitigating Trust Boundary Confusion from Visual Injections on Vision-Language Agentic Systems\"", "authors": "Jiamin Chang, Minhui Xue, Ruoxi Sun, Shuchao Pang, Salil S. Kanhere, Hammond Pearce", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2604.19844", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-04-21", "updated_at": "2026-04-21", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "embodied-agent", "multi-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CV", "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "19", "collection_queries": "language-agent" } }, { "collection": "papers", "path": "papers/items/2026-2604-20994-breaking-mcp-with-function-hijacking-attacks-novel-threats-for-function-calling-.md", "title": "\"Breaking MCP with Function Hijacking Attacks: Novel Threats for Function Calling and Agentic Models\"", "type": "paper", "meta": { "type": "paper", "title": "\"Breaking MCP with Function Hijacking Attacks: Novel Threats for Function Calling and Agentic Models\"", "authors": "Yannis Belkhiter, Giulio Zizzo, Sergio Maffeis, Seshu Tirupathi, John D. Kelleher", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2604.20994", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-04-22", "updated_at": "2026-04-22", "status": "queued", "relevance": "high", "topics": [ "agent-safety", "reasoning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CR", "cs.AI", "cs.CL" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "13", "collection_queries": "function-calling" } }, { "collection": "papers", "path": "papers/items/2026-2604-21190-spatio-adaptive-test-time-orchestration-of-vision-language-agents-for-spatial-re.md", "title": "\"SpatiO: Adaptive Test-Time Orchestration of Vision-Language Agents for Spatial Reasoning\"", "type": "paper", "meta": { "type": "paper", "title": "\"SpatiO: Adaptive Test-Time Orchestration of Vision-Language Agents for Spatial Reasoning\"", "authors": "Chan Yeong Hwang, Miso Choi, Sunghyun On, Jinkyu Kim, Jungbeom Lee", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2604.21190", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-04-23", "updated_at": "2026-06-27", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "multi-agent", "rag", "reasoning" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CV" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "14", "collection_queries": "language-agent" } }, { "collection": "papers", "path": "papers/items/2026-2604-22879-beyond-single-agent-alignment-preventing-context-fragmented-violations-in-multi-.md", "title": "\"Beyond Single-Agent Alignment: Preventing Context-Fragmented Violations in Multi-Agent Systems\"", "type": "paper", "meta": { "type": "paper", "title": "\"Beyond Single-Agent Alignment: Preventing Context-Fragmented Violations in Multi-Agent Systems\"", "authors": "Jie Wu, Ming Gong", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2604.22879", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-04-24", "updated_at": "2026-04-24", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "multi-agent", "rag", "tool-use", "workflow-agent", "world-model" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.MA", "cs.AI", "cs.CR", "cs.LG" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "18", "collection_queries": "agent-safety" } }, { "collection": "papers", "path": "papers/items/2026-2604-23210-discovering-agentic-safety-specifications-from-1-bit-danger-signals.md", "title": "Discovering Agentic Safety Specifications from 1-Bit Danger Signals", "type": "paper", "meta": { "type": "paper", "title": "Discovering Agentic Safety Specifications from 1-Bit Danger Signals", "authors": "Víctor Gallego", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2604.23210", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-04-25", "updated_at": "2026-04-25", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "planning", "rag", "reasoning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI", "cs.CL" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "16", "collection_queries": "agent-safety" } }, { "collection": "papers", "path": "papers/items/2026-2604-23374-ghost-in-the-agent-redefining-information-flow-tracking-for-llm-agents.md", "title": "\"Ghost in the Agent: Redefining Information Flow Tracking for LLM Agents\"", "type": "paper", "meta": { "type": "paper", "title": "\"Ghost in the Agent: Redefining Information Flow Tracking for LLM Agents\"", "authors": "Yuandao Cai, Wensheng Tang, Cheng Wen, Shengchao Qin", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2604.23374", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-04-25", "updated_at": "2026-04-25", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "memory", "reasoning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CR" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "18", "collection_queries": "agent-safety" } }, { "collection": "papers", "path": "papers/items/2026-2604-23459-architecture-matters-for-multi-agent-security.md", "title": "Architecture Matters for Multi-Agent Security", "type": "paper", "meta": { "type": "paper", "title": "Architecture Matters for Multi-Agent Security", "authors": "Ben Hagag, William L. Anderson, Christian Schroeder de Witt, Sarah Scheffler", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2604.23459", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-04-25", "updated_at": "2026-04-25", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "computer-use", "memory", "multi-agent", "planning" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.MA", "cs.CR", "cs.LG" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "18", "collection_queries": "agent-safety" } }, { "collection": "papers", "path": "papers/items/2026-2604-24212-empowering-autonomous-debugging-agents-with-efficient-dynamic-analysis.md", "title": "Empowering Autonomous Debugging Agents with Efficient Dynamic Analysis", "type": "paper", "meta": { "type": "paper", "title": "Empowering Autonomous Debugging Agents with Efficient Dynamic Analysis", "authors": "Jiahong Xiang, Xiaoyang Xu, Xiaopan Chu, Hongliang Tian, Yuqun Zhang", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2604.24212", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-04-27", "updated_at": "2026-04-27", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "coding-agent", "embodied-agent", "memory", "rag", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.SE" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "16", "collection_queries": "autonomous-agent-llm" } }, { "collection": "papers", "path": "papers/items/2026-2604-24826-a-comparative-evaluation-of-ai-agent-security-guardrails.md", "title": "A Comparative Evaluation of AI Agent Security Guardrails", "type": "paper", "meta": { "type": "paper", "title": "A Comparative Evaluation of AI Agent Security Guardrails", "authors": "Qi Li, Jiu Li, Pingtao Wei, Jianjun Xu, Xueyi Wei, Jiwei Shi, Xuan Zhang, Yanhui Yang, et al.", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2604.24826", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-04-27", "updated_at": "2026-04-27", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CR", "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "15", "collection_queries": "agent-safety" } }, { "collection": "papers", "path": "papers/items/2026-2604-25135-fama-failure-aware-meta-agentic-framework-for-open-source-llms-in-interactive-to.md", "title": "\"FAMA: Failure-Aware Meta-Agentic Framework for Open-Source LLMs in Interactive Tool Use Environments\"", "type": "paper", "meta": { "type": "paper", "title": "\"FAMA: Failure-Aware Meta-Agentic Framework for Open-Source LLMs in Interactive Tool Use Environments\"", "authors": "Amir Saeidi, Venkatesh Mishra, Souradeep Mukhopadhyay, Gaowen Liu, Ali Payani, Jayanth Srinivasa, Chitta Baral", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2604.25135", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-04-28", "updated_at": "2026-04-28", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CL" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "18", "collection_queries": "autonomous-agent-llm" } }, { "collection": "papers", "path": "papers/items/2026-2604-25318-cutscene-agent-an-llm-agent-framework-for-automated-3d-cutscene-generation.md", "title": "\"Cutscene Agent: An LLM Agent Framework for Automated 3D Cutscene Generation\"", "type": "paper", "meta": { "type": "paper", "title": "\"Cutscene Agent: An LLM Agent Framework for Automated 3D Cutscene Generation\"", "authors": "Lanshan He, Haozhou Pang, Qi Gan, Xin Shen, Ziwei Zhang, Yibo Liu, Gang Fang, Bo Liu, et al.", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2604.25318", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-04-28", "updated_at": "2026-04-28", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "multi-agent", "planning", "reasoning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.GR", "cs.AI", "cs.CL" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "18", "collection_queries": "function-calling" } }, { "collection": "papers", "path": "papers/items/2026-2604-25555-from-crud-to-autonomous-agents-formal-validation-and-zero-trust-security-for-sem.md", "title": "\"From CRUD to Autonomous Agents: Formal Validation and Zero-Trust Security for Semantic Gateways in AI-Native Enterprise Systems\"", "type": "paper", "meta": { "type": "paper", "title": "\"From CRUD to Autonomous Agents: Formal Validation and Zero-Trust Security for Semantic Gateways in AI-Native Enterprise Systems\"", "authors": "Ignacio Peyrano", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2604.25555", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-04-28", "updated_at": "2026-04-28", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "coding-agent", "tool-use", "workflow-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CR", "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "14", "collection_queries": "autonomous-agent-llm" } }, { "collection": "papers", "path": "papers/items/2026-2604-26274-enforcing-benign-trajectories-a-behavioral-firewall-for-structured-workflow-ai-a.md", "title": "\"Enforcing Benign Trajectories: A Behavioral Firewall for Structured-Workflow AI Agents\"", "type": "paper", "meta": { "type": "paper", "title": "\"Enforcing Benign Trajectories: A Behavioral Firewall for Structured-Workflow AI Agents\"", "authors": "Hung Dang", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2604.26274", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-04-29", "updated_at": "2026-04-29", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "rag", "tool-use", "workflow-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CR", "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "15", "collection_queries": "agent-safety" } }, { "collection": "papers", "path": "papers/items/2026-2604-26959-careguardai-context-aware-multi-agent-guardrails-for-clinical-safety-hallucinati.md", "title": "\"CareGuardAI: Context-Aware Multi-Agent Guardrails for Clinical Safety & Hallucination Mitigation in Patient-Facing LLMs\"", "type": "paper", "meta": { "type": "paper", "title": "\"CareGuardAI: Context-Aware Multi-Agent Guardrails for Clinical Safety & Hallucination Mitigation in Patient-Facing LLMs\"", "authors": "Elham Nasarian, Abhilash Neog, Kwok-Leung Tsui, Niyousha HosseiniChimeh", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2604.26959", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-04-07", "updated_at": "2026-04-07", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "multi-agent", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CY", "cs.AI", "cs.MA" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "16", "collection_queries": "agent-safety" } }, { "collection": "papers", "path": "papers/items/2026-2604-27092-end-to-end-autonomous-scientific-discovery-on-a-real-optical-platform.md", "title": "End-to-end autonomous scientific discovery on a real optical platform", "type": "paper", "meta": { "type": "paper", "title": "End-to-end autonomous scientific discovery on a real optical platform", "authors": "Shuxing Yang, Fujia Chen, Rui Zhao, Junyao Wu, Yize Wang, Haiyao Luo, Ning Han, Qiaolu Chen, et al.", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2604.27092", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-04-29", "updated_at": "2026-04-29", "status": "queued", "relevance": "high", "topics": [ "memory", "planning", "reasoning", "tool-use", "workflow-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI", "physics.optics" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "13", "collection_queries": "autonomous-agent-llm" } }, { "collection": "papers", "path": "papers/items/2026-2604-27464-security-attack-and-defense-strategies-for-autonomous-agent-frameworks-a-layered.md", "title": "\"Security Attack and Defense Strategies for Autonomous Agent Frameworks: A Layered Review with OpenClaw as a Case Study\"", "type": "paper", "meta": { "type": "paper", "title": "\"Security Attack and Defense Strategies for Autonomous Agent Frameworks: A Layered Review with OpenClaw as a Case Study\"", "authors": "Luyao Xu, Xiang Chen", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2604.27464", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-04-30", "updated_at": "2026-04-30", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "planning", "tool-use", "workflow-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CR", "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "14", "collection_queries": "agent-safety, autonomous-agent-llm" } }, { "collection": "papers", "path": "papers/items/2026-2604-27699-bridging-values-and-behavior-a-hierarchical-framework-for-proactive-embodied-age.md", "title": "\"Bridging Values and Behavior: A Hierarchical Framework for Proactive Embodied Agents\"", "type": "paper", "meta": { "type": "paper", "title": "\"Bridging Values and Behavior: A Hierarchical Framework for Proactive Embodied Agents\"", "authors": "Chunhui Zhang, Yuxuan Wang, Aoyang Qin, Yi-Long Lu, Kunlun Wu, Yizhou Wang, Wei Wang", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2604.27699", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-04-30", "updated_at": "2026-04-30", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "embodied-agent", "memory", "planning", "reasoning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "15", "collection_queries": "autonomous-agent-llm" } }, { "collection": "papers", "path": "papers/items/2026-2604-27859-rethinking-agentic-reinforcement-learning-in-large-language-models.md", "title": "Rethinking Agentic Reinforcement Learning In Large Language Models", "type": "paper", "meta": { "type": "paper", "title": "Rethinking Agentic Reinforcement Learning In Large Language Models", "authors": "Fangming Cui, Ruixiao Zhu, Cheng Fang, Sunan Li, Jiahong Li", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2604.27859", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-04-30", "updated_at": "2026-05-15", "status": "queued", "relevance": "high", "topics": [ "memory", "planning", "reasoning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI", "cs.ET" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "14", "collection_queries": "autonomous-agent-llm" } }, { "collection": "papers", "path": "papers/items/2026-2604-28157-flashrt-towards-computationally-and-memory-efficient-red-teaming-for-prompt-inje.md", "title": "\"FlashRT: Towards Computationally and Memory Efficient Red-Teaming for Prompt Injection and Knowledge Corruption\"", "type": "paper", "meta": { "type": "paper", "title": "\"FlashRT: Towards Computationally and Memory Efficient Red-Teaming for Prompt Injection and Knowledge Corruption\"", "authors": "Yanting Wang, Chenlong Yin, Ying Chen, Jinyuan Jia", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2604.28157", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-04-30", "updated_at": "2026-04-30", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "memory", "rag", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CR" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "14", "collection_queries": "autonomous-agent-llm" } }, { "collection": "papers", "path": "papers/items/2026-2605-00081-alignment-contracts-for-agentic-security-systems.md", "title": "Alignment Contracts for Agentic Security Systems", "type": "paper", "meta": { "type": "paper", "title": "Alignment Contracts for Agentic Security Systems", "authors": "Isaac David, Marco Guarnieri, Arthur Gervais", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2605.00081", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-04-30", "updated_at": "2026-04-30", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "planning", "tool-use", "workflow-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CR", "cs.LO" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "15", "collection_queries": "agent-safety" } }, { "collection": "papers", "path": "papers/items/2026-2605-00741-self-adaptive-multi-agent-llm-based-security-pattern-selection-for-iot-systems.md", "title": "Self-Adaptive Multi-Agent LLM-Based Security Pattern Selection for IoT Systems", "type": "paper", "meta": { "type": "paper", "title": "Self-Adaptive Multi-Agent LLM-Based Security Pattern Selection for IoT Systems", "authors": "Saeid Jamshidi, Foutse Khomh, Carol Fung, Kawser Wazed Nafi", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2605.00741", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-05-01", "updated_at": "2026-05-01", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "multi-agent", "reasoning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CR" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "16", "collection_queries": "agent-safety" } }, { "collection": "papers", "path": "papers/items/2026-2605-00845-graph-query-generation-with-constraint-guided-large-language-agents.md", "title": "Graph Query Generation with Constraint-guided Large Language Agents", "type": "paper", "meta": { "type": "paper", "title": "Graph Query Generation with Constraint-guided Large Language Agents", "authors": "Mengying Wang, Nicolaas Jedema, Rahul Pandey, RaviKiran Krishnan, Jens Lehmann, Yinghui Wu", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2605.00845", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-04-09", "updated_at": "2026-04-09", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "computer-use", "reasoning", "workflow-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.DB", "cs.AI", "cs.CL" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "16", "collection_queries": "language-agent" } }, { "collection": "papers", "path": "papers/items/2026-2605-01101-virtual-speech-therapist-a-clinician-in-the-loop-ai-speech-therapy-agent-for-per.md", "title": "\"Virtual Speech Therapist: A Clinician-in-the-Loop AI Speech Therapy Agent for Personalized and Supervised Therapy\"", "type": "paper", "meta": { "type": "paper", "title": "\"Virtual Speech Therapist: A Clinician-in-the-Loop AI Speech Therapy Agent for Personalized and Supervised Therapy\"", "authors": "Shakeel Sheikh, Patrick Marmaroli, MD Sahidullah, Slim Ouni, Fabrice Hirsch, Goncalo Leal, Bjorn W Schuller", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2605.01101", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-05-01", "updated_at": "2026-06-15", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "computer-use", "multi-agent", "planning", "reasoning", "tool-use", "workflow-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI", "cs.CL", "cs.SD", "eess.AS" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "19", "collection_queries": "planning-agent" } }, { "collection": "papers", "path": "papers/items/2026-2605-01644-toward-a-principled-framework-for-agent-safety-measurement.md", "title": "Toward a Principled Framework for Agent Safety Measurement", "type": "paper", "meta": { "type": "paper", "title": "Toward a Principled Framework for Agent Safety Measurement", "authors": "Shuyi Lin, Anshuman Suri, Alina Oprea, Cheng Tan", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2605.01644", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-05-02", "updated_at": "2026-05-02", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "coding-agent", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CR" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "16", "collection_queries": "agent-safety" } }, { "collection": "papers", "path": "papers/items/2026-2605-02240-physicianbench-evaluating-llm-agents-in-real-world-ehr-environments.md", "title": "\"PhysicianBench: Evaluating LLM Agents in Real-World EHR Environments\"", "type": "paper", "meta": { "type": "paper", "title": "\"PhysicianBench: Evaluating LLM Agents in Real-World EHR Environments\"", "authors": "Ruoqi Liu, Imran Q. Mohiuddin, Austin J. Schoeffler, Kavita Renduchintala, Ashwin Nayak, Prasantha L. Vemu, Shivam C. Vedak, Kameron C. Black, et al.", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2605.02240", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-05-04", "updated_at": "2026-05-04", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "planning", "rag", "reasoning", "tool-use", "workflow-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "17", "collection_queries": "planning-agent" } }, { "collection": "papers", "path": "papers/items/2026-2605-03242-enhancing-agent-safety-judgment-controlled-benchmark-rewriting-and-analogical-re.md", "title": "\"Enhancing Agent Safety Judgment: Controlled Benchmark Rewriting and Analogical Reasoning for Deceptive Out-of-Distribution Scenarios\"", "type": "paper", "meta": { "type": "paper", "title": "\"Enhancing Agent Safety Judgment: Controlled Benchmark Rewriting and Analogical Reasoning for Deceptive Out-of-Distribution Scenarios\"", "authors": "Zuoyu Zhang, Yancheng Zhu", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2605.03242", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-05-05", "updated_at": "2026-05-05", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "computer-use", "multi-agent", "rag", "reasoning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "22", "collection_queries": "agent-safety" } }, { "collection": "papers", "path": "papers/items/2026-2605-03312-memflow-intent-driven-memory-orchestration-for-small-language-model-agents.md", "title": "\"MemFlow: Intent-Driven Memory Orchestration for Small Language Model Agents\"", "type": "paper", "meta": { "type": "paper", "title": "\"MemFlow: Intent-Driven Memory Orchestration for Small Language Model Agents\"", "authors": "Jiayi Chen, Yingcong Li, Guiling Wang", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2605.03312", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-05-05", "updated_at": "2026-05-05", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "memory", "planning", "rag", "reasoning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.MA" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "19", "collection_queries": "language-agent" } }, { "collection": "papers", "path": "papers/items/2026-2605-03328-llm-adam-a-generalizable-llm-agent-framework-for-pre-print-anomaly-detection-in-.md", "title": "\"LLM-ADAM: A Generalizable LLM Agent Framework for Pre-Print Anomaly Detection in Additive Manufacturing\"", "type": "paper", "meta": { "type": "paper", "title": "\"LLM-ADAM: A Generalizable LLM Agent Framework for Pre-Print Anomaly Detection in Additive Manufacturing\"", "authors": "Ahmadreza Eslaminia, Chuhan Cai, Cameron Smith, Ruo-Syuan Mei, Shichen Li, Rajiv Malhotra, Klara Nahrstedt, Chenhui Shao", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2605.03328", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-05-05", "updated_at": "2026-05-05", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "computer-use", "planning" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.LG", "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "13", "collection_queries": "planning-agent" } }, { "collection": "papers", "path": "papers/items/2026-2605-03505-lats-rca-language-agent-tree-search-for-root-cause-analysis-in-microservices.md", "title": "\"LATS-RCA: Language Agent Tree Search for Root Cause Analysis in Microservices\"", "type": "paper", "meta": { "type": "paper", "title": "\"LATS-RCA: Language Agent Tree Search for Root Cause Analysis in Microservices\"", "authors": "Alexander Naakka, Yuqing Wang, Mika V Mäntylä", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2605.03505", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-05-05", "updated_at": "2026-06-12", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "computer-use", "rag", "reasoning" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.SE" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "15", "collection_queries": "language-agent" } }, { "collection": "papers", "path": "papers/items/2026-2605-04107-tscg-deterministic-tool-schema-compilation-for-agentic-llm-deployments.md", "title": "\"TSCG: Deterministic Tool-Schema Compilation for Agentic LLM Deployments\"", "type": "paper", "meta": { "type": "paper", "title": "\"TSCG: Deterministic Tool-Schema Compilation for Agentic LLM Deployments\"", "authors": "Furkan Sakizli", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2605.04107", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-05-04", "updated_at": "2026-05-04", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "computer-use", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.SE", "cs.AI", "cs.CL" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "15", "collection_queries": "function-calling" } }, { "collection": "papers", "path": "papers/items/2026-2605-04808-decodingtrust-agent-platform-dtap-a-controllable-and-interactive-red-teaming-pla.md", "title": "\"DecodingTrust-Agent Platform (DTap): A Controllable and Interactive Red-Teaming Platform for AI Agents\"", "type": "paper", "meta": { "type": "paper", "title": "\"DecodingTrust-Agent Platform (DTap): A Controllable and Interactive Red-Teaming Platform for AI Agents\"", "authors": "Zhaorun Chen, Xun Liu, Haibo Tong, Chengquan Guo, Yuzhou Nie, Jiawei Zhang, Mintong Kang, Chejian Xu, et al.", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2605.04808", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-05-06", "updated_at": "2026-05-06", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "coding-agent", "planning", "tool-use", "workflow-agent", "world-model" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "16", "collection_queries": "agent-safety" } }, { "collection": "papers", "path": "papers/items/2026-2605-05242-beyond-semantic-similarity-rethinking-retrieval-for-agentic-search-via-direct-co.md", "title": "\"Beyond Semantic Similarity: Rethinking Retrieval for Agentic Search via Direct Corpus Interaction\"", "type": "paper", "meta": { "type": "paper", "title": "\"Beyond Semantic Similarity: Rethinking Retrieval for Agentic Search via Direct Corpus Interaction\"", "authors": "Zhuofeng Li, Haoxiang Zhang, Cong Wei, Pan Lu, Ping Nie, Yi Lu, Yuyang Bai, Shangbin Feng, et al.", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2605.05242", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-05-03", "updated_at": "2026-05-03", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "planning", "rag", "reasoning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.IR", "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "15", "collection_queries": "language-agent" } }, { "collection": "papers", "path": "papers/items/2026-2605-05704-safeharbor-hierarchical-memory-augmented-guardrail-for-llm-agent-safety.md", "title": "\"SafeHarbor: Hierarchical Memory-Augmented Guardrail for LLM Agent Safety\"", "type": "paper", "meta": { "type": "paper", "title": "\"SafeHarbor: Hierarchical Memory-Augmented Guardrail for LLM Agent Safety\"", "authors": "Zhe Liu, Zonghao Ying, Wenxin Zhang, Quanchen Zou, Deyue Zhang, Dongdong Yang, Xiangzheng Zhang, Hao Peng", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2605.05704", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-05-07", "updated_at": "2026-05-22", "status": "queued", "relevance": "high", "topics": [ "agent-safety", "computer-use", "memory", "reasoning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CR", "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "22", "collection_queries": "agent-safety, autonomous-agent-llm" } }, { "collection": "papers", "path": "papers/items/2026-2605-05716-more-is-not-always-better-cross-component-interference-in-llm-agent-scaffolding.md", "title": "\"More Is Not Always Better: Cross-Component Interference in LLM Agent Scaffolding\"", "type": "paper", "meta": { "type": "paper", "title": "\"More Is Not Always Better: Cross-Component Interference in LLM Agent Scaffolding\"", "authors": "Ming Liu", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2605.05716", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-05-07", "updated_at": "2026-05-07", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "memory", "planning", "rag", "reasoning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI", "cs.CL" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "18", "collection_queries": "planning-agent" } }, { "collection": "papers", "path": "papers/items/2026-2605-06078-milestone-guided-policy-learning-for-long-horizon-language-agents.md", "title": "Milestone-Guided Policy Learning for Long-Horizon Language Agents", "type": "paper", "meta": { "type": "paper", "title": "Milestone-Guided Policy Learning for Long-Horizon Language Agents", "authors": "Zixuan Wang, Yuchen Yan, Hongxing Li, Teng Pan, Dingming Li, Ruiqing Zhang, Weiming Lu, Jun Xiao, et al.", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2605.06078", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-05-07", "updated_at": "2026-05-07", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "computer-use", "planning", "rag", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CL", "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "15", "collection_queries": "language-agent" } }, { "collection": "papers", "path": "papers/items/2026-2605-06713-agentic-ai-and-the-industrialization-of-cyber-offense-forecast-consequences-and-.md", "title": "\"Agentic AI and the Industrialization of Cyber Offense: Forecast, Consequences, and Defensive Priorities for Enterprises and the Mittelstand\"", "type": "paper", "meta": { "type": "paper", "title": "\"Agentic AI and the Industrialization of Cyber Offense: Forecast, Consequences, and Defensive Priorities for Enterprises and the Mittelstand\"", "authors": "Christopher Koch", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2605.06713", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-05-06", "updated_at": "2026-05-06", "status": "queued", "relevance": "high", "topics": [ "agent-safety", "computer-use", "planning", "tool-use", "workflow-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CR", "cs.AI", "cs.HC" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "17", "collection_queries": "agent-safety, planning-agent" } }, { "collection": "papers", "path": "papers/items/2026-2605-06716-from-storage-to-experience-a-survey-on-the-evolution-of-llm-agent-memory-mechani.md", "title": "\"From Storage to Experience: A Survey on the Evolution of LLM Agent Memory Mechanisms\"", "type": "paper", "meta": { "type": "paper", "title": "\"From Storage to Experience: A Survey on the Evolution of LLM Agent Memory Mechanisms\"", "authors": "Jinghao Luo, Yuchen Tian, Chuxue Cao, Ziyang Luo, Hongzhan Lin, Kaixin Li, Chuyi Kong, Ruichao Yang, et al.", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2605.06716", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-05-07", "updated_at": "2026-05-07", "status": "queued", "relevance": "high", "topics": [ "memory", "planning", "rag", "reasoning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI", "cs.CL" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "18", "collection_queries": "planning-agent" } }, { "collection": "papers", "path": "papers/items/2026-2605-06737-a-self-healing-framework-for-reliable-llm-based-autonomous-agents.md", "title": "A Self-Healing Framework for Reliable LLM-Based Autonomous Agents", "type": "paper", "meta": { "type": "paper", "title": "A Self-Healing Framework for Reliable LLM-Based Autonomous Agents", "authors": "Cheonsu Jeong, Younggun Shin", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2605.06737", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-05-07", "updated_at": "2026-05-07", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "computer-use", "multi-agent", "planning", "reasoning", "workflow-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.SE", "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "14", "collection_queries": "autonomous-agent-llm" } }, { "collection": "papers", "path": "papers/items/2026-2605-06812-towards-security-auditable-llm-agents-a-unified-graph-representation.md", "title": "\"Towards Security-Auditable LLM Agents: A Unified Graph Representation\"", "type": "paper", "meta": { "type": "paper", "title": "\"Towards Security-Auditable LLM Agents: A Unified Graph Representation\"", "authors": "Chaofan Li, Lyuye Zhang, Jintao Zhai, Siyue Feng, Xichun Yang, Huahao Wang, Shihan Dou, Yu Ji, et al.", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2605.06812", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-05-07", "updated_at": "2026-05-07", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "memory", "multi-agent", "rag", "reasoning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "21", "collection_queries": "agent-safety" } }, { "collection": "papers", "path": "papers/items/2026-2605-06869-agentick-a-unified-benchmark-for-general-sequential-decision-making-agents.md", "title": "\"Agentick: A Unified Benchmark for General Sequential Decision-Making Agents\"", "type": "paper", "meta": { "type": "paper", "title": "\"Agentick: A Unified Benchmark for General Sequential Decision-Making Agents\"", "authors": "Roger Creus Castanyer, Pablo Samuel Castro, Glen Berseth", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2605.06869", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-05-07", "updated_at": "2026-05-12", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "coding-agent", "multi-agent", "planning", "rag", "reasoning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "23", "collection_queries": "autonomous-agent-llm" } }, { "collection": "papers", "path": "papers/items/2026-2605-06890-beyond-the-black-box-interpretability-of-agentic-ai-tool-use.md", "title": "\"Beyond the Black Box: Interpretability of Agentic AI Tool Use\"", "type": "paper", "meta": { "type": "paper", "title": "\"Beyond the Black Box: Interpretability of Agentic AI Tool Use\"", "authors": "Hariom Tatsat, Ariye Shater", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2605.06890", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-05-07", "updated_at": "2026-07-05", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "planning", "tool-use", "workflow-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI", "cs.MA" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "18", "collection_queries": "function-calling" } }, { "collection": "papers", "path": "papers/items/2026-2605-06957-learning-and-reusing-policy-decompositions-for-hierarchical-generalized-planning.md", "title": "Learning and Reusing Policy Decompositions for Hierarchical Generalized Planning with LLM Agents", "type": "paper", "meta": { "type": "paper", "title": "Learning and Reusing Policy Decompositions for Hierarchical Generalized Planning with LLM Agents", "authors": "Shirin Sohrabi, Haritha Ananthakrishnan, Harsha Kokel, Kavitha Srinivas, Michael Katz", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2605.06957", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-05-07", "updated_at": "2026-05-07", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "planning", "rag" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "14", "collection_queries": "planning-agent" } }, { "collection": "papers", "path": "papers/items/2026-2605-06992-why-does-agentic-safety-fail-to-generalize-across-tasks.md", "title": "Why Does Agentic Safety Fail to Generalize Across Tasks?", "type": "paper", "meta": { "type": "paper", "title": "Why Does Agentic Safety Fail to Generalize Across Tasks?", "authors": "Yonatan Slutzky, Yotam Alexander, Tomer Slor, Yoav Nagel, Nadav Cohen", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2605.06992", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-05-07", "updated_at": "2026-05-07", "status": "queued", "relevance": "high", "topics": [ "agent-safety", "embodied-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.LG", "stat.ML" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "14", "collection_queries": "agent-safety" } }, { "collection": "papers", "path": "papers/items/2026-2605-07112-switchcraft-ai-model-router-for-agentic-tool-calling.md", "title": "\"Switchcraft: AI Model Router for Agentic Tool Calling\"", "type": "paper", "meta": { "type": "paper", "title": "\"Switchcraft: AI Model Router for Agentic Tool Calling\"", "authors": "Sharad Agarwal, Pooria Namyar, Alec Wolman, Rahul Ambavat, Ankur Gupta, Qizheng Zhang", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2605.07112", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-05-08", "updated_at": "2026-05-08", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "reasoning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI", "cs.MA" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "14", "collection_queries": "function-calling" } }, { "collection": "papers", "path": "papers/items/2026-2605-07251-can-agents-price-a-reaction-evaluating-llms-on-chemical-cost-reasoning.md", "title": "Can Agents Price a Reaction? Evaluating LLMs on Chemical Cost Reasoning", "type": "paper", "meta": { "type": "paper", "title": "Can Agents Price a Reaction? Evaluating LLMs on Chemical Cost Reasoning", "authors": "Yuyang Wu, Yue Huang, Shuaike Shen, Xujian Wang, Shuhao Zhang, Qiyao Xue, Weichen Liu, Runtian Gao, et al.", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2605.07251", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-05-08", "updated_at": "2026-05-08", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "planning", "rag", "reasoning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "17", "collection_queries": "planning-agent" } }, { "collection": "papers", "path": "papers/items/2026-2605-07830-cybiasbench-benchmarking-bias-in-llm-agents-for-cyber-attack-scenarios.md", "title": "\"CyBiasBench: Benchmarking Bias in LLM Agents for Cyber-Attack Scenarios\"", "type": "paper", "meta": { "type": "paper", "title": "\"CyBiasBench: Benchmarking Bias in LLM Agents for Cyber-Attack Scenarios\"", "authors": "Taein Lim, Seongyong Ju, Munhyeok Kim, Hyunjun Kim, Hoki Kim", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2605.07830", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-05-08", "updated_at": "2026-05-08", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CR", "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "16", "collection_queries": "autonomous-agent-llm" } }, { "collection": "papers", "path": "papers/items/2026-2605-08374-memq-integrating-q-learning-into-self-evolving-memory-agents-over-provenance-dag.md", "title": "\"MemQ: Integrating Q-Learning into Self-Evolving Memory Agents over Provenance DAGs\"", "type": "paper", "meta": { "type": "paper", "title": "\"MemQ: Integrating Q-Learning into Self-Evolving Memory Agents over Provenance DAGs\"", "authors": "Junwei Liao, Haoting Shi, Ruiwen Zhou, Jiaqian Wang, Shengtao Zhang, Wei Zhang, Ying Wen, Zhiyu Li, et al.", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2605.08374", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-05-08", "updated_at": "2026-05-14", "status": "skimmed", "relevance": "high", "topics": [ "agent-evaluation", "computer-use", "embodied-agent", "memory", "rag", "reasoning", "tool-use" ], "methods": [ "provenance-dag", "td-lambda", "q-guided-retrieval" ], "benchmarks": [ "LLAB", "LiveCodeBench", "MMMU-Pro", "ERQA", "GPQA-Diamond", "BFCL" ], "models": [], "datasets": [ "cs.AI" ], "related_concepts": [ "memory-credit-assignment", "procedural-memory" ], "related_jobs": [], "related_experiments": [ "KC-001-agent-memory-pilot" ], "related_projects": [ "learning/agent-memory" ], "collection_score": "20", "collection_queries": "function-calling" } }, { "collection": "papers", "path": "papers/items/2026-2605-08442-defense-effectiveness-across-architectural-layers-a-mechanistic-evaluation-of-pe.md", "title": "\"Defense effectiveness across architectural layers: a mechanistic evaluation of persistent memory attacks on stateful LLM agents\"", "type": "paper", "meta": { "type": "paper", "title": "\"Defense effectiveness across architectural layers: a mechanistic evaluation of persistent memory attacks on stateful LLM agents\"", "authors": "Jun Wen Leong", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2605.08442", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-05-08", "updated_at": "2026-07-03", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "computer-use", "memory", "rag", "reasoning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CR", "cs.AI", "cs.LG" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "23", "collection_queries": "agent-safety" } }, { "collection": "papers", "path": "papers/items/2026-2605-08763-when-llms-team-up-a-coordinated-attack-framework-for-automated-cyber-intrusions.md", "title": "\"When LLMs Team Up: A Coordinated Attack Framework for Automated Cyber Intrusions\"", "type": "paper", "meta": { "type": "paper", "title": "\"When LLMs Team Up: A Coordinated Attack Framework for Automated Cyber Intrusions\"", "authors": "Minfeng Qi, Tianqing Zhu, Zijie Xu, Congcong Zhu, Qin Wang, Wanlei Zhou", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2605.08763", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-05-09", "updated_at": "2026-05-09", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "multi-agent", "planning", "rag", "tool-use", "workflow-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CR" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "15", "collection_queries": "planning-agent" } }, { "collection": "papers", "path": "papers/items/2026-2605-08876-otora-a-unified-red-teaming-framework-for-reasoning-level-denial-of-service-in-l.md", "title": "\"OTora: A Unified Red Teaming Framework for Reasoning-Level Denial-of-Service in LLM Agents\"", "type": "paper", "meta": { "type": "paper", "title": "\"OTora: A Unified Red Teaming Framework for Reasoning-Level Denial-of-Service in LLM Agents\"", "authors": "Xinyu Li, Ronghui Mu, Lin Li, Tianjin Huang, Gaojie Jin", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2605.08876", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-05-09", "updated_at": "2026-06-07", "status": "queued", "relevance": "high", "topics": [ "computer-use", "reasoning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.LG" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "13", "collection_queries": "autonomous-agent-llm" } }, { "collection": "papers", "path": "papers/items/2026-2605-08964-trustworthy-ai-ensuring-reliability-and-accountability-from-models-to-agents.md", "title": "\"Trustworthy AI: Ensuring Reliability and Accountability from Models to Agents\"", "type": "paper", "meta": { "type": "paper", "title": "\"Trustworthy AI: Ensuring Reliability and Accountability from Models to Agents\"", "authors": "Carol Xuan Long", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2605.08964", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-05-09", "updated_at": "2026-05-09", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "coding-agent", "multi-agent", "rag", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.LG" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "16", "collection_queries": "autonomous-agent-llm" } }, { "collection": "papers", "path": "papers/items/2026-2605-09168-civex-causal-intervention-verification-for-language-agents.md", "title": "\"CIVeX: Causal Intervention Verification for Language Agents\"", "type": "paper", "meta": { "type": "paper", "title": "\"CIVeX: Causal Intervention Verification for Language Agents\"", "authors": "Fabio Rovai", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2605.09168", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-05-09", "updated_at": "2026-05-09", "status": "queued", "relevance": "high", "topics": [ "agent-safety", "reasoning", "tool-use", "workflow-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI", "cs.LG" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "15", "collection_queries": "language-agent" } }, { "collection": "papers", "path": "papers/items/2026-2605-09692-causal-state-binding-predicts-action-control-in-language-agents.md", "title": "Causal state binding predicts action control in language agents", "type": "paper", "meta": { "type": "paper", "title": "Causal state binding predicts action control in language agents", "authors": "Xiao Jia", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2605.09692", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-05-10", "updated_at": "2026-06-01", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "coding-agent", "memory", "planning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "17", "collection_queries": "language-agent" } }, { "collection": "papers", "path": "papers/items/2026-2605-10365-agent-valuebench-a-comprehensive-benchmark-for-evaluating-agent-values.md", "title": "\"Agent-ValueBench: A Comprehensive Benchmark for Evaluating Agent Values\"", "type": "paper", "meta": { "type": "paper", "title": "\"Agent-ValueBench: A Comprehensive Benchmark for Evaluating Agent Values\"", "authors": "Haonan Dong, Qiguan Feng, Kehan Jiang, Haoran Ye, Xin Zhang, Guojie Song", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2605.10365", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-05-11", "updated_at": "2026-05-11", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "15", "collection_queries": "autonomous-agent-llm" } }, { "collection": "papers", "path": "papers/items/2026-2605-10763-matra-modeling-the-attack-surface-of-agentic-ai-systems-openclaw-case-study.md", "title": "\"MATRA: Modeling the Attack Surface of Agentic AI Systems -- OpenClaw Case Study\"", "type": "paper", "meta": { "type": "paper", "title": "\"MATRA: Modeling the Attack Surface of Agentic AI Systems -- OpenClaw Case Study\"", "authors": "Tim Van hamme, Thomas Vissers, Javier Carnerero-Cano, Mario Fritz, Emil C. Lupu, Lieven Desmet, Dinil Mon Divakaran", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2605.10763", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-05-11", "updated_at": "2026-05-11", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "rag", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI", "cs.CR" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "15", "collection_queries": "autonomous-agent-llm" } }, { "collection": "papers", "path": "papers/items/2026-2605-10779-litmus-benchmarking-behavioral-jailbreaks-of-llm-agents-in-real-os-environments.md", "title": "\"LITMUS: Benchmarking Behavioral Jailbreaks of LLM Agents in Real OS Environments\"", "type": "paper", "meta": { "type": "paper", "title": "\"LITMUS: Benchmarking Behavioral Jailbreaks of LLM Agents in Real OS Environments\"", "authors": "Chiyu Zhang, Huiqin Yang, Bendong Jiang, Xiaolei Zhang, Yiran Zhao, Ruyi Chen, Lu Zhou, Xiaogang Xu, et al.", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2605.10779", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-05-11", "updated_at": "2026-05-11", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "multi-agent", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CR", "cs.CL" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "19", "collection_queries": "autonomous-agent-llm" } }, { "collection": "papers", "path": "papers/items/2026-2605-10870-remember-the-decision-not-the-description-a-rate-distortion-framework-for-agent-.md", "title": "\"Remember the Decision, Not the Description: A Rate-Distortion Framework for Agent Memory\"", "type": "paper", "meta": { "type": "paper", "title": "\"Remember the Decision, Not the Description: A Rate-Distortion Framework for Agent Memory\"", "authors": "Mingxi Zou, Zhihan Guo, Langzhang Liang, Zhuo Wang, Qifan Wang, Qingsong Wen, Irwin King, Lizhen Qu, et al.", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2605.10870", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-05-11", "updated_at": "2026-05-11", "status": "skimmed", "relevance": "high", "topics": [ "agent-evaluation", "memory", "planning" ], "methods": [ "decision-rate-distortion", "k-slot-memory", "certified-memory-split" ], "benchmarks": [ "LoCoMo", "synthetic-decision-diagnostics" ], "models": [], "datasets": [ "cs.AI" ], "related_concepts": [ "decision-sufficient-state", "selective-forgetting" ], "related_jobs": [], "related_experiments": [ "KC-001-agent-memory-pilot" ], "related_projects": [ "learning/agent-memory" ], "collection_score": "13", "collection_queries": "language-agent" } }, { "collection": "papers", "path": "papers/items/2026-2605-11039-the-granularity-mismatch-in-agent-security-argument-level-provenance-solves-enfo.md", "title": "\"The Granularity Mismatch in Agent Security: Argument-Level Provenance Solves Enforcement and Isolates the LLM Reasoning Bottleneck\"", "type": "paper", "meta": { "type": "paper", "title": "\"The Granularity Mismatch in Agent Security: Argument-Level Provenance Solves Enforcement and Isolates the LLM Reasoning Bottleneck\"", "authors": "Linfeng Fan, Ziwei Li, Yuan Tian, Yichen Wang, Rongsheng Li, Xiong Wang", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2605.11039", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-05-11", "updated_at": "2026-05-11", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "planning", "rag", "reasoning", "tool-use", "workflow-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CR", "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "15", "collection_queries": "agent-safety" } }, { "collection": "papers", "path": "papers/items/2026-2605-11225-pivot-bridging-planning-and-execution-in-llm-agents-via-trajectory-refinement.md", "title": "\"PIVOT: Bridging Planning and Execution in LLM Agents via Trajectory Refinement\"", "type": "paper", "meta": { "type": "paper", "title": "\"PIVOT: Bridging Planning and Execution in LLM Agents via Trajectory Refinement\"", "authors": "Tuo Zhang, Alin-Ionut Popa, Yan Xu, Rui Song, Dimitrios Dimitriadis", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2605.11225", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-05-11", "updated_at": "2026-05-11", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "coding-agent", "planning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI", "cs.LG", "cs.MA" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "17", "collection_queries": "autonomous-agent-llm, planning-agent" } }, { "collection": "papers", "path": "papers/items/2026-2605-11388-deep-reasoning-in-general-purpose-agents-via-structured-meta-cognition.md", "title": "Deep Reasoning in General Purpose Agents via Structured Meta-Cognition", "type": "paper", "meta": { "type": "paper", "title": "Deep Reasoning in General Purpose Agents via Structured Meta-Cognition", "authors": "Dean Light, Michael Theologitis, Kshitish Ghate, Shuyue Stella Li, Benjamin Newman, Chirag Shah, Aylin Caliskan, Pang Wei Koh, et al.", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2605.11388", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-05-12", "updated_at": "2026-05-12", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "computer-use", "planning", "rag", "reasoning" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CL", "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "15", "collection_queries": "planning-agent" } }, { "collection": "papers", "path": "papers/items/2026-2605-11534-prism-planning-and-reasoning-with-intent-in-simulated-embodied-environments.md", "title": "\"PRISM: : Planning and Reasoning with Intent in Simulated Embodied Environments\"", "type": "paper", "meta": { "type": "paper", "title": "\"PRISM: : Planning and Reasoning with Intent in Simulated Embodied Environments\"", "authors": "Yunn Kang Lim, Pengzhan Sun, Ziyi Bai, Xun Xu, Angela Yao, Xulei Yang, Shijie Li", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2605.11534", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-05-12", "updated_at": "2026-05-12", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "embodied-agent", "memory", "planning", "rag", "reasoning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.RO" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "15", "collection_queries": "planning-agent" } }, { "collection": "papers", "path": "papers/items/2026-2605-11633-can-llm-agents-respond-to-disasters-benchmarking-heterogeneous-geospatial-reason.md", "title": "Can LLM Agents Respond to Disasters? Benchmarking Heterogeneous Geospatial Reasoning in Emergency Operations", "type": "paper", "meta": { "type": "paper", "title": "Can LLM Agents Respond to Disasters? Benchmarking Heterogeneous Geospatial Reasoning in Emergency Operations", "authors": "Junjue Wang, Weihao Xuan, Heli Qi, Pengyu Dai, Kunyi Liu, Hongruixuan Chen, Zhuo Zheng, Junshi Xia, et al.", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2605.11633", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-05-12", "updated_at": "2026-05-12", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "planning", "rag", "reasoning", "tool-use", "workflow-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "21", "collection_queries": "planning-agent" } }, { "collection": "papers", "path": "papers/items/2026-2605-11882-on-policy-self-evolution-via-failure-trajectories-for-agentic-safety-alignment.md", "title": "On-Policy Self-Evolution via Failure Trajectories for Agentic Safety Alignment", "type": "paper", "meta": { "type": "paper", "title": "On-Policy Self-Evolution via Failure Trajectories for Agentic Safety Alignment", "authors": "Bo Yin, Qi Li, Xinchao Wang", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2605.11882", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-05-12", "updated_at": "2026-05-12", "status": "queued", "relevance": "high", "topics": [ "agent-safety", "reasoning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "15", "collection_queries": "agent-safety" } }, { "collection": "papers", "path": "papers/items/2026-2605-11928-when-simulation-lies-a-sim-to-real-benchmark-and-domain-randomized-rl-recipe-for.md", "title": "\"When Simulation Lies: A Sim-to-Real Benchmark and Domain-Randomized RL Recipe for Tool-Use Agents\"", "type": "paper", "meta": { "type": "paper", "title": "\"When Simulation Lies: A Sim-to-Real Benchmark and Domain-Randomized RL Recipe for Tool-Use Agents\"", "authors": "Xiaolin Zhou, Aojie Yuan, Zheng Luo, Zipeng Ling, Xixiao Pan, Yicheng Gao, Haiyue Zhang, Jiate Li, et al.", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2605.11928", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-05-12", "updated_at": "2026-05-12", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "tool-use", "world-model" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "13", "collection_queries": "function-calling, language-agent" } }, { "collection": "papers", "path": "papers/items/2026-2605-11946-counterfactual-trace-auditing-of-llm-agent-skills.md", "title": "Counterfactual Trace Auditing of LLM Agent Skills", "type": "paper", "meta": { "type": "paper", "title": "Counterfactual Trace Auditing of LLM Agent Skills", "authors": "Xiaolin Zhou, Jinbo Liu, Li Li, Ryan A. Rossi, Xiyang Hu", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2605.11946", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-05-12", "updated_at": "2026-05-28", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "coding-agent", "planning", "rag" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "16", "collection_queries": "planning-agent" } }, { "collection": "papers", "path": "papers/items/2026-2605-12015-skillsafetybench-evaluating-agent-safety-under-skill-facing-attack-surfaces.md", "title": "\"SkillSafetyBench: Evaluating Agent Safety under Skill-Facing Attack Surfaces\"", "type": "paper", "meta": { "type": "paper", "title": "\"SkillSafetyBench: Evaluating Agent Safety under Skill-Facing Attack Surfaces\"", "authors": "Chang Jin, An Wang, Zeming Wei, Kai Wang, Biaojie Zeng, Qiaosheng Zhang, Chao Yang, Jingjing Qu, et al.", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2605.12015", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-05-12", "updated_at": "2026-05-27", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "computer-use", "memory", "reasoning", "tool-use", "workflow-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CR", "cs.AI", "cs.CL", "cs.LG", "cs.MA" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "20", "collection_queries": "agent-safety" } }, { "collection": "papers", "path": "papers/items/2026-2605-12061-sage-a-self-evolving-agentic-graph-memory-engine-for-structure-aware-associative.md", "title": "\"SAGE: A Self-Evolving Agentic Graph-Memory Engine for Structure-Aware Associative Memory\"", "type": "paper", "meta": { "type": "paper", "title": "\"SAGE: A Self-Evolving Agentic Graph-Memory Engine for Structure-Aware Associative Memory\"", "authors": "Juntong Wang, Haoyue Zhao, guanghui Pan, Xiyuan Wang, Yanbo Wang, Qiyan Deng, Muhan Zhang", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2605.12061", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-05-12", "updated_at": "2026-05-12", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "memory", "planning", "rag", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "18", "collection_queries": "language-agent" } }, { "collection": "papers", "path": "papers/items/2026-2605-12260-prism-pareto-efficient-retrieval-over-intent-aware-structured-memory-for-long-ho.md", "title": "\"PRISM: Pareto-Efficient Retrieval over Intent-Aware Structured Memory for Long-Horizon Agents\"", "type": "paper", "meta": { "type": "paper", "title": "\"PRISM: Pareto-Efficient Retrieval over Intent-Aware Structured Memory for Long-Horizon Agents\"", "authors": "Jingyi Peng, Zhongwei Wan, Weiting Liu, Qiuzhuang Sun", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2605.12260", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-05-12", "updated_at": "2026-05-22", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "memory", "planning", "rag", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CL" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "17", "collection_queries": "language-agent" } }, { "collection": "papers", "path": "papers/items/2026-2605-13481-personalai-2-0-enhancing-knowledge-graph-traversal-retrieval-with-planning-mecha.md", "title": "\"PersonalAI 2.0: Enhancing knowledge graph traversal/retrieval with planning mechanism for Personalized LLM Agents\"", "type": "paper", "meta": { "type": "paper", "title": "\"PersonalAI 2.0: Enhancing knowledge graph traversal/retrieval with planning mechanism for Personalized LLM Agents\"", "authors": "Mikhail Menschikov, Matvey Iskornev, Alexander Kharitonov, Alina Bogdanova, Mikhail Belkin, Ekaterina Lisitsyna, Artyom Sosedka, Victoria Dochkina, et al.", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2605.13481", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-05-13", "updated_at": "2026-05-13", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "computer-use", "planning", "rag", "reasoning" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CL" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "17", "collection_queries": "planning-agent" } }, { "collection": "papers", "path": "papers/items/2026-2605-13542-realicu-do-llm-agents-understand-long-context-icu-data-a-benchmark-beyond-behavi.md", "title": "\"RealICU: Do LLM Agents Understand Long-Context ICU Data? A Benchmark Beyond Behavior Imitation\"", "type": "paper", "meta": { "type": "paper", "title": "\"RealICU: Do LLM Agents Understand Long-Context ICU Data? A Benchmark Beyond Behavior Imitation\"", "authors": "Chengzhi Shen, Weixiang Shen, Tobias Susetzky, Chen, Chen, Jun Li, Yuyuan Liu, Xuepeng Zhang, et al.", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2605.13542", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-05-13", "updated_at": "2026-05-13", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "memory", "planning", "reasoning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI", "cs.CL", "cs.LG", "cs.MA" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "20", "collection_queries": "agent-memory" } }, { "collection": "papers", "path": "papers/items/2026-2605-13618-openaaas-an-open-agent-as-a-service-framework-for-distributed-materials-informat.md", "title": "\"OpenAaaS: An Open Agent-as-a-Service Framework for Distributed Materials-Informatics Research\"", "type": "paper", "meta": { "type": "paper", "title": "\"OpenAaaS: An Open Agent-as-a-Service Framework for Distributed Materials-Informatics Research\"", "authors": "Peng Kang, Bixuan Li, Xiaoya Huang, Shuo Shi, Weiqiao Zhou, Zhen Li, Yu Liu, Lei Zheng", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2605.13618", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-05-13", "updated_at": "2026-05-13", "status": "queued", "relevance": "high", "topics": [ "memory", "multi-agent", "planning", "rag", "reasoning", "workflow-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cond-mat.mtrl-sci", "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "14", "collection_queries": "autonomous-agent-llm" } }, { "collection": "papers", "path": "papers/items/2026-2605-13716-skillops-managing-llm-agent-skill-libraries-as-self-maintaining-software-ecosyst.md", "title": "\"SkillOps: Managing LLM Agent Skill Libraries as Self-Maintaining Software Ecosystems\"", "type": "paper", "meta": { "type": "paper", "title": "\"SkillOps: Managing LLM Agent Skill Libraries as Self-Maintaining Software Ecosystems\"", "authors": "Hongji Pu, Xinyuan Song, Liang Zhao", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2605.13716", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-05-13", "updated_at": "2026-05-13", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "planning", "rag" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.SE", "cs.MA" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "14", "collection_queries": "planning-agent" } }, { "collection": "papers", "path": "papers/items/2026-2605-14126-reinforcement-learning-for-tool-calling-agents-in-fast-healthcare-interoperabili.md", "title": "Reinforcement Learning for Tool-Calling Agents in Fast Healthcare Interoperability Resources (FHIR)", "type": "paper", "meta": { "type": "paper", "title": "Reinforcement Learning for Tool-Calling Agents in Fast Healthcare Interoperability Resources (FHIR)", "authors": "Marius S. Knorr, Robert Müller, Jan P. Bremer, Nils Schweingruber", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2605.14126", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-05-13", "updated_at": "2026-05-13", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "planning", "rag", "reasoning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.LG", "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "15", "collection_queries": "planning-agent" } }, { "collection": "papers", "path": "papers/items/2026-2605-14290-web-agents-should-adopt-the-plan-then-execute-paradigm.md", "title": "Web Agents Should Adopt the Plan-Then-Execute Paradigm", "type": "paper", "meta": { "type": "paper", "title": "Web Agents Should Adopt the Plan-Then-Execute Paradigm", "authors": "Julien Piet, Annabella Chow, Yiwei Hou, Muxi Lyu, Sylvie Venuto, Jinhao Zhu, Raluca Ada Popa, David Wagner", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2605.14290", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-05-14", "updated_at": "2026-05-14", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "computer-use", "planning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CR", "cs.AI", "cs.CL", "cs.SE" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "14", "collection_queries": "planning-agent" } }, { "collection": "papers", "path": "papers/items/2026-2605-14322-are-agents-ready-to-teach-a-multi-stage-benchmark-for-real-world-teaching-workfl.md", "title": "Are Agents Ready to Teach? A Multi-Stage Benchmark for Real-World Teaching Workflows", "type": "paper", "meta": { "type": "paper", "title": "Are Agents Ready to Teach? A Multi-Stage Benchmark for Real-World Teaching Workflows", "authors": "Zixin Chen, Peng Liu, Rui Sheng, Haobo Li, Jianhong Tu, Xiaodong Deng, Kashun Shum, Dayiheng Liu, et al.", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2605.14322", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-05-14", "updated_at": "2026-05-20", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "tool-use", "workflow-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "13", "collection_queries": "language-agent" } }, { "collection": "papers", "path": "papers/items/2026-2605-14421-memlineage-lineage-guided-enforcement-for-llm-agent-memory.md", "title": "\"MemLineage: Lineage-Guided Enforcement for LLM Agent Memory\"", "type": "paper", "meta": { "type": "paper", "title": "\"MemLineage: Lineage-Guided Enforcement for LLM Agent Memory\"", "authors": "Ciyan Ouyang, Rui Hou", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2605.14421", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-05-14", "updated_at": "2026-05-14", "status": "skimmed", "relevance": "high", "topics": [ "agent-evaluation", "computer-use", "memory", "tool-use" ], "methods": [ "cryptographic-provenance", "derivation-lineage", "sensitive-action-gate" ], "benchmarks": [ "custom-cross-session-memory-harness", "AgentDojo-bridge" ], "models": [], "datasets": [ "cs.CR", "cs.AI" ], "related_concepts": [ "memory-provenance", "memory-poisoning" ], "related_jobs": [], "related_experiments": [ "KC-001-agent-memory-pilot" ], "related_projects": [ "learning/agent-memory" ], "collection_score": "17", "collection_queries": "agent-memory" } }, { "collection": "papers", "path": "papers/items/2026-2605-14460-exploiting-llm-agent-supply-chains-via-payload-less-skills.md", "title": "Exploiting LLM Agent Supply Chains via Payload-less Skills", "type": "paper", "meta": { "type": "paper", "title": "Exploiting LLM Agent Supply Chains via Payload-less Skills", "authors": "Xinyu Liu, Yukai Zhao, Xing Hu, Xin Xia", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2605.14460", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-05-14", "updated_at": "2026-05-14", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "coding-agent", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CR", "cs.SE" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "15", "collection_queries": "autonomous-agent-llm" } }, { "collection": "papers", "path": "papers/items/2026-2605-14498-groupmembench-benchmarking-llm-agent-memory-in-multi-party-conversations.md", "title": "\"GroupMemBench: Benchmarking LLM Agent Memory in Multi-Party Conversations\"", "type": "paper", "meta": { "type": "paper", "title": "\"GroupMemBench: Benchmarking LLM Agent Memory in Multi-Party Conversations\"", "authors": "Jingbo Yang, Kwei-Herng Lai, Xiaowen Wang, Shiyu Chang, Yaar Harari, Evgeniy Gabrilovich", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2605.14498", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-05-14", "updated_at": "2026-05-16", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "computer-use", "memory", "rag", "reasoning" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CL" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "21", "collection_queries": "agent-memory" } }, { "collection": "papers", "path": "papers/items/2026-2605-14527-lang2mlip-end-to-end-language-to-machine-learning-interatomic-potential-developm.md", "title": "\"Lang2MLIP: End-to-End Language-to-Machine Learning Interatomic Potential Development with Autonomous Agentic Workflows\"", "type": "paper", "meta": { "type": "paper", "title": "\"Lang2MLIP: End-to-End Language-to-Machine Learning Interatomic Potential Development with Autonomous Agentic Workflows\"", "authors": "Wenwen Li, Yuki Orimo, Nontawat Charoenphakdee", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2605.14527", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-05-14", "updated_at": "2026-05-14", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "multi-agent", "tool-use", "workflow-agent", "world-model" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.LG", "cond-mat.mtrl-sci", "physics.comp-ph" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "17", "collection_queries": "autonomous-agent-llm" } }, { "collection": "papers", "path": "papers/items/2026-2605-14892-beyond-individual-intelligence-surveying-collaboration-failure-attribution-and-s.md", "title": "\"Beyond Individual Intelligence: Surveying Collaboration, Failure Attribution, and Self-Evolution in LLM-based Multi-Agent Systems\"", "type": "paper", "meta": { "type": "paper", "title": "\"Beyond Individual Intelligence: Surveying Collaboration, Failure Attribution, and Self-Evolution in LLM-based Multi-Agent Systems\"", "authors": "Shihao Qi, Jie Ma, Rui Xing, Wei Guo, Xiao Huang, Zhitao Gao, Jianhao Deng, Jun Liu, et al.", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2605.14892", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-05-14", "updated_at": "2026-05-15", "status": "queued", "relevance": "high", "topics": [ "agent-safety", "multi-agent", "planning", "rag", "reasoning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "17", "collection_queries": "autonomous-agent-llm" } }, { "collection": "papers", "path": "papers/items/2026-2605-14906-memlens-benchmarking-multimodal-long-term-memory-in-large-vision-language-models.md", "title": "\"MemLens: Benchmarking Multimodal Long-Term Memory in Large Vision-Language Models\"", "type": "paper", "meta": { "type": "paper", "title": "\"MemLens: Benchmarking Multimodal Long-Term Memory in Large Vision-Language Models\"", "authors": "Xiyu Ren, Zhaowei Wang, Yiming Du, Zhongwei Xie, Chi Liu, Xinlin Yang, Haoyue Feng, Wenjun Pan, et al.", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2605.14906", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-05-14", "updated_at": "2026-05-14", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "memory", "rag", "reasoning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CV" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "16", "collection_queries": "agent-memory" } }, { "collection": "papers", "path": "papers/items/2026-2605-14932-toward-securing-ai-agents-like-operating-systems.md", "title": "Toward Securing AI Agents Like Operating Systems", "type": "paper", "meta": { "type": "paper", "title": "Toward Securing AI Agents Like Operating Systems", "authors": "Lukas Pirch, Micha Horlboge, Patrick Großmann, Syeda Mahnur Asif, Klim Kireev, Thorsten Holz, Konrad Rieck", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2605.14932", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-05-14", "updated_at": "2026-05-14", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "computer-use", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CR" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "16", "collection_queries": "autonomous-agent-llm" } }, { "collection": "papers", "path": "papers/items/2026-2605-15040-orchard-an-open-source-agentic-modeling-framework.md", "title": "\"Orchard: An Open-Source Agentic Modeling Framework\"", "type": "paper", "meta": { "type": "paper", "title": "\"Orchard: An Open-Source Agentic Modeling Framework\"", "authors": "Baolin Peng, Wenlin Yao, Qianhui Wu, Hao Cheng, Xiao Yu, Rui Yang, Tao Ge, Alessandro Sordoni, et al.", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2605.15040", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-05-14", "updated_at": "2026-05-21", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "coding-agent", "computer-use", "planning", "reasoning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI", "cs.CL" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "20", "collection_queries": "autonomous-agent-llm" } }, { "collection": "papers", "path": "papers/items/2026-2605-15128-memeye-a-visual-centric-evaluation-framework-for-multimodal-agent-memory.md", "title": "\"MemEye: A Visual-Centric Evaluation Framework for Multimodal Agent Memory\"", "type": "paper", "meta": { "type": "paper", "title": "\"MemEye: A Visual-Centric Evaluation Framework for Multimodal Agent Memory\"", "authors": "Minghao Guo, Qingyue Jiao, Zeru Shi, Yihao Quan, Boxuan Zhang, Danrui Li, Liwei Che, Wujiang Xu, et al.", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2605.15128", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-05-14", "updated_at": "2026-05-14", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "memory", "reasoning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CV", "cs.CL", "cs.IR" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "16", "collection_queries": "agent-memory" } }, { "collection": "papers", "path": "papers/items/2026-2605-15206-agentstop-terminating-local-ai-agents-early-to-save-energy-in-consumer-devices.md", "title": "\"AgentStop: Terminating Local AI Agents Early to Save Energy in Consumer Devices\"", "type": "paper", "meta": { "type": "paper", "title": "\"AgentStop: Terminating Local AI Agents Early to Save Energy in Consumer Devices\"", "authors": "Dzung Pham, Kleomenis Katevas, Ali Shahin Shamsabadi, Hamed Haddadi", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2605.15206", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-05-01", "updated_at": "2026-05-01", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "coding-agent", "rag", "reasoning", "tool-use", "workflow-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.LG", "cs.AI", "cs.DC" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "20", "collection_queries": "autonomous-agent-llm" } }, { "collection": "papers", "path": "papers/items/2026-2605-15625-colpackagent-agent-skill-guided-hard-particle-monte-carlo-workflows-for-colloida.md", "title": "\"ColPackAgent: Agent-Skill-Guided Hard-Particle Monte Carlo Workflows for Colloidal Packing\"", "type": "paper", "meta": { "type": "paper", "title": "\"ColPackAgent: Agent-Skill-Guided Hard-Particle Monte Carlo Workflows for Colloidal Packing\"", "authors": "Lijie Ding, Changwoo Do", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2605.15625", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-05-15", "updated_at": "2026-05-15", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "computer-use", "planning", "tool-use", "workflow-agent", "world-model" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI", "cond-mat.soft" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "14", "collection_queries": "planning-agent" } }, { "collection": "papers", "path": "papers/items/2026-2605-15701-h-mem-a-novel-memory-mechanism-for-evolving-and-retrieving-agent-memory-via-a-hy.md", "title": "\"H-Mem: A Novel Memory Mechanism for Evolving and Retrieving Agent Memory via a Hybrid Structure\"", "type": "paper", "meta": { "type": "paper", "title": "\"H-Mem: A Novel Memory Mechanism for Evolving and Retrieving Agent Memory via a Hybrid Structure\"", "authors": "Jiawei Yu, Yixiang Fang, Xilin Liu, Yuchi Ma", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2605.15701", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-05-15", "updated_at": "2026-05-15", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "memory", "rag" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CL", "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "14", "collection_queries": "agent-memory" } }, { "collection": "papers", "path": "papers/items/2026-2605-15710-smmbench-a-benchmark-for-source-distributed-multimodal-agent-memory.md", "title": "\"SMMBench: A Benchmark for Source-Distributed Multimodal Agent Memory\"", "type": "paper", "meta": { "type": "paper", "title": "\"SMMBench: A Benchmark for Source-Distributed Multimodal Agent Memory\"", "authors": "Huacan Chai, Yukai Wang, Yingxuan Yang, Dan Peng, Yuanyi Song, Zhihui Fu, Weiwen Liu, Jianghao Lin, et al.", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2605.15710", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-05-15", "updated_at": "2026-05-15", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "memory", "rag", "reasoning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CL" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "16", "collection_queries": "agent-memory" } }, { "collection": "papers", "path": "papers/items/2026-2605-15759-dimmem-dimensional-structuring-for-efficient-long-term-agent-memory.md", "title": "\"DimMem: Dimensional Structuring for Efficient Long-Term Agent Memory\"", "type": "paper", "meta": { "type": "paper", "title": "\"DimMem: Dimensional Structuring for Efficient Long-Term Agent Memory\"", "authors": "Wentao Qiu, Haotian Hu, Fanyi Wang, Jinwei Kong, Yu Zhang", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2605.15759", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-05-15", "updated_at": "2026-05-24", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "memory", "rag", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CL" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "16", "collection_queries": "agent-memory" } }, { "collection": "papers", "path": "papers/items/2026-2605-16233-forge-self-evolving-agent-memory-with-no-weight-updates-via-population-broadcast.md", "title": "\"FORGE: Self-Evolving Agent Memory With No Weight Updates via Population Broadcast\"", "type": "paper", "meta": { "type": "paper", "title": "\"FORGE: Self-Evolving Agent Memory With No Weight Updates via Population Broadcast\"", "authors": "Igor Bogdanov, Chung-Horng Lung, Thomas Kunz, Jie Gao, Adrian Taylor, Marzia Zaman", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2605.16233", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-05-15", "updated_at": "2026-05-15", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "memory", "rag", "reasoning" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI", "cs.CL", "cs.LG", "cs.MA", "eess.SY" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "16", "collection_queries": "agent-memory" } }, { "collection": "papers", "path": "papers/items/2026-2605-16481-visual-agentic-memory-enabling-online-long-video-understanding-via-online-indexi.md", "title": "\"Visual Agentic Memory: Enabling Online Long Video Understanding via Online Indexing, Hierarchical Memory, and Agentic Retrieval\"", "type": "paper", "meta": { "type": "paper", "title": "\"Visual Agentic Memory: Enabling Online Long Video Understanding via Online Indexing, Hierarchical Memory, and Agentic Retrieval\"", "authors": "Aiden Yiliu Li, Nels Numan, Anthony Steed", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2605.16481", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-05-15", "updated_at": "2026-05-15", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "memory", "planning", "rag", "reasoning" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CV", "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "18", "collection_queries": "agent-memory" } }, { "collection": "papers", "path": "papers/items/2026-2605-16821-multi-paradigm-agent-interaction-in-practice-a-systematic-analysis-of-generator-.md", "title": "\"Multi-Paradigm Agent Interaction in Practice:A Systematic Analysis of Generator-Evaluator, ReAct Loop,and Adversarial Evaluation in the buddyMe Framework\"", "type": "paper", "meta": { "type": "paper", "title": "\"Multi-Paradigm Agent Interaction in Practice:A Systematic Analysis of Generator-Evaluator, ReAct Loop,and Adversarial Evaluation in the buddyMe Framework\"", "authors": "Xiaohua Wang, Chao Han, Kai Yu, XiaoLiang Xu, Liang Wang", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2605.16821", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-05-16", "updated_at": "2026-05-16", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "computer-use", "memory", "multi-agent", "planning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "18", "collection_queries": "planning-agent" } }, { "collection": "papers", "path": "papers/items/2026-2605-17075-a-red-teaming-framework-for-evaluating-robustness-of-ai-enabled-security-orchest.md", "title": "A Red Teaming Framework for Evaluating Robustness of AI-enabled Security Orchestration, Automation, and Response Systems", "type": "paper", "meta": { "type": "paper", "title": "A Red Teaming Framework for Evaluating Robustness of AI-enabled Security Orchestration, Automation, and Response Systems", "authors": "Ayan Javeed Shaikh, Nathaniel D. Bastian, Ankit Shah", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2605.17075", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-05-16", "updated_at": "2026-05-16", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "planning", "tool-use", "workflow-agent", "world-model" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CR" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "13", "collection_queries": "autonomous-agent-llm" } }, { "collection": "papers", "path": "papers/items/2026-2605-17348-taming-zombie-agents-a-markov-state-aware-framework-for-resilient-multi-agent-ev.md", "title": "\"Taming \\\"Zombie'' Agents: A Markov State-Aware Framework for Resilient Multi-Agent Evolution\"", "type": "paper", "meta": { "type": "paper", "title": "\"Taming \\\"Zombie'' Agents: A Markov State-Aware Framework for Resilient Multi-Agent Evolution\"", "authors": "Taolin Zhang, Pukun Zhao, Qizhou Chen, Jiuheng Wan, Chen Chen, Xiaofeng He, Chengyu Wang, Richang Hong", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2605.17348", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-05-17", "updated_at": "2026-05-17", "status": "queued", "relevance": "high", "topics": [ "agent-safety", "memory", "multi-agent", "reasoning" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CL" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "15", "collection_queries": "agent-memory" } }, { "collection": "papers", "path": "papers/items/2026-2605-17453-trust-no-tool-evaluating-and-defending-llm-agents-under-untrusted-tool-feedback.md", "title": "\"Trust No Tool: Evaluating and Defending LLM Agents under Untrusted Tool Feedback\"", "type": "paper", "meta": { "type": "paper", "title": "\"Trust No Tool: Evaluating and Defending LLM Agents under Untrusted Tool Feedback\"", "authors": "Lecheng Yan, Ruizhe Li, Xicheng Han, Wenxi Li, Binwu Wang, Longyue Wang, Chenyang Lyu, Guanhua Chen", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2605.17453", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-05-17", "updated_at": "2026-05-17", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CR", "cs.CL" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "15", "collection_queries": "agent-safety" } }, { "collection": "papers", "path": "papers/items/2026-2605-17625-episodic-semantic-memory-architecture-for-long-horizon-scientific-agents.md", "title": "Episodic-Semantic Memory Architecture for Long-Horizon Scientific Agents", "type": "paper", "meta": { "type": "paper", "title": "Episodic-Semantic Memory Architecture for Long-Horizon Scientific Agents", "authors": "Nikola Milosevic", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2605.17625", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-05-17", "updated_at": "2026-05-17", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "memory", "planning", "rag", "reasoning", "tool-use", "workflow-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "18", "collection_queries": "agent-memory" } }, { "collection": "papers", "path": "papers/items/2026-2605-18284-commitdistill-a-lightweight-knowledge-centric-memory-layer-for-software-reposito.md", "title": "\"CommitDistill: A Lightweight Knowledge-Centric Memory Layer for Software Repositories\"", "type": "paper", "meta": { "type": "paper", "title": "\"CommitDistill: A Lightweight Knowledge-Centric Memory Layer for Software Repositories\"", "authors": "Divya Chukkapalli, Thejesh Avula, Aditya Aggarwal, Harsimran Singh, Amith Tallanki", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2605.18284", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-05-18", "updated_at": "2026-05-18", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "coding-agent", "computer-use", "memory", "rag", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.SE", "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "17", "collection_queries": "agent-memory" } }, { "collection": "papers", "path": "papers/items/2026-2605-18502-the-distance-based-formation-controller-design-for-multi-agent-systems-in-port-h.md", "title": "The distance-based formation controller design for multi-agent systems in port-Hamiltonian form", "type": "paper", "meta": { "type": "paper", "title": "The distance-based formation controller design for multi-agent systems in port-Hamiltonian form", "authors": "Jingyi Zhao, Yongxin Wu, Héctor García de Marina, Yuhu Wu, Yann Le Gorrec", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2605.18502", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-05-18", "updated_at": "2026-05-18", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "multi-agent", "tool-use", "world-model" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "math.OC" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "14", "collection_queries": "agent-safety" } }, { "collection": "papers", "path": "papers/items/2026-2605-18652-mementogui-learning-agentic-multimodal-memory-control-for-long-horizon-gui-agent.md", "title": "\"MementoGUI: Learning Agentic Multimodal Memory Control for Long-Horizon GUI Agents\"", "type": "paper", "meta": { "type": "paper", "title": "\"MementoGUI: Learning Agentic Multimodal Memory Control for Long-Horizon GUI Agents\"", "authors": "Ziyun Zeng, Hang Hua, Bocheng Zou, Mu Cai, Rogerio Feris, Jiebo Luo", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2605.18652", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-05-18", "updated_at": "2026-05-18", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "computer-use", "memory", "planning", "rag", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CV" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "22", "collection_queries": "agent-memory" } }, { "collection": "papers", "path": "papers/items/2026-2605-18672-position-a-three-layer-probabilistic-assume-guarantee-architecture-is-structural.md", "title": "\"Position: A Three-Layer Probabilistic Assume-Guarantee Architecture Is Structurally Required for Safe LLM Agent Deployment\"", "type": "paper", "meta": { "type": "paper", "title": "\"Position: A Three-Layer Probabilistic Assume-Guarantee Architecture Is Structurally Required for Safe LLM Agent Deployment\"", "authors": "S. Bensalem, Y. Dong, M. Franzle, X. Huang, J. Kroger, D. Nickovic, A. Nouri, R. Roy, et al.", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2605.18672", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-05-18", "updated_at": "2026-05-18", "status": "queued", "relevance": "high", "topics": [ "agent-safety", "multi-agent", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "15", "collection_queries": "agent-safety" } }, { "collection": "papers", "path": "papers/items/2026-2605-18930-oep-poisoning-self-evolving-llm-agents-via-locally-correct-but-non-transferable-.md", "title": "\"OEP: Poisoning Self-Evolving LLM Agents via Locally Correct but Non-Transferable Experiences\"", "type": "paper", "meta": { "type": "paper", "title": "\"OEP: Poisoning Self-Evolving LLM Agents via Locally Correct but Non-Transferable Experiences\"", "authors": "Kaixiang Wang, Jiong Lou, Zhaojiacheng Zhou, Jie Li", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2605.18930", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-05-18", "updated_at": "2026-05-18", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "memory", "reasoning" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CR", "cs.AI", "cs.LG" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "17", "collection_queries": "agent-memory" } }, { "collection": "papers", "path": "papers/items/2026-2605-19604-formal-skill-programmable-runtime-skills-for-efficient-and-accurate-llm-agents.md", "title": "\"Formal Skill: Programmable Runtime Skills for Efficient and Accurate LLM Agents\"", "type": "paper", "meta": { "type": "paper", "title": "\"Formal Skill: Programmable Runtime Skills for Efficient and Accurate LLM Agents\"", "authors": "Xi Zhang, Meijun Gao, Yuntian Zhao, Xinyu Tan, Yilun Yao, Feiyu Wang, Yanshu Wang, Dingsiyi, et al.", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2605.19604", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-05-19", "updated_at": "2026-05-19", "status": "queued", "relevance": "high", "topics": [ "rag", "reasoning", "tool-use", "workflow-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "14", "collection_queries": "function-calling" } }, { "collection": "papers", "path": "papers/items/2026-2605-19952-rethinking-how-to-remember-beyond-atomic-facts-in-lifelong-llm-agent-memory.md", "title": "\"Rethinking How to Remember: Beyond Atomic Facts in Lifelong LLM Agent Memory\"", "type": "paper", "meta": { "type": "paper", "title": "\"Rethinking How to Remember: Beyond Atomic Facts in Lifelong LLM Agent Memory\"", "authors": "Jingwei Sun, Jianing Zhu, Jiangchao Yao, Tongliang Liu, Bo Han", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2605.19952", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-05-19", "updated_at": "2026-05-19", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "memory", "rag", "reasoning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CL" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "18", "collection_queries": "agent-memory" } }, { "collection": "papers", "path": "papers/items/2026-2605-20306-wildroadbench-a-wild-aerial-road-damage-grounding-benchmark-for-vision-language-.md", "title": "\"WildRoadBench: A Wild Aerial Road-Damage Grounding Benchmark for Vision-Language Models and Autonomous Agents\"", "type": "paper", "meta": { "type": "paper", "title": "\"WildRoadBench: A Wild Aerial Road-Damage Grounding Benchmark for Vision-Language Models and Autonomous Agents\"", "authors": "Bingnan Liu, Chenhang Cui, Rui Huang, Jiani Luo, Zhirong Shen, Tinghao Wang, Xiande Huang, Lingbei Meng, et al.", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2605.20306", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-05-19", "updated_at": "2026-06-02", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "coding-agent", "rag", "reasoning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CV", "cs.LG" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "15", "collection_queries": "autonomous-agent-llm" } }, { "collection": "papers", "path": "papers/items/2026-2605-20315-mix-quant-quantized-prefilling-precise-decoding-for-agentic-llms.md", "title": "\"Mix-Quant: Quantized Prefilling, Precise Decoding for Agentic LLMs\"", "type": "paper", "meta": { "type": "paper", "title": "\"Mix-Quant: Quantized Prefilling, Precise Decoding for Agentic LLMs\"", "authors": "Haiquan Lu, Zigeng Chen, Gongfan Fang, Xinyin Ma, Xinchao Wang", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2605.20315", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-05-19", "updated_at": "2026-05-19", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "coding-agent", "memory", "planning", "rag", "tool-use", "workflow-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CL" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "21", "collection_queries": "planning-agent" } }, { "collection": "papers", "path": "papers/items/2026-2605-20616-auto-dreamer-learning-offline-memory-consolidation-for-language-agents.md", "title": "\"Auto-Dreamer: Learning Offline Memory Consolidation for Language Agents\"", "type": "paper", "meta": { "type": "paper", "title": "\"Auto-Dreamer: Learning Offline Memory Consolidation for Language Agents\"", "authors": "Chongrui Ye, Yuxiang Liu, Yu Wang, Haofei Yu, Yining Zhao, Ge Liu, Julian McAuley, Jiaxuan You", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2605.20616", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-05-20", "updated_at": "2026-05-20", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "memory", "rag", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CL" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "16", "collection_queries": "agent-memory, language-agent" } }, { "collection": "papers", "path": "papers/items/2026-2605-20833-memgym-a-long-horizon-memory-environment-for-llm-agents.md", "title": "\"MemGym: a Long-Horizon Memory Environment for LLM Agents\"", "type": "paper", "meta": { "type": "paper", "title": "\"MemGym: a Long-Horizon Memory Environment for LLM Agents\"", "authors": "Wujiang Xu, Yu Wang, Kai Mei, Kaiqu Liang, Zhenting Wang, Mingyu Jin, Han Zhang, Shi-Xiong Zhang, et al.", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2605.20833", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-05-20", "updated_at": "2026-05-20", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "coding-agent", "computer-use", "embodied-agent", "memory", "planning", "rag", "reasoning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CL" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "26", "collection_queries": "agent-memory" } }, { "collection": "papers", "path": "papers/items/2026-2605-20874-governance-by-construction-for-generalist-agents.md", "title": "Governance by Construction for Generalist Agents", "type": "paper", "meta": { "type": "paper", "title": "Governance by Construction for Generalist Agents", "authors": "Segev Shlomov, Iftach Shoham, Alon Oved, Ido Levy, Sami Marreed, Harold Ship, Offer Akrabi, Sergey Zeltyn, et al.", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2605.20874", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-05-20", "updated_at": "2026-05-20", "status": "queued", "relevance": "high", "topics": [ "agent-safety", "computer-use", "planning", "reasoning", "tool-use", "workflow-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI", "cs.SE" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "15", "collection_queries": "planning-agent" } }, { "collection": "papers", "path": "papers/items/2026-2605-21240-apex-autonomous-policy-exploration-for-self-evolving-llm-agents.md", "title": "\"APEX: Autonomous Policy Exploration for Self-Evolving LLM Agents\"", "type": "paper", "meta": { "type": "paper", "title": "\"APEX: Autonomous Policy Exploration for Self-Evolving LLM Agents\"", "authors": "Yibo Li, Jiashuo Yang, Zhi Zheng, Zhiyuan Hu, Yuan Sui, Shizun Wang, Yufei He, Bryan Hooi", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2605.21240", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-05-20", "updated_at": "2026-05-20", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "memory", "planning", "reasoning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.LG", "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "17", "collection_queries": "planning-agent" } }, { "collection": "papers", "path": "papers/items/2026-2605-21740-smdd-bench-can-llms-solve-real-world-small-molecule-drug-design-tasks.md", "title": "\"SMDD-Bench: Can LLMs Solve Real-World Small Molecule Drug Design Tasks?\"", "type": "paper", "meta": { "type": "paper", "title": "\"SMDD-Bench: Can LLMs Solve Real-World Small Molecule Drug Design Tasks?\"", "authors": "Kevin Han, Renfei Zhang, Kathy Wei, Hamed Mahdavi, Niloofar Mireshghallah, Amir Barati Farimani", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2605.21740", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-05-20", "updated_at": "2026-05-24", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "planning", "rag", "reasoning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "15", "collection_queries": "planning-agent" } }, { "collection": "papers", "path": "papers/items/2026-2605-22154-idlespec-exploiting-idle-time-via-speculative-planning-for-llm-agents.md", "title": "\"IdleSpec: Exploiting Idle Time via Speculative Planning for LLM Agents\"", "type": "paper", "meta": { "type": "paper", "title": "\"IdleSpec: Exploiting Idle Time via Speculative Planning for LLM Agents\"", "authors": "Daewon Choi, Kyunghyun Park, Woomin Song, Saket Dingliwal, Sai Muralidhar Jayanthi, Jinwoo Shin, Aram Galstyan", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2605.22154", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-05-21", "updated_at": "2026-05-21", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "computer-use", "planning", "rag", "reasoning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "17", "collection_queries": "planning-agent" } }, { "collection": "papers", "path": "papers/items/2026-2605-22321-benchmarking-autonomous-agents-against-temporal-spatial-and-semantic-evasions.md", "title": "Benchmarking Autonomous Agents against Temporal, Spatial, and Semantic Evasions", "type": "paper", "meta": { "type": "paper", "title": "Benchmarking Autonomous Agents against Temporal, Spatial, and Semantic Evasions", "authors": "Jianan Ma, Xiaohu Du, Ruixiao Lin, Yaoxiang Bian, Jialuo Chen, Jingyi Wang, Xiaofang Yang, Shiwen Cui, et al.", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2605.22321", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-05-21", "updated_at": "2026-05-21", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "memory", "rag", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CR", "cs.AI", "cs.SE" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "15", "collection_queries": "autonomous-agent-llm" } }, { "collection": "papers", "path": "papers/items/2026-2605-22643-boiling-the-frog-a-multi-turn-benchmark-for-agentic-safety.md", "title": "\"Boiling the Frog: A Multi-Turn Benchmark for Agentic Safety\"", "type": "paper", "meta": { "type": "paper", "title": "\"Boiling the Frog: A Multi-Turn Benchmark for Agentic Safety\"", "authors": "Piercosma Bisconti, Matteo Prandi, Federico Pierucci, Federico Sartore, Enrico Panai, Laura Caroli, Yue Zhu, Adam Leon Smith, et al.", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2605.22643", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-05-21", "updated_at": "2026-05-22", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "memory", "rag", "tool-use", "workflow-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CL" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "20", "collection_queries": "agent-safety" } }, { "collection": "papers", "path": "papers/items/2026-2605-23067-what-training-data-teaches-rl-memory-agents-an-empirical-study-of-curriculum-eff.md", "title": "\"What Training Data Teaches RL Memory Agents: An Empirical Study of Curriculum Effects in Memory-Augmented QA\"", "type": "paper", "meta": { "type": "paper", "title": "\"What Training Data Teaches RL Memory Agents: An Empirical Study of Curriculum Effects in Memory-Augmented QA\"", "authors": "Xinjie He, Zhiyuan Lin, Su Liu, Jialun Wu, Qiyang Xie, Weikai Zhou, Shuai Xiao", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2605.23067", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-05-21", "updated_at": "2026-05-21", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "memory", "reasoning" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CL" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "16", "collection_queries": "agent-memory" } }, { "collection": "papers", "path": "papers/items/2026-2605-23574-push-your-agent-measuring-and-enforcing-quantitative-goal-persistence-in-long-ho.md", "title": "\"Push Your Agent: Measuring and Enforcing Quantitative Goal Persistence in Long-Horizon LLM Agents\"", "type": "paper", "meta": { "type": "paper", "title": "\"Push Your Agent: Measuring and Enforcing Quantitative Goal Persistence in Long-Horizon LLM Agents\"", "authors": "Yuandao Cai, Yuzhang Zhu, Liyou Gao, Wensheng Tang, Shengchao Qin", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2605.23574", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-05-22", "updated_at": "2026-05-22", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "coding-agent", "planning", "rag", "reasoning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.LG", "cs.SE" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "19", "collection_queries": "language-agent" } }, { "collection": "papers", "path": "papers/items/2026-2605-23636-rf-instrument-agent-rfia-empowering-rf-instruments-with-natural-language-underst.md", "title": "\"RF Instrument Agent (RFIA): Empowering RF Instruments with Natural Language Understanding, Scheduling and Execution of Complex Tasks\"", "type": "paper", "meta": { "type": "paper", "title": "\"RF Instrument Agent (RFIA): Empowering RF Instruments with Natural Language Understanding, Scheduling and Execution of Complex Tasks\"", "authors": "Chunhui Li, Wei Fan", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2605.23636", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-05-22", "updated_at": "2026-05-22", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "planning", "rag", "tool-use", "workflow-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "eess.SY" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "15", "collection_queries": "language-agent" } }, { "collection": "papers", "path": "papers/items/2026-2605-23723-memaudit-post-hoc-auditing-of-poisoned-agent-memory-via-causal-attribution-and-s.md", "title": "\"MemAudit: Post-hoc Auditing of Poisoned Agent Memory via Causal Attribution and Structural Anomaly Detection\"", "type": "paper", "meta": { "type": "paper", "title": "\"MemAudit: Post-hoc Auditing of Poisoned Agent Memory via Causal Attribution and Structural Anomaly Detection\"", "authors": "Zhewen Tan, Yilun Yao, Huiyan Jin, Wenhan Yu, Guoan Wang, Mengyuan Fan, liang lu, Feng Liu, et al.", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2605.23723", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-05-22", "updated_at": "2026-05-22", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "memory", "planning", "reasoning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "17", "collection_queries": "agent-memory" } }, { "collection": "papers", "path": "papers/items/2026-2605-23899-from-raw-experience-to-skill-consumption-a-systematic-study-of-model-generated-a.md", "title": "\"From Raw Experience to Skill Consumption: A Systematic Study of Model-Generated Agent Skills\"", "type": "paper", "meta": { "type": "paper", "title": "\"From Raw Experience to Skill Consumption: A Systematic Study of Model-Generated Agent Skills\"", "authors": "Zisu Huang, Jingwen Xu, Yifan Yang, Ziyang Gong, Qihao Yang, Muzhao Tian, Xiaohua Wang, Changze Lv, et al.", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2605.23899", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-05-22", "updated_at": "2026-05-22", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "coding-agent", "computer-use", "rag", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "13", "collection_queries": "language-agent" } }, { "collection": "papers", "path": "papers/items/2026-2605-23986-memforest-an-efficient-agent-memory-system-with-hierarchical-temporal-indexing.md", "title": "\"MemForest: An Efficient Agent Memory System with Hierarchical Temporal Indexing\"", "type": "paper", "meta": { "type": "paper", "title": "\"MemForest: An Efficient Agent Memory System with Hierarchical Temporal Indexing\"", "authors": "Han Chen, Zining Zhang, Wenqi Pei, Bingsheng He, Ming Wu, Jason Zeng, Michael Heinrich, Wei Wu, et al.", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2605.23986", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-05-16", "updated_at": "2026-05-16", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "memory", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.DB", "cs.AI", "cs.MA" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "15", "collection_queries": "agent-memory" } }, { "collection": "papers", "path": "papers/items/2026-2605-24069-when-the-manual-lies-a-realistic-benchmark-to-evaluate-mcp-poisoning-attacks-for.md", "title": "\"When the Manual Lies: A Realistic Benchmark to Evaluate MCP Poisoning Attacks for LLM Agents\"", "type": "paper", "meta": { "type": "paper", "title": "\"When the Manual Lies: A Realistic Benchmark to Evaluate MCP Poisoning Attacks for LLM Agents\"", "authors": "Shi Liu, Xuehai Tang, Xikang Yang, Liang Lin, Biyu Zhou, Wenjie Xiao, Wantao Liu", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2605.24069", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-05-22", "updated_at": "2026-05-22", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "planning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CR", "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "19", "collection_queries": "planning-agent" } }, { "collection": "papers", "path": "papers/items/2026-2605-24216-agent-tom-learning-to-monitor-autonomous-llm-agents-via-theory-of-mind-reasoning.md", "title": "\"Agent-ToM: Learning to Monitor Autonomous LLM Agents via Theory-of-Mind Reasoning\"", "type": "paper", "meta": { "type": "paper", "title": "\"Agent-ToM: Learning to Monitor Autonomous LLM Agents via Theory-of-Mind Reasoning\"", "authors": "Nesreen K. Ahmed, Nima Nafisi", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2605.24216", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-05-22", "updated_at": "2026-05-22", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "computer-use", "memory", "planning", "reasoning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.LG", "cs.AI", "cs.CL", "cs.CR" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "19", "collection_queries": "autonomous-agent-llm" } }, { "collection": "papers", "path": "papers/items/2026-2605-24219-beyond-final-answers-auditing-trajectory-level-hallucinations-in-multi-agent-ind.md", "title": "\"Beyond Final Answers: Auditing Trajectory-Level Hallucinations in Multi-Agent Industrial Workflows\"", "type": "paper", "meta": { "type": "paper", "title": "\"Beyond Final Answers: Auditing Trajectory-Level Hallucinations in Multi-Agent Industrial Workflows\"", "authors": "Harshada Badave, Santosh Borse, Andrea Gomez, Harshitha Narahari, Sara Carter, Vishwa Bhatt, Aishani Rachakonda, Shuxin Lin, et al.", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2605.24219", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-05-22", "updated_at": "2026-05-26", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "multi-agent", "tool-use", "workflow-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "17", "collection_queries": "autonomous-agent-llm" } }, { "collection": "papers", "path": "papers/items/2026-2605-24220-polar-agentic-rl-on-any-harness-at-scale.md", "title": "\"Polar: Agentic RL on Any Harness at Scale\"", "type": "paper", "meta": { "type": "paper", "title": "\"Polar: Agentic RL on Any Harness at Scale\"", "authors": "Binfeng Xu, Hao Zhang, Shaokun Zhang, Songyang Han, Mingjie Liu, Jian Hu, Shizhe Diao, Zhenghui Jin, et al.", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2605.24220", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-05-22", "updated_at": "2026-05-22", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "coding-agent", "multi-agent", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.DC" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "19", "collection_queries": "language-agent" } }, { "collection": "papers", "path": "papers/items/2026-2605-24309-reframing-llm-agent-security-as-an-agent-human-interaction-problem.md", "title": "Reframing LLM Agent Security as an Agent-Human Interaction Problem", "type": "paper", "meta": { "type": "paper", "title": "Reframing LLM Agent Security as an Agent-Human Interaction Problem", "authors": "Peiran Wang, Ying Li, Yuan Tian", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2605.24309", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-05-23", "updated_at": "2026-05-23", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CR" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "14", "collection_queries": "agent-safety" } }, { "collection": "papers", "path": "papers/items/2026-2605-24659-iterinject-indirect-prompt-injection-against-llm-agents-via-feedback-guided-iter.md", "title": "\"IterInject: Indirect Prompt Injection Against LLM Agents via Feedback-Guided Iterative Optimization\"", "type": "paper", "meta": { "type": "paper", "title": "\"IterInject: Indirect Prompt Injection Against LLM Agents via Feedback-Guided Iterative Optimization\"", "authors": "Zixuan Chen, Jiaxiang Chen, Li Luo, Ke Xu, Xiaoxiang Huang, Tanfeng Sun, Xinghao Jiang", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2605.24659", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-05-23", "updated_at": "2026-05-23", "status": "queued", "relevance": "high", "topics": [ "agent-safety", "coding-agent", "computer-use", "planning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.LG" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "17", "collection_queries": "planning-agent" } }, { "collection": "papers", "path": "papers/items/2026-2605-24812-core-code-collaborative-reinforcement-learning-for-code-generation.md", "title": "\"CoRe-Code: Collaborative Reinforcement Learning for Code Generation\"", "type": "paper", "meta": { "type": "paper", "title": "\"CoRe-Code: Collaborative Reinforcement Learning for Code Generation\"", "authors": "Zhihao Dou, Qinjian Zhao, Zhongwei Wan, Xiaoyu Xia, Sumon Biswas", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2605.24812", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-05-24", "updated_at": "2026-05-24", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "coding-agent", "memory", "multi-agent", "planning", "rag" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "16", "collection_queries": "planning-agent" } }, { "collection": "papers", "path": "papers/items/2026-2605-25141-llm-agent-based-renewable-energy-forecasting-using-edge-and-iot-data-a-review-of.md", "title": "LLM Agent Based Renewable Energy Forecasting Using Edge and IoT Data A Review of Solar Wind Weather and Grid Aware Decision Support", "type": "paper", "meta": { "type": "paper", "title": "LLM Agent Based Renewable Energy Forecasting Using Edge and IoT Data A Review of Solar Wind Weather and Grid Aware Decision Support", "authors": "Pavan Manjunath, Thomas Pruefer", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2605.25141", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-05-24", "updated_at": "2026-05-24", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "computer-use", "planning", "rag", "reasoning", "tool-use", "workflow-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CL", "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "18", "collection_queries": "planning-agent" } }, { "collection": "papers", "path": "papers/items/2026-2605-25200-grouptravelbench-benchmarking-llm-agents-on-multi-person-travel-planning.md", "title": "\"GroupTravelBench: Benchmarking LLM Agents on Multi-Person Travel Planning\"", "type": "paper", "meta": { "type": "paper", "title": "\"GroupTravelBench: Benchmarking LLM Agents on Multi-Person Travel Planning\"", "authors": "Xiang Cheng, Yulan Hu, Lulu Zheng, Zheng Pan, Xin Li, Yong Liu", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2605.25200", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-05-24", "updated_at": "2026-06-03", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "planning", "reasoning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CL" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "19", "collection_queries": "planning-agent" } }, { "collection": "papers", "path": "papers/items/2026-2605-25310-tool-call-dependency-structure-is-linearly-decodable-in-llm-agent-residual-strea.md", "title": "Tool-Call Dependency Structure is Linearly Decodable in LLM Agent Residual Streams", "type": "paper", "meta": { "type": "paper", "title": "Tool-Call Dependency Structure is Linearly Decodable in LLM Agent Residual Streams", "authors": "Tianda Sun, Dimitar Kazakov", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2605.25310", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-05-25", "updated_at": "2026-05-25", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "planning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CL" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "14", "collection_queries": "planning-agent" } }, { "collection": "papers", "path": "papers/items/2026-2605-25393-decision-making-with-lightweight-confidence-aware-language-model-for-autonomous-.md", "title": "Decision-Making with Lightweight Confidence-Aware Language Model for Autonomous Driving", "type": "paper", "meta": { "type": "paper", "title": "Decision-Making with Lightweight Confidence-Aware Language Model for Autonomous Driving", "authors": "Ruoyu Yao, Ruiguo Zhong, Pei Liu, Mingxing Peng, Rui Yang, Jun Ma", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2605.25393", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-05-25", "updated_at": "2026-05-25", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "multi-agent", "planning", "rag", "reasoning", "tool-use", "workflow-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.RO" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "14", "collection_queries": "rag-agent" } }, { "collection": "papers", "path": "papers/items/2026-2605-25435-security-of-openclaw-agents-fundamentals-attacks-and-countermeasures.md", "title": "\"Security of OpenClaw Agents: Fundamentals, Attacks, and Countermeasures\"", "type": "paper", "meta": { "type": "paper", "title": "\"Security of OpenClaw Agents: Fundamentals, Attacks, and Countermeasures\"", "authors": "Yuntao Wang, Jianle Ba, Han Liu, Yanghe Pan, Jintao Wei, Zhou Su, Tom H. Luan, Linkang Du", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2605.25435", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-05-25", "updated_at": "2026-05-25", "status": "queued", "relevance": "high", "topics": [ "agent-safety", "computer-use", "memory", "multi-agent", "reasoning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "16", "collection_queries": "autonomous-agent-llm" } }, { "collection": "papers", "path": "papers/items/2026-2605-25920-can-llms-time-travel-enhancing-temporal-consistency-in-legal-agentic-search-thro.md", "title": "Can LLMs Time Travel? Enhancing Temporal Consistency in Legal Agentic Search through Reinforcement Learning", "type": "paper", "meta": { "type": "paper", "title": "Can LLMs Time Travel? Enhancing Temporal Consistency in Legal Agentic Search through Reinforcement Learning", "authors": "Wei Fan, Yining Zhou, Mufan Zhang, Yanbing Weng, Yiran HU, Tianshi Zheng, Baixuan Xu, Chunyang Li, et al.", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2605.25920", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-05-25", "updated_at": "2026-05-25", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "rag", "reasoning" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CL", "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "14", "collection_queries": "rag-agent" } }, { "collection": "papers", "path": "papers/items/2026-2605-26165-tool-schema-compression-enables-agentic-rag-under-constrained-context-budgets.md", "title": "Tool-Schema Compression Enables Agentic RAG Under Constrained Context Budgets", "type": "paper", "meta": { "type": "paper", "title": "Tool-Schema Compression Enables Agentic RAG Under Constrained Context Budgets", "authors": "Furkan Sakizli", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2605.26165", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-05-24", "updated_at": "2026-05-24", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "rag", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.SE", "cs.AI", "cs.CL" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "13", "collection_queries": "rag-agent" } }, { "collection": "papers", "path": "papers/items/2026-2605-26252-is-agent-memory-a-database-rethinking-data-foundations-for-long-term-ai-agent-me.md", "title": "Is Agent Memory a Database? Rethinking Data Foundations for Long-Term AI Agent Memory", "type": "paper", "meta": { "type": "paper", "title": "Is Agent Memory a Database? Rethinking Data Foundations for Long-Term AI Agent Memory", "authors": "Abdelghny Orogat, Essam Mansour", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2605.26252", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-05-25", "updated_at": "2026-05-25", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "memory", "rag", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI", "cs.DB" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "15", "collection_queries": "agent-memory" } }, { "collection": "papers", "path": "papers/items/2026-2605-26305-experiments-in-agentic-ai-for-science.md", "title": "Experiments in Agentic AI for Science", "type": "paper", "meta": { "type": "paper", "title": "Experiments in Agentic AI for Science", "authors": "Judy Fox, Geoffrey Fox", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2605.26305", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-05-25", "updated_at": "2026-05-29", "status": "queued", "relevance": "high", "topics": [ "rag", "reasoning", "tool-use", "workflow-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI", "eess.SY", "hep-ph" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "15", "collection_queries": "autonomous-agent-llm, rag-agent" } }, { "collection": "papers", "path": "papers/items/2026-2605-26497-aligning-provenance-with-authorization-a-dual-graph-defense-for-llm-agents.md", "title": "\"Aligning Provenance with Authorization: A Dual-Graph Defense for LLM Agents\"", "type": "paper", "meta": { "type": "paper", "title": "\"Aligning Provenance with Authorization: A Dual-Graph Defense for LLM Agents\"", "authors": "Peiran Wang, Ying Li, Yuan Tian", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2605.26497", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-05-26", "updated_at": "2026-05-26", "status": "queued", "relevance": "high", "topics": [ "agent-safety", "reasoning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CR" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "13", "collection_queries": "agent-safety" } }, { "collection": "papers", "path": "papers/items/2026-2605-26720-towards-feedback-to-plan-decisions-for-self-evolving-llm-agents-in-cuda-kernel-g.md", "title": "Towards Feedback-to-Plan Decisions for Self-Evolving LLM Agents in CUDA Kernel Generation", "type": "paper", "meta": { "type": "paper", "title": "Towards Feedback-to-Plan Decisions for Self-Evolving LLM Agents in CUDA Kernel Generation", "authors": "Yee Hin Chong, Jiaming Wu, Youhui Zhang, Peng Qu", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2605.26720", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-05-26", "updated_at": "2026-05-26", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "planning", "reasoning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "15", "collection_queries": "planning-agent" } }, { "collection": "papers", "path": "papers/items/2026-2605-26926-from-norms-to-indicators-n2i-rag-an-agentic-retrieval-augmented-generation-frame.md", "title": "\"From Norms to Indicators (N2I-RAG): An Agentic Retrieval-Augmented Generation Framework for Legal Indicator Computation\"", "type": "paper", "meta": { "type": "paper", "title": "\"From Norms to Indicators (N2I-RAG): An Agentic Retrieval-Augmented Generation Framework for Legal Indicator Computation\"", "authors": "Youssef Al Mouatamid, Marie Bonnin, Jihad Zahir", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2605.26926", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-05-26", "updated_at": "2026-05-26", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "planning", "rag" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "15", "collection_queries": "rag-agent" } }, { "collection": "papers", "path": "papers/items/2026-2605-27123-rethinking-agentic-rag-toward-llm-driven-logical-retrieval-beyond-embeddings.md", "title": "\"Rethinking Agentic RAG: Toward LLM-Driven Logical Retrieval Beyond Embeddings\"", "type": "paper", "meta": { "type": "paper", "title": "\"Rethinking Agentic RAG: Toward LLM-Driven Logical Retrieval Beyond Embeddings\"", "authors": "Yuqi Zeng, Qixiang Deng, Yulei Wan, Ruiquan Jiang, Xiaoqing Zheng, Xuanjing Huang", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2605.27123", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-05-26", "updated_at": "2026-05-26", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "computer-use", "rag" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.IR" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "13", "collection_queries": "rag-agent" } }, { "collection": "papers", "path": "papers/items/2026-2605-27134-scaling-benchmarking-and-reasoning-of-vision-language-agents-for-mobile-gui-navi.md", "title": "Scaling, Benchmarking, and Reasoning of Vision-Language Agents for Mobile GUI Navigation", "type": "paper", "meta": { "type": "paper", "title": "Scaling, Benchmarking, and Reasoning of Vision-Language Agents for Mobile GUI Navigation", "authors": "Heng Qu, Yike Liu, Renren Jin, Wenzong Zhang, Pengzhi Gao, Wei Liu, Jian Luan", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2605.27134", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-05-26", "updated_at": "2026-05-26", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "computer-use", "embodied-agent", "rag", "reasoning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "20", "collection_queries": "language-agent" } }, { "collection": "papers", "path": "papers/items/2026-2605-27240-enpmr-bench-benchmarking-proactive-memory-retrieval-for-emotional-support-agents.md", "title": "\"ENPMR-Bench: Benchmarking Proactive Memory Retrieval for Emotional Support Agents\"", "type": "paper", "meta": { "type": "paper", "title": "\"ENPMR-Bench: Benchmarking Proactive Memory Retrieval for Emotional Support Agents\"", "authors": "Xing Fu, Yulin Hu, Mengtong Ji, Haozhen Li, Yixin Sun, Weixiang Zhao, Yanyan Zhao, Bing Qin", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2605.27240", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-05-26", "updated_at": "2026-05-26", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "memory", "rag", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CL" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "19", "collection_queries": "language-agent" } }, { "collection": "papers", "path": "papers/items/2026-2605-27333-finharness-an-inline-lifecycle-safety-harness-for-finance-llm-agents.md", "title": "\"FinHarness: An Inline Lifecycle Safety Harness for Finance LLM Agents\"", "type": "paper", "meta": { "type": "paper", "title": "\"FinHarness: An Inline Lifecycle Safety Harness for Finance LLM Agents\"", "authors": "Haoxuan Jia, Yang Liu, Bin Chong, Yingguang Yang, Yancheng Chen, Jiayu Liang, Qian Li, Hanning Lu, et al.", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2605.27333", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-05-26", "updated_at": "2026-05-26", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "planning", "tool-use", "workflow-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CL" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "15", "collection_queries": "planning-agent" } }, { "collection": "papers", "path": "papers/items/2026-2605-27366-muse-autoskill-self-evolving-agents-via-skill-creation-memory-management-and-eva.md", "title": "\"MUSE-Autoskill: Self-Evolving Agents via Skill Creation, Memory, Management, and Evaluation\"", "type": "paper", "meta": { "type": "paper", "title": "\"MUSE-Autoskill: Self-Evolving Agents via Skill Creation, Memory, Management, and Evaluation\"", "authors": "Huawei Lin, Peng Li, Jie Song, Fuxin Jiang, Tieying Zhang", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2605.27366", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-05-26", "updated_at": "2026-07-03", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "memory" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI", "cs.CL", "cs.LG", "cs.MA" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "15", "collection_queries": "agent-memory" } }, { "collection": "papers", "path": "papers/items/2026-2605-27690-traces-proactive-safety-auditing-for-multi-turn-llm-agents-via-trajectory-state-.md", "title": "\"TRACES: Proactive Safety Auditing for Multi-Turn LLM Agents via Trajectory-State Modeling\"", "type": "paper", "meta": { "type": "paper", "title": "\"TRACES: Proactive Safety Auditing for Multi-Turn LLM Agents via Trajectory-State Modeling\"", "authors": "Jiaqian Li, Yanshu Li, Boxuan Zhang, Ruixiang Tang, Kuan-Hao Huang", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2605.27690", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-05-26", "updated_at": "2026-05-26", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "computer-use", "planning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CL", "cs.LG" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "18", "collection_queries": "agent-safety" } }, { "collection": "papers", "path": "papers/items/2026-2605-27762-peam-parametric-embodied-agent-memory-through-contrastive-internalization-of-exp.md", "title": "\"PEAM: Parametric Embodied Agent Memory through Contrastive Internalization of Experience in Minecraft\"", "type": "paper", "meta": { "type": "paper", "title": "\"PEAM: Parametric Embodied Agent Memory through Contrastive Internalization of Experience in Minecraft\"", "authors": "Yuchen Guo, Junli Gong, Weicheng Wang, Hongmin Cai, Yiu-ming Cheung, Weifeng Su", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2605.27762", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-05-26", "updated_at": "2026-06-01", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "embodied-agent", "memory", "planning", "rag", "reasoning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "17", "collection_queries": "agent-memory" } }, { "collection": "papers", "path": "papers/items/2026-2605-27825-mrmmia-membership-inference-attacks-on-memory-in-chat-agents.md", "title": "\"MRMMIA: Membership Inference Attacks on Memory in Chat Agents\"", "type": "paper", "meta": { "type": "paper", "title": "\"MRMMIA: Membership Inference Attacks on Memory in Chat Agents\"", "authors": "Kai Chen, Yan Pang, Tianhao Wang", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2605.27825", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-05-27", "updated_at": "2026-05-27", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "memory", "rag", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CR", "cs.LG" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "14", "collection_queries": "agent-memory" } }, { "collection": "papers", "path": "papers/items/2026-2605-27935-do-agents-think-deeper-a-mechanistic-investigation-of-layer-wise-dynamics-in-seq.md", "title": "Do Agents Think Deeper? A Mechanistic Investigation of Layer-Wise Dynamics in Sequential Planning", "type": "paper", "meta": { "type": "paper", "title": "Do Agents Think Deeper? A Mechanistic Investigation of Layer-Wise Dynamics in Sequential Planning", "authors": "Zhenyu Cui, Xiangzhong Luo", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2605.27935", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-05-27", "updated_at": "2026-05-27", "status": "queued", "relevance": "high", "topics": [ "planning", "reasoning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "14", "collection_queries": "autonomous-agent-llm, planning-agent" } }, { "collection": "papers", "path": "papers/items/2026-2605-28046-memcog-from-memory-as-tool-to-memory-as-cognition-in-conversational-agents.md", "title": "\"MemCog: From Memory-as-Tool to Memory-as-Cognition in Conversational Agents\"", "type": "paper", "meta": { "type": "paper", "title": "\"MemCog: From Memory-as-Tool to Memory-as-Cognition in Conversational Agents\"", "authors": "Zihan Li, Xingyu Fan, Feifei Li, Wenhui Que", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2605.28046", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-05-27", "updated_at": "2026-05-27", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "embodied-agent", "memory", "rag", "reasoning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI", "cs.CL" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "15", "collection_queries": "agent-memory" } }, { "collection": "papers", "path": "papers/items/2026-2605-28120-legalgraphrag-multi-agent-graph-retrieval-augmented-generation-for-reliable-lega.md", "title": "\"LegalGraphRAG: Multi-Agent Graph Retrieval-Augmented Generation for Reliable Legal Reasoning\"", "type": "paper", "meta": { "type": "paper", "title": "\"LegalGraphRAG: Multi-Agent Graph Retrieval-Augmented Generation for Reliable Legal Reasoning\"", "authors": "Zerui Chen, Qinggang Zhang, Zhishang Xiang, Zhimin Wei, Linfeng Gao, Xiao Huang, Zhihong Zhang, Jinsong Su", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2605.28120", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-05-27", "updated_at": "2026-05-27", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "multi-agent", "rag", "reasoning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CL", "cs.AI", "cs.MA" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "15", "collection_queries": "rag-agent" } }, { "collection": "papers", "path": "papers/items/2026-2605-28175-mixture-of-experts-knowledge-graph-retrieval-augmented-generation-for-multi-agen.md", "title": "Mixture-of-Experts Knowledge Graph Retrieval-Augmented Generation for Multi-Agent LLM-based Recommendation", "type": "paper", "meta": { "type": "paper", "title": "Mixture-of-Experts Knowledge Graph Retrieval-Augmented Generation for Multi-Agent LLM-based Recommendation", "authors": "Shijie Wang, Chengyi Liu, Yujuan Ding, Shanru Lin, See-Kiong Ng, Xu Xin, Wenqi Fan", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2605.28175", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-05-27", "updated_at": "2026-05-29", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "multi-agent", "rag" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.IR" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "14", "collection_queries": "rag-agent" } }, { "collection": "papers", "path": "papers/items/2026-2605-28424-skill0-5-joint-skill-internalization-and-utilization-for-out-of-distribution-gen.md", "title": "\"Skill0.5: Joint Skill Internalization and Utilization for Out-of-Distribution Generalization in Agentic Reinforcement Learning\"", "type": "paper", "meta": { "type": "paper", "title": "\"Skill0.5: Joint Skill Internalization and Utilization for Out-of-Distribution Generalization in Agentic Reinforcement Learning\"", "authors": "Jiapeng Zhu, Jianxiang Yu, Yibo Zhao, Chengcheng Han, Qi Gu, Xunliang Cai, Xiang Li, Weining Qian", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2605.28424", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-05-27", "updated_at": "2026-05-27", "status": "queued", "relevance": "high", "topics": [ "agent-safety", "memory" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CL" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "13", "collection_queries": "autonomous-agent-llm" } }, { "collection": "papers", "path": "papers/items/2026-2605-28607-adaptive-multimodal-agents-based-framework-for-automatic-workflow-execution.md", "title": "Adaptive Multimodal Agents-Based Framework for Automatic Workflow Execution", "type": "paper", "meta": { "type": "paper", "title": "Adaptive Multimodal Agents-Based Framework for Automatic Workflow Execution", "authors": "Susanna Cifani, Mario Luca Bernardi, Marta Cimitile", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2605.28607", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-05-27", "updated_at": "2026-05-27", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "computer-use", "embodied-agent", "multi-agent", "planning", "rag", "workflow-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI", "cs.CL" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "15", "collection_queries": "rag-agent" } }, { "collection": "papers", "path": "papers/items/2026-2605-28617-lacuna-safe-agents-as-recursive-program-holes.md", "title": "\"LACUNA: Safe Agents as Recursive Program Holes\"", "type": "paper", "meta": { "type": "paper", "title": "\"LACUNA: Safe Agents as Recursive Program Holes\"", "authors": "Yaoyu Zhao, Yichen Xu, Oliver Bračevac, Cao Nguyen Pham, Frank Zhengqing Wu, Martin Odersky", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2605.28617", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-05-27", "updated_at": "2026-05-27", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "planning", "rag", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI", "cs.PL" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "13", "collection_queries": "planning-agent" } }, { "collection": "papers", "path": "papers/items/2026-2605-28787-do-agents-need-semantic-metadata-a-comparative-study-in-agentic-data-retrieval.md", "title": "Do Agents Need Semantic Metadata? A Comparative Study in Agentic Data Retrieval", "type": "paper", "meta": { "type": "paper", "title": "Do Agents Need Semantic Metadata? A Comparative Study in Agentic Data Retrieval", "authors": "Shiyu Chen, Tarfah Alrashed, Alon Halevy, Natasha Noy", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2605.28787", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-05-27", "updated_at": "2026-05-27", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "rag", "tool-use", "workflow-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.IR", "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "15", "collection_queries": "autonomous-agent-llm" } }, { "collection": "papers", "path": "papers/items/2026-2605-28835-genesisfunc-multi-agent-data-generation-for-accurate-and-generalizable-function-.md", "title": "\"GenesisFunc: Multi-Agent Data Generation for Accurate and Generalizable Function-Calling\"", "type": "paper", "meta": { "type": "paper", "title": "\"GenesisFunc: Multi-Agent Data Generation for Accurate and Generalizable Function-Calling\"", "authors": "Hao-Xiang Xu, Chong Deng, Jiaqing Liu, Wen Wang, Qian Chen, Lujia Bao, Xiangang Li, Zhen-Hua Ling", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2605.28835", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-04-10", "updated_at": "2026-04-10", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "multi-agent", "rag", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CL", "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "16", "collection_queries": "function-calling" } }, { "collection": "papers", "path": "papers/items/2026-2605-28850-representation-signatures-and-risk-feedback-alignment-in-llm-trading-agents.md", "title": "Representation Signatures and Risk-Feedback Alignment in LLM Trading Agents", "type": "paper", "meta": { "type": "paper", "title": "Representation Signatures and Risk-Feedback Alignment in LLM Trading Agents", "authors": "Weicheng Xue", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2605.28850", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-05-16", "updated_at": "2026-05-30", "status": "queued", "relevance": "high", "topics": [ "agent-safety", "coding-agent", "memory", "planning", "reasoning", "tool-use", "world-model" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.LG", "q-fin.CP" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "15", "collection_queries": "planning-agent" } }, { "collection": "papers", "path": "papers/items/2026-2605-29341-worldmemarena-evaluating-multimodal-agent-memory-through-action-world-interactio.md", "title": "\"WorldMemArena: Evaluating Multimodal Agent Memory Through Action-World Interaction\"", "type": "paper", "meta": { "type": "paper", "title": "\"WorldMemArena: Evaluating Multimodal Agent Memory Through Action-World Interaction\"", "authors": "Chengzhi Liu, Yuzhe Yang, Sophia Xiao Pu, Yepeng Liu, Lin Long, Yichen Guo, Nuo Chen, Zhaotian Weng, et al.", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2605.29341", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-05-28", "updated_at": "2026-06-01", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "memory", "planning", "rag", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CV", "cs.CL" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "18", "collection_queries": "agent-memory, rag-agent" } }, { "collection": "papers", "path": "papers/items/2026-2605-29630-entity-collision-a-stratified-protocol-for-attributing-retrieval-lift-in-agent-m.md", "title": "\"Entity-Collision: A Stratified Protocol for Attributing Retrieval Lift in Agent Memory\"", "type": "paper", "meta": { "type": "paper", "title": "\"Entity-Collision: A Stratified Protocol for Attributing Retrieval Lift in Agent Memory\"", "authors": "Youwang Deng", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2605.29630", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-05-28", "updated_at": "2026-05-28", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "memory", "rag", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CL", "cs.AI", "cs.IR" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "13", "collection_queries": "agent-memory" } }, { "collection": "papers", "path": "papers/items/2026-2605-29640-vikingmem-a-memory-base-management-system-for-stateful-llm-based-applications.md", "title": "\"VikingMem: A Memory Base Management System for Stateful LLM-based Applications\"", "type": "paper", "meta": { "type": "paper", "title": "\"VikingMem: A Memory Base Management System for Stateful LLM-based Applications\"", "authors": "Jiajie Fu, Junwen Chen, Mengzhao Wang, Aoxiang He, Maojia Sheng, Xiangyu Ke, Yifan Zhu, Yunjun Gao", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2605.29640", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-05-28", "updated_at": "2026-06-12", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "memory", "rag", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "14", "collection_queries": "agent-memory" } }, { "collection": "papers", "path": "papers/items/2026-2605-29653-ptcg-bench-can-llm-agents-master-pok-mon-trading-card-game.md", "title": "\"PTCG-Bench: Can LLM Agents Master Pokémon Trading Card Game?\"", "type": "paper", "meta": { "type": "paper", "title": "\"PTCG-Bench: Can LLM Agents Master Pokémon Trading Card Game?\"", "authors": "Dongdong Hua, Yifei Sun, Renhong Huang, Feng Gao, Chunping Wang, Yang Yang", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2605.29653", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-05-28", "updated_at": "2026-05-28", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "13", "collection_queries": "agent-evaluation, autonomous-agent-llm" } }, { "collection": "papers", "path": "papers/items/2026-2605-29676-notation-matters-a-benchmark-study-of-token-optimized-formats-in-agentic-ai-syst.md", "title": "\"Notation Matters: A Benchmark Study of Token-Optimized Formats in Agentic AI Systems\"", "type": "paper", "meta": { "type": "paper", "title": "\"Notation Matters: A Benchmark Study of Token-Optimized Formats in Agentic AI Systems\"", "authors": "Lorenz Kutschka, Bernhard Geiger", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2605.29676", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-05-28", "updated_at": "2026-06-17", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI", "cs.CL" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "15", "collection_queries": "agent-evaluation" } }, { "collection": "papers", "path": "papers/items/2026-2605-29790-evolve-as-a-team-collaborative-self-evolution-for-llm-based-multi-agent-systems.md", "title": "\"Evolve as a Team: Collaborative Self-Evolution for LLM-based Multi-Agent Systems\"", "type": "paper", "meta": { "type": "paper", "title": "\"Evolve as a Team: Collaborative Self-Evolution for LLM-based Multi-Agent Systems\"", "authors": "Zhezheng Hao, Tianfu Wang, Huanshuo Dong, Ziyan Liu, Hong Wang, Xiankun Lin, Qiang Lin, Can Wang, et al.", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2605.29790", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-05-28", "updated_at": "2026-05-28", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "multi-agent", "planning" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.MA", "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "14", "collection_queries": "agent-evaluation" } }, { "collection": "papers", "path": "papers/items/2026-2605-29801-agentdog-1-5-a-lightweight-and-scalable-alignment-framework-for-ai-agent-safety-.md", "title": "\"AgentDoG 1.5: A Lightweight and Scalable Alignment Framework for AI Agent Safety and Security\"", "type": "paper", "meta": { "type": "paper", "title": "\"AgentDoG 1.5: A Lightweight and Scalable Alignment Framework for AI Agent Safety and Security\"", "authors": "Dongrui Liu, Yu Li, Zhonghao Yang, Peng Wang, Guanxu Chen, Yuejin Xie, Qinghua Mao, Wanying Qu, et al.", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2605.29801", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-05-28", "updated_at": "2026-05-28", "status": "queued", "relevance": "high", "topics": [ "agent-safety", "computer-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI", "cs.CL", "cs.CR", "cs.CV", "cs.LG" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "14", "collection_queries": "agent-safety" } }, { "collection": "papers", "path": "papers/items/2026-2605-29861-towards-verifiable-multimodal-deep-research-a-multi-agent-harness-for-interleave.md", "title": "\"Towards Verifiable Multimodal Deep Research: A Multi-Agent Harness for Interleaved Report Generation\"", "type": "paper", "meta": { "type": "paper", "title": "\"Towards Verifiable Multimodal Deep Research: A Multi-Agent Harness for Interleaved Report Generation\"", "authors": "Chenghao Zhang, Guanting Dong, Yufan Liu, Tong Zhao, Xiaoxi Li, Zhicheng Dou", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2605.29861", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-05-28", "updated_at": "2026-06-03", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "memory", "multi-agent", "planning", "rag", "reasoning", "tool-use", "workflow-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CL", "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "22", "collection_queries": "autonomous-agent-llm" } }, { "collection": "papers", "path": "papers/items/2026-2605-29960-hijacking-agent-memory-stealthy-trojan-attacks-through-conversational-interactio.md", "title": "\"Hijacking Agent Memory: Stealthy Trojan Attacks Through Conversational Interaction\"", "type": "paper", "meta": { "type": "paper", "title": "\"Hijacking Agent Memory: Stealthy Trojan Attacks Through Conversational Interaction\"", "authors": "Hongtao Wang, Se Yang, Yu Chen, Puzhuo Liu", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2605.29960", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-05-28", "updated_at": "2026-05-28", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "memory", "rag", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CR", "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "16", "collection_queries": "agent-memory" } }, { "collection": "papers", "path": "papers/items/2026-2605-30058-heart-bench-do-llm-agents-exhibit-human-like-psychology.md", "title": "\"HEART-Bench: Do LLM Agents Exhibit Human-like Psychology?\"", "type": "paper", "meta": { "type": "paper", "title": "\"HEART-Bench: Do LLM Agents Exhibit Human-like Psychology?\"", "authors": "Weihan Peng, Chenxu Zhang, Qianao Wang, Yuling Shi, Heng Lian, Qihong Mao, Jiahao Pang, Chunliang Feng, et al.", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2605.30058", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-05-28", "updated_at": "2026-05-28", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "computer-use", "memory", "planning", "reasoning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CL" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "17", "collection_queries": "planning-agent" } }, { "collection": "papers", "path": "papers/items/2026-2605-30090-directorbench-diagnosing-long-form-video-generation-with-personalized-multi-agen.md", "title": "\"DirectorBench: Diagnosing Long-Form Video Generation with Personalized Multi-Agent Evaluation\"", "type": "paper", "meta": { "type": "paper", "title": "\"DirectorBench: Diagnosing Long-Form Video Generation with Personalized Multi-Agent Evaluation\"", "authors": "Jiamin Chen, Qianben Chen, Jiawen Zhang, Yidi Wu, Yuchen Li, Xiaokun Zhang, Wangchunshu Zhou, Chen Ma", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2605.30090", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-05-28", "updated_at": "2026-05-28", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "multi-agent", "rag", "tool-use", "workflow-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CL", "cs.CV" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "20", "collection_queries": "agent-evaluation" } }, { "collection": "papers", "path": "papers/items/2026-2605-30407-exploring-autonomous-agentic-data-engineering-for-model-specialization.md", "title": "Exploring Autonomous Agentic Data Engineering for Model Specialization", "type": "paper", "meta": { "type": "paper", "title": "Exploring Autonomous Agentic Data Engineering for Model Specialization", "authors": "Yujie Luo, Xiangyuan Ru, Jingsheng Zheng, Jingjing Wang, Yuqi Zhu, Jintian Zhang, Runnan Fang, Kewei Xu, et al.", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2605.30407", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-05-28", "updated_at": "2026-06-08", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "computer-use", "planning", "workflow-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CL", "cs.AI", "cs.IR", "cs.LG" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "14", "collection_queries": "autonomous-agent-llm" } }, { "collection": "papers", "path": "papers/items/2026-2605-30604-an-organization-scoped-llm-agent-runtime-architecture-for-regulated-cybersecurit.md", "title": "An Organization-Scoped LLM Agent Runtime Architecture for Regulated Cybersecurity Operations", "type": "paper", "meta": { "type": "paper", "title": "An Organization-Scoped LLM Agent Runtime Architecture for Regulated Cybersecurity Operations", "authors": "George Fatouros, Georgios Makridis, George Kousiouris, John Soldatos, Dimosthenis Kyriazis", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2605.30604", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-05-28", "updated_at": "2026-05-28", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "memory", "planning", "rag", "tool-use", "workflow-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CR", "cs.AI", "cs.CL", "cs.IR" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "19", "collection_queries": "planning-agent" } }, { "collection": "papers", "path": "papers/items/2026-2605-30690-elasticmem-latent-memory-as-a-learnable-resource-for-llm-agents.md", "title": "\"ElasticMem: Latent Memory as a Learnable Resource for LLM Agents\"", "type": "paper", "meta": { "type": "paper", "title": "\"ElasticMem: Latent Memory as a Learnable Resource for LLM Agents\"", "authors": "Tao Feng, Chongrui Ye, Tianyang Luo, Jingjun Xu, Xueqiang Xu, Haozhen Zhang, Ge Liu, Jiaxuan You", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2605.30690", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-05-29", "updated_at": "2026-05-29", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "embodied-agent", "memory", "planning", "rag", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CL" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "19", "collection_queries": "planning-agent" } }, { "collection": "papers", "path": "papers/items/2026-2605-30711-sage-a-novelty-gate-for-efficient-memory-evolution-in-agentic-llms.md", "title": "\"SAGE: A Novelty Gate for Efficient Memory Evolution in Agentic LLMs\"", "type": "paper", "meta": { "type": "paper", "title": "\"SAGE: A Novelty Gate for Efficient Memory Evolution in Agentic LLMs\"", "authors": "Sijia Wang, Dhanajit Brahma, Ricardo Henao", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2605.30711", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-05-29", "updated_at": "2026-06-18", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "memory", "rag", "reasoning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CL", "cs.AI", "cs.LG", "stat.ML" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "18", "collection_queries": "agent-memory" } }, { "collection": "papers", "path": "papers/items/2026-2605-30858-forecastcompass-guiding-agentic-forecasting-with-adaptive-factor-memory.md", "title": "\"ForecastCompass: Guiding Agentic Forecasting with Adaptive Factor Memory\"", "type": "paper", "meta": { "type": "paper", "title": "\"ForecastCompass: Guiding Agentic Forecasting with Adaptive Factor Memory\"", "authors": "Yurui Chang, Yongkang Du, Yuanpu Cao, Jinghui Chen, Lu Lin", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2605.30858", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-05-29", "updated_at": "2026-05-29", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "computer-use", "memory", "rag", "reasoning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.LG" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "17", "collection_queries": "agent-memory" } }, { "collection": "papers", "path": "papers/items/2026-2605-30883-trace-task-aware-adaptive-self-evolving-agentic-jailbreaking.md", "title": "\"TRACE: Task-Aware Adaptive Self-Evolving Agentic Jailbreaking\"", "type": "paper", "meta": { "type": "paper", "title": "\"TRACE: Task-Aware Adaptive Self-Evolving Agentic Jailbreaking\"", "authors": "Churui Zeng, Weiwei Qi, Kedong Xiu, Tianhang Zheng, Chaochao Lu, Liang He, Zhan Qin, Kui Ren", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2605.30883", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-05-29", "updated_at": "2026-05-29", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "coding-agent", "computer-use", "planning", "rag", "tool-use", "workflow-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CR" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "20", "collection_queries": "planning-agent" } }, { "collection": "papers", "path": "papers/items/2026-2605-30907-bluefin-benchmarking-llm-agents-on-financial-spreadsheets.md", "title": "\"BlueFin: Benchmarking LLM Agents on Financial Spreadsheets\"", "type": "paper", "meta": { "type": "paper", "title": "\"BlueFin: Benchmarking LLM Agents on Financial Spreadsheets\"", "authors": "Srivatsa Kundurthy, Clara Na, Colton Moraine, Anoushka Mohta, Case Winter, George Fang, John Ling, Emma Strubell, et al.", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2605.30907", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-05-29", "updated_at": "2026-05-29", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "rag" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.SE", "cs.AI", "cs.CL", "cs.LG" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "17", "collection_queries": "agent-evaluation" } }, { "collection": "papers", "path": "papers/items/2026-2605-30947-extending-ai-for-research-to-the-humanities-a-multi-agent-framework-for-evidence.md", "title": "\"Extending AI for Research to the Humanities: A Multi-Agent Framework for Evidence-Grounded Scholarship\"", "type": "paper", "meta": { "type": "paper", "title": "\"Extending AI for Research to the Humanities: A Multi-Agent Framework for Evidence-Grounded Scholarship\"", "authors": "Yating Pan, Jiajun Zhang, Jun Wang, Qi Su", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2605.30947", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-05-29", "updated_at": "2026-06-03", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "multi-agent", "rag", "reasoning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CL" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "16", "collection_queries": "rag-agent" } }, { "collection": "papers", "path": "papers/items/2026-2605-31075-task-focused-memorization-for-multimodal-agents.md", "title": "Task-Focused Memorization for Multimodal Agents", "type": "paper", "meta": { "type": "paper", "title": "Task-Focused Memorization for Multimodal Agents", "authors": "Tao Zou, Yichen He, Tian Qiu, Yuan Lin, Hang Li", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2605.31075", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-05-29", "updated_at": "2026-05-29", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "computer-use", "embodied-agent", "memory" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CV" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "13", "collection_queries": "agent-memory" } }, { "collection": "papers", "path": "papers/items/2026-2605-31268-mellum2-technical-report.md", "title": "Mellum2 Technical Report", "type": "paper", "meta": { "type": "paper", "title": "Mellum2 Technical Report", "authors": "Marko Kojic, Ivan Bondyrev, Aral de Moor, Joseph Shtok, Petr Borovlev, Kseniia Lysaniuk, Madeeswaran Kannan, Ivan Dolgov, et al.", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2605.31268", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-05-29", "updated_at": "2026-05-29", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "coding-agent", "reasoning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CL" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "13", "collection_queries": "function-calling" } }, { "collection": "papers", "path": "papers/items/2026-2605-31278-industrializing-prediction-powered-inference-the-glide-library-for-reliable-gena.md", "title": "\"Industrializing Prediction-Powered Inference: The GLIDE Library for Reliable GenAI and Agentic Systems Evaluation\"", "type": "paper", "meta": { "type": "paper", "title": "\"Industrializing Prediction-Powered Inference: The GLIDE Library for Reliable GenAI and Agentic Systems Evaluation\"", "authors": "Grégoire Martinon, Ibrahim Merad, Mohammed Raki", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2605.31278", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-05-29", "updated_at": "2026-06-04", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI", "cs.LG", "stat.ME" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "15", "collection_queries": "agent-evaluation" } }, { "collection": "papers", "path": "papers/items/2026-2605-31308-tracegraph-shared-decision-landscapes-for-diagnosing-and-improving-agent-traject.md", "title": "\"TraceGraph: Shared Decision Landscapes for Diagnosing and Improving Agent Trajectories\"", "type": "paper", "meta": { "type": "paper", "title": "\"TraceGraph: Shared Decision Landscapes for Diagnosing and Improving Agent Trajectories\"", "authors": "Junjie Nian, Kang Chen, Ge Zhang, Yixin Cao, Yugang Jiang", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2605.31308", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-05-29", "updated_at": "2026-05-29", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "coding-agent", "computer-use", "embodied-agent", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "13", "collection_queries": "agent-evaluation" } }, { "collection": "papers", "path": "papers/items/2026-2605-31377-dynatree-dynamic-agentic-retrieval-tree-for-time-sensitive-news-retrieval.md", "title": "\"DynaTree: Dynamic Agentic Retrieval Tree for Time-Sensitive News Retrieval\"", "type": "paper", "meta": { "type": "paper", "title": "\"DynaTree: Dynamic Agentic Retrieval Tree for Time-Sensitive News Retrieval\"", "authors": "Siyuan Qi, Xinyuan Wang, Yingxuan Yang, Haochuan Guo, Jianghao Lin, Weiwen Liu, Yong Yu, Weinan Zhang", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2605.31377", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-05-29", "updated_at": "2026-05-29", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "planning", "rag", "reasoning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.IR", "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "16", "collection_queries": "rag-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-00198-bagen-are-llm-agents-budget-aware.md", "title": "\"BAGEN: Are LLM Agents Budget-Aware?\"", "type": "paper", "meta": { "type": "paper", "title": "\"BAGEN: Are LLM Agents Budget-Aware?\"", "authors": "Yuxiang Lin, Zihan Wang, Mengyang Liu, Yuxuan Shan, Longju Bai, Junyao Zhang, Xing Jin, Boshan Chen, et al.", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.00198", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-05-29", "updated_at": "2026-05-29", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "planning", "rag", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.LG", "cs.AI", "cs.CL" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "14", "collection_queries": "planning-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-00341-rogue-misaligned-agent-behavior-arising-from-ordinary-computer-use.md", "title": "\"ROGUE: Misaligned Agent Behavior Arising from Ordinary Computer Use\"", "type": "paper", "meta": { "type": "paper", "title": "\"ROGUE: Misaligned Agent Behavior Arising from Ordinary Computer Use\"", "authors": "Jeremy Tien, Abishek Anand, Yu-Rou Tuan, Yuchen Shen, J. Zico Kolter, Aran Nayebi", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.00341", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-05-29", "updated_at": "2026-05-29", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "computer-use", "tool-use", "workflow-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.LG", "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "16", "collection_queries": "agent-safety" } }, { "collection": "papers", "path": "papers/items/2026-2606-00610-memgraphrag-memory-based-multi-agent-system-for-graph-retrieval-augmented-genera.md", "title": "\"MemGraphRAG: Memory-based Multi-Agent System for Graph Retrieval-Augmented Generation\"", "type": "paper", "meta": { "type": "paper", "title": "\"MemGraphRAG: Memory-based Multi-Agent System for Graph Retrieval-Augmented Generation\"", "authors": "Chuanjie Wu, Zhishang Xiang, Yunbo Tang, Zerui Chen, Qinggang Zhang, Jinsong Su", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.00610", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-05-30", "updated_at": "2026-05-30", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "memory", "multi-agent", "rag", "reasoning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.IR", "cs.AI", "cs.MA" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "20", "collection_queries": "rag-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-00611-trace-trajectory-risk-aware-compression-for-long-horizon-agent-safety.md", "title": "\"TRACE: Trajectory Risk-Aware Compression for Long-Horizon Agent Safety\"", "type": "paper", "meta": { "type": "paper", "title": "\"TRACE: Trajectory Risk-Aware Compression for Long-Horizon Agent Safety\"", "authors": "Zhepei Hong, Lin Wang, Liting Li, Haokai Ma, Junfeng Fang, Fei Shen, Dan Zhang, Xiang Wang", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.00611", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-05-30", "updated_at": "2026-05-30", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "planning" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "14", "collection_queries": "agent-safety" } }, { "collection": "papers", "path": "papers/items/2026-2606-00619-mempro-agentic-memory-systems-as-evolvable-programs.md", "title": "\"MemPro: Agentic Memory Systems as Evolvable Programs\"", "type": "paper", "meta": { "type": "paper", "title": "\"MemPro: Agentic Memory Systems as Evolvable Programs\"", "authors": "Qingshan Liu, Guoqing Wang, Wen Wu, Jingqi Huang, Xinqi Tao, Dejia Song, Jie Zhou, Liang He", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.00619", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-05-30", "updated_at": "2026-05-30", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "computer-use", "memory", "planning", "rag" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CL", "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "16", "collection_queries": "agent-memory" } }, { "collection": "papers", "path": "papers/items/2026-2606-00644-foresci-evaluating-llm-agents-for-forward-looking-ai-research-judgment.md", "title": "\"ForeSci: Evaluating LLM Agents for Forward-Looking AI Research Judgment\"", "type": "paper", "meta": { "type": "paper", "title": "\"ForeSci: Evaluating LLM Agents for Forward-Looking AI Research Judgment\"", "authors": "Qiuyu Tian, Haojie Yin, Yingce Xia, Youyong Kong, Zequn Liu", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.00644", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-05-30", "updated_at": "2026-06-04", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "rag" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "13", "collection_queries": "rag-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-00756-comic-collaborative-memory-and-insights-circulation-for-long-horizon-llm-agents-.md", "title": "\"CoMIC: Collaborative Memory and Insights Circulation for Long-Horizon LLM Agents in Cloud-Edge Systems\"", "type": "paper", "meta": { "type": "paper", "title": "\"CoMIC: Collaborative Memory and Insights Circulation for Long-Horizon LLM Agents in Cloud-Edge Systems\"", "authors": "Yannan Wang, Longli Yang, Zhen Liu, Abhishek Kumar, Carsten Maple", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.00756", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-05-30", "updated_at": "2026-05-30", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "computer-use", "memory", "planning", "rag", "reasoning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "21", "collection_queries": "planning-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-00914-adversarial-feeds-steer-llm-agent-decisions-against-their-defaults.md", "title": "Adversarial Feeds Steer LLM Agent Decisions Against Their Defaults", "type": "paper", "meta": { "type": "paper", "title": "Adversarial Feeds Steer LLM Agent Decisions Against Their Defaults", "authors": "Rana Muhammad Usman", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.00914", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-05-30", "updated_at": "2026-05-30", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "rag", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI", "cs.CL", "cs.CR" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "15", "collection_queries": "agent-evaluation" } }, { "collection": "papers", "path": "papers/items/2026-2606-00915-autonomous-agentic-design-for-photonics.md", "title": "Autonomous agentic design for photonics", "type": "paper", "meta": { "type": "paper", "title": "Autonomous agentic design for photonics", "authors": "Prashanta Kharel, Amin Khavasi, Xinzhong Chen, Tyler W. Hughes", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.00915", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-05-30", "updated_at": "2026-05-30", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "computer-use", "tool-use", "world-model" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "physics.optics" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "15", "collection_queries": "autonomous-agent-llm" } }, { "collection": "papers", "path": "papers/items/2026-2606-00922-a-machine-to-machine-knowledge-guided-llm-agent-for-generalizable-radiotherapy-t.md", "title": "A Machine-to-Machine Knowledge-Guided LLM Agent for Generalizable Radiotherapy Treatment Planning", "type": "paper", "meta": { "type": "paper", "title": "A Machine-to-Machine Knowledge-Guided LLM Agent for Generalizable Radiotherapy Treatment Planning", "authors": "Md Mainul Abrar, Xun Jia, Yujie Chi", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.00922", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-05-30", "updated_at": "2026-05-30", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "computer-use", "planning", "reasoning" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "physics.med-ph", "cs.RO" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "16", "collection_queries": "planning-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-00939-fincom-a-financial-multi-agent-demo-with-disagree-or-commit-deliberation.md", "title": "\"FinCom: A Financial Multi-Agent Demo with Disagree-or-Commit Deliberation\"", "type": "paper", "meta": { "type": "paper", "title": "\"FinCom: A Financial Multi-Agent Demo with Disagree-or-Commit Deliberation\"", "authors": "Chao Peter Yang, Zixiao Tan, Kaisen Yao, Ziyu Zhou, Eleanor Jiang, Michael Wu", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.00939", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-05-31", "updated_at": "2026-05-31", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "multi-agent", "rag", "reasoning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.MA" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "19", "collection_queries": "agent-evaluation" } }, { "collection": "papers", "path": "papers/items/2026-2606-01041-expweaver-llm-agents-learn-from-experience-via-latent-rag.md", "title": "\"ExpWeaver: LLM Agents Learn from Experience via Latent RAG\"", "type": "paper", "meta": { "type": "paper", "title": "\"ExpWeaver: LLM Agents Learn from Experience via Latent RAG\"", "authors": "Tao Feng, Tianyang Luo, Jingjun Xu, Zhigang Hua, Yan Xie, Shuang Yang, Ge Liu, Jiaxuan You", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.01041", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-05-31", "updated_at": "2026-05-31", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "coding-agent", "planning", "rag", "reasoning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CL" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "17", "collection_queries": "planning-agent, rag-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-01138-memorywire-a-vendor-neutral-wire-format-for-agent-memory-operations.md", "title": "\"memorywire: A Vendor-Neutral Wire Format for Agent Memory Operations\"", "type": "paper", "meta": { "type": "paper", "title": "\"memorywire: A Vendor-Neutral Wire Format for Agent Memory Operations\"", "authors": "Thamilvendhan Munirathinam", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.01138", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-05-31", "updated_at": "2026-06-03", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "memory", "rag", "workflow-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CR", "cs.AI", "cs.DC" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "13", "collection_queries": "agent-memory" } }, { "collection": "papers", "path": "papers/items/2026-2606-01166-braveguard-from-open-world-threats-to-safer-computer-use-agents.md", "title": "\"BraveGuard: From Open-World Threats to Safer Computer-Use Agents\"", "type": "paper", "meta": { "type": "paper", "title": "\"BraveGuard: From Open-World Threats to Safer Computer-Use Agents\"", "authors": "Yunhao Feng, Xiaohu Du, Xinhao Deng, Yifan Ding, Ming Wen, Yixu Wang, Yuxiang Xie, Baihui Zheng, et al.", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.01166", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-05-31", "updated_at": "2026-06-02", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "computer-use", "rag", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CR", "cs.CL" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "13", "collection_queries": "agent-safety" } }, { "collection": "papers", "path": "papers/items/2026-2606-01185-skill-issues-data-centric-optimization-of-lakehouse-agents.md", "title": "\"\\\"Skill issues'': data-centric optimization of lakehouse agents\"", "type": "paper", "meta": { "type": "paper", "title": "\"\\\"Skill issues'': data-centric optimization of lakehouse agents\"", "authors": "Nicole Rose Schneider, Davide Ghilardi, Giacomo Piccinini, Jacopo Tagliabue", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.01185", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-05-31", "updated_at": "2026-05-31", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "coding-agent", "planning", "reasoning", "tool-use", "workflow-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "13", "collection_queries": "agent-evaluation" } }, { "collection": "papers", "path": "papers/items/2026-2606-01199-can-llm-agents-sustain-long-horizon-organizational-dynamics.md", "title": "Can LLM Agents Sustain Long-Horizon Organizational Dynamics?", "type": "paper", "meta": { "type": "paper", "title": "Can LLM Agents Sustain Long-Horizon Organizational Dynamics?", "authors": "Xuancheng Zhu, Yang Yue, Shuaibing Wan, Zihan Dou, Xiaohan Zhang, Yongrui Liu, Guoshun Nan", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.01199", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-05-31", "updated_at": "2026-05-31", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "memory", "multi-agent", "planning", "rag", "workflow-agent", "world-model" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "22", "collection_queries": "language-agent, planning-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-01222-rag-driven-multi-agent-llm-framework-with-task-decomposition-for-beyond-5g-auto-.md", "title": "RAG-driven Multi-Agent LLM Framework with Task Decomposition for Beyond 5G Auto-Configuration", "type": "paper", "meta": { "type": "paper", "title": "RAG-driven Multi-Agent LLM Framework with Task Decomposition for Beyond 5G Auto-Configuration", "authors": "İrşat Emin Sarıdaş, Onur Salan, Ali Görçin, Ibrahim Hokelek, Hakan Ali Çırpan", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.01222", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-05-31", "updated_at": "2026-05-31", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "multi-agent", "planning", "rag", "reasoning" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "eess.SP" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "16", "collection_queries": "rag-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-01385-bridging-requirements-and-architecture-multi-agent-orchestration-with-external-k.md", "title": "\"Bridging Requirements and Architecture: Multi-Agent Orchestration with External Knowledge and Hierarchical Memory\"", "type": "paper", "meta": { "type": "paper", "title": "\"Bridging Requirements and Architecture: Multi-Agent Orchestration with External Knowledge and Hierarchical Memory\"", "authors": "Ruiyin Li, Yiran Zhang, Xiyu Zhou, Yangxiao Cai, Peng Liang, Weisong Sun, Jifeng Xuan, Zhi Jin, et al.", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.01385", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-05-31", "updated_at": "2026-05-31", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "coding-agent", "memory", "multi-agent", "rag", "reasoning", "workflow-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.SE", "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "21", "collection_queries": "rag-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-01416-self-healing-agentic-orchestrators-for-reliable-tool-augmented-large-language-mo.md", "title": "Self-Healing Agentic Orchestrators for Reliable Tool-Augmented Large Language Model Systems", "type": "paper", "meta": { "type": "paper", "title": "Self-Healing Agentic Orchestrators for Reliable Tool-Augmented Large Language Model Systems", "authors": "Rahul Suresh Babu, Adarsh Agrawal", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.01416", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-05-31", "updated_at": "2026-05-31", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "computer-use", "memory", "planning", "rag", "reasoning", "tool-use", "workflow-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "20", "collection_queries": "planning-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-01528-joint-agent-memory-and-exploration-learning-via-novelty-signals.md", "title": "Joint Agent Memory and Exploration Learning via Novelty Signals", "type": "paper", "meta": { "type": "paper", "title": "Joint Agent Memory and Exploration Learning via Novelty Signals", "authors": "Shizuo Tian, Xiaohong Weng, Rui Kong, Yuxuan Chen, Guohong Liu, Yuebing Song, Jiacheng Liu, Yuchen Li, et al.", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.01528", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-01", "updated_at": "2026-06-01", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "computer-use", "memory", "rag", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "17", "collection_queries": "agent-memory" } }, { "collection": "papers", "path": "papers/items/2026-2606-01613-techrag-evidence-gated-multimodal-agentic-rag-for-technical-literature-reasoning.md", "title": "\"TechRAG: Evidence-Gated Multimodal Agentic RAG for Technical Literature Reasoning\"", "type": "paper", "meta": { "type": "paper", "title": "\"TechRAG: Evidence-Gated Multimodal Agentic RAG for Technical Literature Reasoning\"", "authors": "Kanwar Bharat Singh", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.01613", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-01", "updated_at": "2026-06-13", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "coding-agent", "computer-use", "multi-agent", "planning", "rag", "reasoning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.IR", "cs.AI", "cs.MA" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "19", "collection_queries": "rag-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-01815-crab-bench-evaluating-llm-agents-under-complex-task-dependencies-and-human-align.md", "title": "\"CRAB-Bench: Evaluating LLM Agents under Complex Task Dependencies and Human-aligned User Simulation\"", "type": "paper", "meta": { "type": "paper", "title": "\"CRAB-Bench: Evaluating LLM Agents under Complex Task Dependencies and Human-aligned User Simulation\"", "authors": "Danqing Wang, Akshay Sivaraman, Lei Li", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.01815", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-01", "updated_at": "2026-06-01", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "tool-use", "world-model" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CL" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "15", "collection_queries": "agent-evaluation" } }, { "collection": "papers", "path": "papers/items/2026-2606-01961-automedbench-towards-medical-autoresearch-with-agentic-ai-models.md", "title": "\"AutoMedBench: Towards Medical AutoResearch with Agentic AI Models\"", "type": "paper", "meta": { "type": "paper", "title": "\"AutoMedBench: Towards Medical AutoResearch with Agentic AI Models\"", "authors": "Junqi Liu, Selena Song, Yuhan Wang, Jiawei Mao, Hardy Chen, Xiaoke Huang, Tianhao Qi, Pengfei Guo, et al.", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.01961", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-01", "updated_at": "2026-06-03", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "planning", "rag", "workflow-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "13", "collection_queries": "agent-evaluation" } }, { "collection": "papers", "path": "papers/items/2026-2606-02109-badger-bridging-agentic-and-deterministic-evaluation-for-generative-enterprise-r.md", "title": "\"BADGER: Bridging Agentic and Deterministic Evaluation for Generative Enterprise Reasoning\"", "type": "paper", "meta": { "type": "paper", "title": "\"BADGER: Bridging Agentic and Deterministic Evaluation for Generative Enterprise Reasoning\"", "authors": "Shannon Serrao, Soumitra Chatterjee, Dorina Strori, Abhishek Sharma, Nathan Miller", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.02109", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-01", "updated_at": "2026-06-01", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "rag", "reasoning", "tool-use", "workflow-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "20", "collection_queries": "agent-evaluation" } }, { "collection": "papers", "path": "papers/items/2026-2606-02302-seclaw-spec-driven-security-task-synthesis-for-evaluating-autonomous-agents.md", "title": "\"SeClaw: Spec-Driven Security Task Synthesis for Evaluating Autonomous Agents\"", "type": "paper", "meta": { "type": "paper", "title": "\"SeClaw: Spec-Driven Security Task Synthesis for Evaluating Autonomous Agents\"", "authors": "Hao Cheng, Changtao Miao, Tianle Song, Yin Wu, He Liu, Erjia Xiao, Junchi Chen, Xiaoyu Shi, et al.", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.02302", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-01", "updated_at": "2026-06-01", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "memory", "rag", "tool-use", "workflow-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CR", "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "18", "collection_queries": "agent-safety, autonomous-agent-llm" } }, { "collection": "papers", "path": "papers/items/2026-2606-02372-comap-co-evolving-world-models-and-agent-policies-for-llm-agents.md", "title": "\"COMAP: Co-Evolving World Models and Agent Policies for LLM Agents\"", "type": "paper", "meta": { "type": "paper", "title": "\"COMAP: Co-Evolving World Models and Agent Policies for LLM Agents\"", "authors": "Youwei Liu, Jian Wang, Hanlin Wang, Wenjie Li", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.02372", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-01", "updated_at": "2026-06-01", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "embodied-agent", "planning", "reasoning", "tool-use", "world-model" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI", "cs.CL" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "19", "collection_queries": "language-agent, planning-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-02380-spade-bench-evaluating-spontaneous-strategic-deception-in-agents-via-plan-action.md", "title": "\"SPADE-Bench: Evaluating Spontaneous Strategic Deception in Agents via Plan-Action Divergence\"", "type": "paper", "meta": { "type": "paper", "title": "\"SPADE-Bench: Evaluating Spontaneous Strategic Deception in Agents via Plan-Action Divergence\"", "authors": "Yuyan Bu, Haowei Li, Qirui Zheng, Bowen Dong, Kaiyue Yang, Jiaming Ji, Yingshui Tan, Wenxin Li, et al.", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.02380", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-01", "updated_at": "2026-06-28", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "computer-use", "planning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CL", "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "15", "collection_queries": "agent-safety" } }, { "collection": "papers", "path": "papers/items/2026-2606-02388-policy-and-world-modeling-co-training-for-language-agents.md", "title": "Policy and World Modeling Co-Training for Language Agents", "type": "paper", "meta": { "type": "paper", "title": "Policy and World Modeling Co-Training for Language Agents", "authors": "Ning Lu, Baijiong Lin, Shengcai Liu, Jiahao Wu, Haoze Lv, Yanbin Wei, Lingting Zhu, Shengju Qian, et al.", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.02388", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-01", "updated_at": "2026-06-01", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "tool-use", "world-model" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.LG", "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "15", "collection_queries": "language-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-02404-k-browsecomp-a-web-browsing-agent-benchmark-grounded-in-korean-contexts.md", "title": "\"K-BrowseComp: A Web Browsing Agent Benchmark Grounded in Korean Contexts\"", "type": "paper", "meta": { "type": "paper", "title": "\"K-BrowseComp: A Web Browsing Agent Benchmark Grounded in Korean Contexts\"", "authors": "Nahyun Lee, Dongkeun Yoon, Guijin Son, Geewook Kim, Dayoon Ko, Jeonghun Park, Haneul Yoo, Jaewon Cho, et al.", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.02404", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-01", "updated_at": "2026-06-01", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "reasoning" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CL" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "14", "collection_queries": "agent-evaluation" } }, { "collection": "papers", "path": "papers/items/2026-2606-02461-agentcl-toward-rigorous-evaluation-of-continual-learning-in-language-agents.md", "title": "\"AgentCL: Toward Rigorous Evaluation of Continual Learning in Language Agents\"", "type": "paper", "meta": { "type": "paper", "title": "\"AgentCL: Toward Rigorous Evaluation of Continual Learning in Language Agents\"", "authors": "Yiheng Shu, Bernal Jiménez Gutiérrez, Saisri Padmaja Jonnalagedda, Yuguang Yao, Huan Sun, Yu Su", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.02461", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-01", "updated_at": "2026-06-02", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "coding-agent", "computer-use", "memory", "rag", "reasoning", "tool-use", "workflow-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI", "cs.CL" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "21", "collection_queries": "language-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-02497-bridging-the-last-mile-of-time-series-forecasting-with-llm-agents.md", "title": "Bridging the Last Mile of Time Series Forecasting with LLM Agents", "type": "paper", "meta": { "type": "paper", "title": "Bridging the Last Mile of Time Series Forecasting with LLM Agents", "authors": "Yuhua Liao, Zetian Wang, Qiangqiang Nie, Zhenhua Zhang", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.02497", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-01", "updated_at": "2026-06-01", "status": "queued", "relevance": "high", "topics": [ "agent-safety", "memory", "planning", "reasoning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "16", "collection_queries": "planning-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-02812-traj-evolve-a-self-evolving-multi-agent-system-for-patient-trajectory-modeling-i.md", "title": "\"Traj-Evolve: A Self-Evolving Multi-Agent System for Patient Trajectory Modeling in Lung Cancer Early Detection\"", "type": "paper", "meta": { "type": "paper", "title": "\"Traj-Evolve: A Self-Evolving Multi-Agent System for Patient Trajectory Modeling in Lung Cancer Early Detection\"", "authors": "Sihang Zeng, Matthew Thompson, Ruth Etzioni, Meliha Yetisgen", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.02812", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-01", "updated_at": "2026-06-01", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "memory", "multi-agent", "rag", "reasoning" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI", "cs.CL" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "17", "collection_queries": "agent-memory" } }, { "collection": "papers", "path": "papers/items/2026-2606-02965-what-benchmarks-don-t-measure-the-case-for-evaluating-abstention-competence-in-a.md", "title": "\"What Benchmarks Don't Measure: The Case for Evaluating Abstention Competence in Autonomous Agents\"", "type": "paper", "meta": { "type": "paper", "title": "\"What Benchmarks Don't Measure: The Case for Evaluating Abstention Competence in Autonomous Agents\"", "authors": "Victor Ojewale, Suresh Venkatasubramanian", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.02965", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-01", "updated_at": "2026-06-01", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "computer-use", "tool-use", "workflow-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "14", "collection_queries": "agent-evaluation" } }, { "collection": "papers", "path": "papers/items/2026-2606-03108-evotrainer-co-evolving-llm-policies-and-training-harnesses-for-autonomous-agenti.md", "title": "\"EvoTrainer: Co-Evolving LLM Policies and Training Harnesses for Autonomous Agentic Reinforcement Learning\"", "type": "paper", "meta": { "type": "paper", "title": "\"EvoTrainer: Co-Evolving LLM Policies and Training Harnesses for Autonomous Agentic Reinforcement Learning\"", "authors": "Guhong Chen, Yingcheng Shi, Yongbin Li, Binhua Li, Xander Xu, Hu Wei, Shiwen Ni, Min Yang, et al.", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.03108", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-02", "updated_at": "2026-06-12", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "coding-agent", "planning", "reasoning" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "15", "collection_queries": "autonomous-agent-llm" } }, { "collection": "papers", "path": "papers/items/2026-2606-03135-uncertainty-aware-clarification-in-llm-agents-with-information-gain.md", "title": "Uncertainty-Aware Clarification in LLM Agents with Information Gain", "type": "paper", "meta": { "type": "paper", "title": "Uncertainty-Aware Clarification in LLM Agents with Information Gain", "authors": "Mengyi Deng, Zhiwei Li, Xin Li, Tingyu Zhu, Ying Zhao, Zhijiang Guo, Wei Wang", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.03135", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-02", "updated_at": "2026-06-02", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "computer-use", "rag", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "15", "collection_queries": "agent-evaluation" } }, { "collection": "papers", "path": "papers/items/2026-2606-03157-clinicalmc-a-benchmark-for-multi-course-clinical-decision-making-with-large-lang.md", "title": "\"ClinicalMC: A Benchmark for Multi-Course Clinical Decision-Making with Large Language Models\"", "type": "paper", "meta": { "type": "paper", "title": "\"ClinicalMC: A Benchmark for Multi-Course Clinical Decision-Making with Large Language Models\"", "authors": "Ruihui Hou, Siyi Zhu, Ziyue Huai, Guangya Yu, Yongqi Fan, Chunming Wang, Tong Ruan", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.03157", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-02", "updated_at": "2026-06-02", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "multi-agent", "rag" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "13", "collection_queries": "agent-evaluation" } }, { "collection": "papers", "path": "papers/items/2026-2606-03197-memtrain-self-supervised-context-memory-training.md", "title": "\"MemTrain: Self-Supervised Context Memory Training\"", "type": "paper", "meta": { "type": "paper", "title": "\"MemTrain: Self-Supervised Context Memory Training\"", "authors": "Ziheng Li, Xingrun Xing, Haoqing Wang, Zhi-Hong Deng, Yehui Tang", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.03197", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-02", "updated_at": "2026-06-02", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "memory", "planning", "rag", "reasoning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CL" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "16", "collection_queries": "agent-memory" } }, { "collection": "papers", "path": "papers/items/2026-2606-03329-infomem-training-long-context-memory-agents-with-answer-conditioned-information-.md", "title": "\"InfoMem: Training Long-Context Memory Agents with Answer-Conditioned Information Gain\"", "type": "paper", "meta": { "type": "paper", "title": "\"InfoMem: Training Long-Context Memory Agents with Answer-Conditioned Information Gain\"", "authors": "Tiancheng Han, Yong Li, Wuzhou Yu, Qiaosheng Zhang, Wenqi Shao", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.03329", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-02", "updated_at": "2026-06-02", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "memory", "rag", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "14", "collection_queries": "agent-memory" } }, { "collection": "papers", "path": "papers/items/2026-2606-03374-emem-a-hybrid-spatio-temporal-memory-system-for-embodied-agents.md", "title": "\"eMEM: A Hybrid Spatio-Temporal Memory System For Embodied Agents\"", "type": "paper", "meta": { "type": "paper", "title": "\"eMEM: A Hybrid Spatio-Temporal Memory System For Embodied Agents\"", "authors": "A. Haroon Rasheed, Maria Kabtoul", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.03374", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-02", "updated_at": "2026-06-22", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "embodied-agent", "memory", "planning", "rag", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.RO" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "19", "collection_queries": "agent-memory, rag-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-03544-sage-a-quantitative-evaluation-of-socialized-evolution-in-agent-ecosystems.md", "title": "\"SAGE: A Quantitative Evaluation of Socialized Evolution in Agent Ecosystems\"", "type": "paper", "meta": { "type": "paper", "title": "\"SAGE: A Quantitative Evaluation of Socialized Evolution in Agent Ecosystems\"", "authors": "Linyue Pan, Yaoming Zhu, Lin Qiu, Xuezhi Cao, Xunliang Cai", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.03544", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-02", "updated_at": "2026-06-02", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "planning", "reasoning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI", "cs.CL" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "13", "collection_queries": "language-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-03657-diagnosing-knowledge-gaps-in-llm-tool-use-an-agentic-benchmark-for-novel-api-acq.md", "title": "\"Diagnosing Knowledge Gaps in LLM Tool Use: An Agentic Benchmark for Novel API Acquisition\"", "type": "paper", "meta": { "type": "paper", "title": "\"Diagnosing Knowledge Gaps in LLM Tool Use: An Agentic Benchmark for Novel API Acquisition\"", "authors": "Jinnuo Liu, Yue Peng, Jinhan Niu, Hongyi Wen", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.03657", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-02", "updated_at": "2026-06-02", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "coding-agent", "rag", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "20", "collection_queries": "agent-evaluation" } }, { "collection": "papers", "path": "papers/items/2026-2606-03895-agent-libos-a-runtime-substrate-for-capability-controlled-self-evolving-llm-agen.md", "title": "\"Agent libOS: A Runtime Substrate for Capability-Controlled Self-Evolving LLM Agents\"", "type": "paper", "meta": { "type": "paper", "title": "\"Agent libOS: A Runtime Substrate for Capability-Controlled Self-Evolving LLM Agents\"", "authors": "Yingqi Zhang", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.03895", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-02", "updated_at": "2026-06-29", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "memory", "planning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.OS", "cs.AI", "cs.CR" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "18", "collection_queries": "planning-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-04051-rubas-rubric-based-reinforcement-learning-for-agent-safety.md", "title": "\"RUBAS: Rubric-Based Reinforcement Learning for Agent Safety\"", "type": "paper", "meta": { "type": "paper", "title": "\"RUBAS: Rubric-Based Reinforcement Learning for Agent Safety\"", "authors": "Xian Qi Loye, Qinglin Su, Zhexin Zhang, Shiyao Cui, Qi Zhu, Fei Mi, Hongning Wang, Minlie Huang", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.04051", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-02", "updated_at": "2026-06-02", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.LG", "cs.AI", "cs.CR" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "17", "collection_queries": "agent-safety" } }, { "collection": "papers", "path": "papers/items/2026-2606-04120-salimory-orchestrating-cognitive-memory-for-conversational-agents.md", "title": "\"SaliMory: Orchestrating Cognitive Memory for Conversational Agents\"", "type": "paper", "meta": { "type": "paper", "title": "\"SaliMory: Orchestrating Cognitive Memory for Conversational Agents\"", "authors": "Kai Zhang, Xinyuan Zhang, Hongda Jiang, Shiun-Zu Kuo, Hyokun Yun, Ejaz Ahmed, Shereen Oraby, Ziyun Li, et al.", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.04120", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-02", "updated_at": "2026-06-02", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "memory", "rag", "reasoning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CL", "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "15", "collection_queries": "agent-memory" } }, { "collection": "papers", "path": "papers/items/2026-2606-04296-the-saturation-trap-and-the-subjectivity-of-intervention-timing-why-affect-based.md", "title": "\"The Saturation Trap and the Subjectivity of Intervention Timing: Why Affect-Based Triggers and LLM Judges Fail to Time Interventions on Autonomous Agents\"", "type": "paper", "meta": { "type": "paper", "title": "\"The Saturation Trap and the Subjectivity of Intervention Timing: Why Affect-Based Triggers and LLM Judges Fail to Time Interventions on Autonomous Agents\"", "authors": "Manvendra Modgil", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.04296", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-02", "updated_at": "2026-06-02", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "coding-agent", "planning", "reasoning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "15", "collection_queries": "autonomous-agent-llm" } }, { "collection": "papers", "path": "papers/items/2026-2606-04315-exploring-cross-scenario-generality-of-agentic-memory-systems-diagnostics-and-a-.md", "title": "\"Exploring Cross-Scenario Generality of Agentic Memory Systems: Diagnostics and a Strong Baseline\"", "type": "paper", "meta": { "type": "paper", "title": "\"Exploring Cross-Scenario Generality of Agentic Memory Systems: Diagnostics and a Strong Baseline\"", "authors": "Zhikai Chen, Jialiang Gu, Junyu Yin, Xianxuan Long, Shenglai Zeng, Xiaoze Liu, Kai Guo, Keren Zhou, et al.", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.04315", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-03", "updated_at": "2026-06-03", "status": "skimmed", "relevance": "high", "topics": [ "agent-evaluation", "memory", "planning", "rag", "tool-use" ], "methods": [ "agentic-memory-harness", "schema-diagnostics", "active-retrieval" ], "benchmarks": [ "LoCoMo", "HotpotQA", "AMABench", "ALFWorld" ], "models": [], "datasets": [ "cs.AI" ], "related_concepts": [ "schema-commitment", "agentic-retrieval" ], "related_jobs": [], "related_experiments": [ "KC-001-agent-memory-pilot" ], "related_projects": [ "learning/agent-memory" ], "collection_score": "19", "collection_queries": "agent-memory" } }, { "collection": "papers", "path": "papers/items/2026-2606-04555-temporal-order-matters-for-agentic-memory-segment-trees-for-long-horizon-agents.md", "title": "\"Temporal Order Matters for Agentic Memory: Segment Trees for Long-Horizon Agents\"", "type": "paper", "meta": { "type": "paper", "title": "\"Temporal Order Matters for Agentic Memory: Segment Trees for Long-Horizon Agents\"", "authors": "Yifan Simon Liu, Liam Gallagher, Faeze Moradi Kalarde, Jiazhou Liang, Armin Toroghi, Scott Sanner", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.04555", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-03", "updated_at": "2026-06-03", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "memory", "planning", "rag" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CL", "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "18", "collection_queries": "agent-memory" } }, { "collection": "papers", "path": "papers/items/2026-2606-04599-plan-first-judge-later-run-better-a-dmaic-inspired-agentic-system-for-industrial.md", "title": "\"Plan First, Judge Later, Run Better: A DMAIC-Inspired Agentic System for Industrial Anomaly Detection\"", "type": "paper", "meta": { "type": "paper", "title": "\"Plan First, Judge Later, Run Better: A DMAIC-Inspired Agentic System for Industrial Anomaly Detection\"", "authors": "Yongzi Yu, Ao Li, Le Wang, Ziyue Li, Fugee Tsung, Yuxuan Liang, Man Li", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.04599", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-03", "updated_at": "2026-06-03", "status": "queued", "relevance": "high", "topics": [ "agent-safety", "multi-agent", "planning", "rag", "workflow-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI", "cs.CE" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "17", "collection_queries": "planning-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-04628-rampart-registry-based-agentic-memory-with-priority-aware-runtime-transformation.md", "title": "\"RAMPART: Registry-based Agentic Memory with Priority-Aware Runtime Transformation\"", "type": "paper", "meta": { "type": "paper", "title": "\"RAMPART: Registry-based Agentic Memory with Priority-Aware Runtime Transformation\"", "authors": "Nikodem Tomczak", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.04628", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-03", "updated_at": "2026-06-03", "status": "queued", "relevance": "high", "topics": [ "memory" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CL", "cs.MA" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "14", "collection_queries": "agent-memory" } }, { "collection": "papers", "path": "papers/items/2026-2606-04780-personatree-structured-lifecycle-memory-for-person-understanding-in-llm-agents.md", "title": "\"PersonaTree: Structured Lifecycle Memory for Person Understanding in LLM Agents\"", "type": "paper", "meta": { "type": "paper", "title": "\"PersonaTree: Structured Lifecycle Memory for Person Understanding in LLM Agents\"", "authors": "Yubo Hou, Jingwei Song, Hongbo Zhang, Zhisheng Chen, Bang Xiao, Tao Wan, Zengchang Qin", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.04780", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-03", "updated_at": "2026-06-03", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "computer-use", "memory", "rag", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CL" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "20", "collection_queries": "agent-memory" } }, { "collection": "papers", "path": "papers/items/2026-2606-04874-agent-planning-benchmark-a-diagnostic-framework-for-planning-capabilities-in-llm.md", "title": "\"Agent Planning Benchmark: A Diagnostic Framework for Planning Capabilities in LLM Agents\"", "type": "paper", "meta": { "type": "paper", "title": "\"Agent Planning Benchmark: A Diagnostic Framework for Planning Capabilities in LLM Agents\"", "authors": "Haoyu Sun, Wenxuan Wang, Mingyang Song, Jujie He, Weinan Zhang, Yang Liu, Yang Yang, Yu Cheng", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.04874", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-03", "updated_at": "2026-06-05", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "computer-use", "planning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CL" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "19", "collection_queries": "agent-evaluation, planning-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-04990-from-agent-traces-to-trust-a-survey-of-evidence-tracing-and-execution-provenance.md", "title": "\"From Agent Traces to Trust: A Survey of Evidence Tracing and Execution Provenance in LLM Agents\"", "type": "paper", "meta": { "type": "paper", "title": "\"From Agent Traces to Trust: A Survey of Evidence Tracing and Execution Provenance in LLM Agents\"", "authors": "Yiqi Wang, Jiaqi Zhang, Taotao Cai, Zirui Liu, Qingqiang Sun, Zequn Sun, Zhangkai Wu, Manqing Dong, et al.", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.04990", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-03", "updated_at": "2026-06-28", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "memory", "multi-agent", "planning", "rag", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CR", "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "21", "collection_queries": "planning-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-05241-search-time-contamination-in-deep-research-agents-measuring-performance-inflatio.md", "title": "\"Search-Time Contamination in Deep Research Agents: Measuring Performance Inflation in Public Benchmark Evaluation\"", "type": "paper", "meta": { "type": "paper", "title": "\"Search-Time Contamination in Deep Research Agents: Measuring Performance Inflation in Public Benchmark Evaluation\"", "authors": "Yongjie Wang, Xinyue Zhang, Kunhong Yao, Zhiwei Zeng, Kaisong Song, Jun Lin, Zhiqi Shen", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.05241", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-03", "updated_at": "2026-06-03", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "rag", "reasoning" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CR", "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "16", "collection_queries": "agent-evaluation" } }, { "collection": "papers", "path": "papers/items/2026-2606-05263-policy-conditioned-counterfactual-credit-for-verifiable-reinforcement-learning-o.md", "title": "Policy-Conditioned Counterfactual Credit for Verifiable Reinforcement Learning of Long-Horizon Language Agents", "type": "paper", "meta": { "type": "paper", "title": "Policy-Conditioned Counterfactual Credit for Verifiable Reinforcement Learning of Long-Horizon Language Agents", "authors": "Renwei Meng", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.05263", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-03", "updated_at": "2026-06-03", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "planning", "rag", "reasoning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.LG", "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "14", "collection_queries": "language-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-05414-when-evidence-is-sparse-weakly-supervised-early-failure-alerting-in-dialogs-and-.md", "title": "\"When Evidence is Sparse: Weakly Supervised Early Failure Alerting in Dialogs and LLM-Agent Trajectories\"", "type": "paper", "meta": { "type": "paper", "title": "\"When Evidence is Sparse: Weakly Supervised Early Failure Alerting in Dialogs and LLM-Agent Trajectories\"", "authors": "Avinash Baidya, Xinran Liang, Ruocheng Guo, Xiang Gao, Kamalika Das", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.05414", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-03", "updated_at": "2026-06-03", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "planning", "rag", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CL", "cs.AI", "cs.HC", "cs.LG" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "14", "collection_queries": "planning-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-05436-ten-headache-specialists-versus-artificial-intelligence-for-clinical-literature-.md", "title": "\"Ten Headache Specialists versus Artificial Intelligence for Clinical Literature Summarization: A Critical Evaluation and Comparison\"", "type": "paper", "meta": { "type": "paper", "title": "\"Ten Headache Specialists versus Artificial Intelligence for Clinical Literature Summarization: A Critical Evaluation and Comparison\"", "authors": "Alejandro Lozano, Keiko Ihara, Ping-Hao Yang, Carrie E. Robertson, Jennifer Stern, Allan Purdy, Hsiangkuo Yuan, Pengfei Zhang, et al.", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.05436", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-03", "updated_at": "2026-06-03", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "computer-use", "rag", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI", "cs.CL", "cs.IR" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "13", "collection_queries": "rag-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-05463-psebench-a-controllable-and-verifiable-benchmark-for-evaluating-llms-in-patient-.md", "title": "\"PSEBench: A Controllable and Verifiable Benchmark for Evaluating LLMs in Patient Safety Event Triage\"", "type": "paper", "meta": { "type": "paper", "title": "\"PSEBench: A Controllable and Verifiable Benchmark for Evaluating LLMs in Patient Safety Event Triage\"", "authors": "Keqi Han, Ryan Young, Annabel Strauss, Lindsey Hughes, Katharine M. Nesbitt, Nicole Schueler, Che Ngufor, Carl Yang, et al.", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.05463", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-03", "updated_at": "2026-06-09", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "reasoning", "tool-use", "workflow-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "15", "collection_queries": "agent-evaluation" } }, { "collection": "papers", "path": "papers/items/2026-2606-05548-adk-arena-evaluating-agent-development-kits-via-llm-as-a-developer.md", "title": "\"ADK Arena: Evaluating Agent Development Kits via LLM-as-a-Developer\"", "type": "paper", "meta": { "type": "paper", "title": "\"ADK Arena: Evaluating Agent Development Kits via LLM-as-a-Developer\"", "authors": "Jintao Huang, Xiaomin Li, Gaurav Mittal, Yu Hu", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.05548", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-04", "updated_at": "2026-06-04", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "coding-agent", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.SE", "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "14", "collection_queries": "agent-evaluation, autonomous-agent-llm" } }, { "collection": "papers", "path": "papers/items/2026-2606-05558-autoregressive-diffusion-world-models-for-off-policy-evaluation-of-llm-agents.md", "title": "Autoregressive Diffusion World Models for Off-Policy Evaluation of LLM Agents", "type": "paper", "meta": { "type": "paper", "title": "Autoregressive Diffusion World Models for Off-Policy Evaluation of LLM Agents", "authors": "Kaixuan Liu, Guojun Xiong, Weinan Zhang, Shengpu Tang", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.05558", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-04", "updated_at": "2026-06-04", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "computer-use", "tool-use", "world-model" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.LG" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "18", "collection_queries": "agent-evaluation" } }, { "collection": "papers", "path": "papers/items/2026-2606-05622-adaplanbench-evaluating-adaptive-planning-in-large-language-model-agents-under-w.md", "title": "\"AdaPlanBench: Evaluating Adaptive Planning in Large Language Model Agents under World and User Constraints\"", "type": "paper", "meta": { "type": "paper", "title": "\"AdaPlanBench: Evaluating Adaptive Planning in Large Language Model Agents under World and User Constraints\"", "authors": "Jiayu Liu, Cheng Qian, Zhenhailong Wang, Bingxuan Li, Jiateng Liu, Heng Wang, Jeonghwan Kim, Yumeng Wang, et al.", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.05622", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-04", "updated_at": "2026-06-04", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "planning", "rag", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CL" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "13", "collection_queries": "planning-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-05658-agent-orchestrated-adaptive-rag-a-comparative-study-on-structured-and-multi-hop-.md", "title": "\"Agent-Orchestrated Adaptive RAG: A Comparative Study on Structured and Multi-Hop Retrieval\"", "type": "paper", "meta": { "type": "paper", "title": "\"Agent-Orchestrated Adaptive RAG: A Comparative Study on Structured and Multi-Hop Retrieval\"", "authors": "Anuj Maharjan, Devinder Kaur, Richard Molyet", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.05658", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-04", "updated_at": "2026-06-04", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "rag", "reasoning" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.IR", "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "13", "collection_queries": "rag-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-05684-adamem-test-time-adaptive-memory-for-language-agents.md", "title": "\"AdaMEM: Test-Time Adaptive Memory for Language Agents\"", "type": "paper", "meta": { "type": "paper", "title": "\"AdaMEM: Test-Time Adaptive Memory for Language Agents\"", "authors": "Yunxiang Zhang, Yiheng Li, Ali Payani, Lu Wang", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.05684", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-04", "updated_at": "2026-06-04", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "computer-use", "memory", "planning", "rag", "reasoning" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "19", "collection_queries": "agent-memory, language-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-05711-beyond-tokens-a-unified-framework-for-latent-communication-in-llm-based-multi-ag.md", "title": "\"Beyond tokens: a unified framework for latent communication in LLM-based multi-agent systems\"", "type": "paper", "meta": { "type": "paper", "title": "\"Beyond tokens: a unified framework for latent communication in LLM-based multi-agent systems\"", "authors": "Yingzhuo Liu", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.05711", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-04", "updated_at": "2026-06-05", "status": "queued", "relevance": "high", "topics": [ "agent-safety", "computer-use", "multi-agent", "planning", "reasoning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CL" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "16", "collection_queries": "language-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-05805-from-risk-classification-to-action-plan-remediation-a-guardrail-feedback-driven-.md", "title": "\"From Risk Classification to Action Plan Remediation: A Guardrail Feedback Driven Framework for LLM Agents\"", "type": "paper", "meta": { "type": "paper", "title": "\"From Risk Classification to Action Plan Remediation: A Guardrail Feedback Driven Framework for LLM Agents\"", "authors": "Yuhao Sun, Jiacheng Zhang, Shaanan Cohney, Zhexin Zhang, Feng Liu, Xingliang Yuan", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.05805", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-04", "updated_at": "2026-06-04", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "computer-use", "planning", "rag", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "17", "collection_queries": "planning-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-06054-beyond-similarity-trustworthy-memory-search-for-personal-ai-agents.md", "title": "\"Beyond Similarity: Trustworthy Memory Search for Personal AI Agents\"", "type": "paper", "meta": { "type": "paper", "title": "\"Beyond Similarity: Trustworthy Memory Search for Personal AI Agents\"", "authors": "Jiawen Zhang, Kejia Chen, Jiachen Ma, Yangfan Hu, Lipeng He, Yechao Zhang, Jian Liu, Xiaohu Yang, et al.", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.06054", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-04", "updated_at": "2026-06-04", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "memory", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "17", "collection_queries": "agent-memory" } }, { "collection": "papers", "path": "papers/items/2026-2606-06090-beyond-semantic-organization-memory-as-execution-state-management-for-long-horiz.md", "title": "\"Beyond Semantic Organization: Memory as Execution State Management for Long-Horizon Agents\"", "type": "paper", "meta": { "type": "paper", "title": "\"Beyond Semantic Organization: Memory as Execution State Management for Long-Horizon Agents\"", "authors": "Yaoqi Chen, Haibin Lai, Yuru Feng, Chuyu Han, Qianxi Zhang, Baotong Lu, Menghao Li, Xinjiang Wang, et al.", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.06090", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-04", "updated_at": "2026-06-04", "status": "skimmed", "relevance": "high", "topics": [ "computer-use", "memory", "planning", "rag", "tool-use" ], "methods": [ "hierarchical-execution-state-tree", "grow-compress-maintain-revise", "error-branch-isolation" ], "benchmarks": [ "MemoryArena" ], "models": [], "datasets": [ "cs.AI" ], "related_concepts": [ "execution-state", "long-horizon-agent" ], "related_jobs": [], "related_experiments": [ "KC-001-agent-memory-pilot" ], "related_projects": [ "learning/agent-memory" ], "collection_score": "16", "collection_queries": "agent-memory, rag-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-06388-humans-almanac-a-human-collaboration-dataset-of-action-level-mental-model-annota.md", "title": "\"Humans' ALMANAC: A Human Collaboration Dataset of Action-Level Mental Model Annotations for Agent Collaboration\"", "type": "paper", "meta": { "type": "paper", "title": "\"Humans' ALMANAC: A Human Collaboration Dataset of Action-Level Mental Model Annotations for Agent Collaboration\"", "authors": "Jiaju Chen, Yuxuan Lu, Jiayi Su, Chaoran Chen, Songlin Xiao, Zheng Zhang, Yun Wang, Yunyao Li, et al.", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.06388", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-04", "updated_at": "2026-06-06", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "computer-use", "multi-agent", "planning", "reasoning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI", "cs.CL" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "16", "collection_queries": "planning-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-06399-collabsim-a-cscw-grounded-methodology-for-investigating-collaborative-competence.md", "title": "\"CollabSim: A CSCW-Grounded Methodology for Investigating Collaborative Competence of LLM Agents through Controlled Multi-Agent Experiments\"", "type": "paper", "meta": { "type": "paper", "title": "\"CollabSim: A CSCW-Grounded Methodology for Investigating Collaborative Competence of LLM Agents through Controlled Multi-Agent Experiments\"", "authors": "Jiaju Chen, Bo Sun, Yuxuan Lu, Yun Wang, Dakuo Wang, Bingsheng Yao", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.06399", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-04", "updated_at": "2026-06-06", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "multi-agent", "planning", "reasoning", "tool-use", "world-model" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CL" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "22", "collection_queries": "planning-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-06448-agent-memory-characterization-and-system-implications-of-stateful-long-horizon-w.md", "title": "\"Agent Memory: Characterization and System Implications of Stateful Long-Horizon Workloads\"", "type": "paper", "meta": { "type": "paper", "title": "\"Agent Memory: Characterization and System Implications of Stateful Long-Horizon Workloads\"", "authors": "Yasmine Omri, Ziyu Gan, Zachary Broveak, Robin Geens, Zexue He, Alex Pentland, Marian Verhelst, Tsachy Weissman, et al.", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.06448", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-04", "updated_at": "2026-06-04", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "memory", "planning", "rag", "reasoning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "19", "collection_queries": "agent-memory" } }, { "collection": "papers", "path": "papers/items/2026-2606-06462-benchmark-everything-everywhere-all-at-once.md", "title": "Benchmark Everything Everywhere All at Once", "type": "paper", "meta": { "type": "paper", "title": "Benchmark Everything Everywhere All at Once", "authors": "Shiyun Xiong, Dongming Wu, Peiwen Sun, Yuang Ai, Bokang Yang, Wencheng Han, Xiao-Hui Li, Xiangyu Yue", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.06462", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-04", "updated_at": "2026-06-04", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "coding-agent", "reasoning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "14", "collection_queries": "autonomous-agent-llm" } }, { "collection": "papers", "path": "papers/items/2026-2606-06473-mlevolve-a-self-evolving-framework-for-automated-machine-learning-algorithm-disc.md", "title": "\"MLEvolve: A Self-Evolving Framework for Automated Machine Learning Algorithm Discovery\"", "type": "paper", "meta": { "type": "paper", "title": "\"MLEvolve: A Self-Evolving Framework for Automated Machine Learning Algorithm Discovery\"", "authors": "Shangheng Du, Xiangchao Yan, Jinxin Shi, Zongsheng Cao, Shiyang Feng, Zichen Liang, Boyuan Sun, Tianshuo Peng, et al.", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.06473", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-04", "updated_at": "2026-06-04", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "coding-agent", "memory", "multi-agent", "planning", "rag" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI", "cs.CL" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "14", "collection_queries": "planning-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-07314-qbuglm-an-agentic-benchmarking-framework-for-llm-based-quantum-software-debuggin.md", "title": "\"QBugLM: An Agentic Benchmarking Framework for LLM-based Quantum Software Debugging\"", "type": "paper", "meta": { "type": "paper", "title": "\"QBugLM: An Agentic Benchmarking Framework for LLM-based Quantum Software Debugging\"", "authors": "An B. B. Pham, Hoa T. Nguyen, Muhammad Usman", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.07314", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-05", "updated_at": "2026-06-05", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "coding-agent", "multi-agent", "reasoning", "world-model" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.SE", "cs.ET", "quant-ph" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "19", "collection_queries": "agent-evaluation" } }, { "collection": "papers", "path": "papers/items/2026-2606-07379-do-coding-agents-deceive-us-detecting-and-preventing-cheating-via-capped-evaluat.md", "title": "Do Coding Agents Deceive Us? Detecting and Preventing Cheating via Capped Evaluation with Randomized Tests", "type": "paper", "meta": { "type": "paper", "title": "Do Coding Agents Deceive Us? Detecting and Preventing Cheating via Capped Evaluation with Randomized Tests", "authors": "Thanawat Lodkaew, Johannes Ackermann, Soichiro Nishimori, Nontawat Charoenphakdee, Masashi Sugiyama, Takashi Ishida", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.07379", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-05", "updated_at": "2026-06-08", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "coding-agent", "rag" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.LG", "cs.AI", "cs.CL", "stat.ME" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "14", "collection_queries": "agent-evaluation" } }, { "collection": "papers", "path": "papers/items/2026-2606-07402-m-3-exam-benchmarking-multimodal-memory-for-realistic-user-agent-interactions.md", "title": "\"M$^3$Exam: Benchmarking Multimodal Memory for Realistic User-Agent Interactions\"", "type": "paper", "meta": { "type": "paper", "title": "\"M$^3$Exam: Benchmarking Multimodal Memory for Realistic User-Agent Interactions\"", "authors": "Zhengjun Huang, Wenxuan Liu, Zhoujin Tian, Wei Chen, Junle Chen, Yuqian Wu, Fangyuan Zhang, Qintian Guo, et al.", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.07402", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-05", "updated_at": "2026-06-05", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "memory", "rag", "reasoning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CL" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "20", "collection_queries": "language-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-07591-researchclawbench-a-benchmark-for-end-to-end-autonomous-scientific-research.md", "title": "\"ResearchClawBench: A Benchmark for End-to-End Autonomous Scientific Research\"", "type": "paper", "meta": { "type": "paper", "title": "\"ResearchClawBench: A Benchmark for End-to-End Autonomous Scientific Research\"", "authors": "Wanghan Xu, Shuo Li, Tianlin Ye, Qinglong Cao, Yixin Chen, Hengjian Gao, Yiheng Wang, Qi Li, et al.", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.07591", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-05-28", "updated_at": "2026-07-03", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "coding-agent", "rag" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.LG", "cs.AI", "cs.CL" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "13", "collection_queries": "autonomous-agent-llm" } }, { "collection": "papers", "path": "papers/items/2026-2606-07595-visualleakbench-reproducible-action-boundary-propagation-failures-in-vision-lang.md", "title": "\"VisualLeakBench: Reproducible Action-Boundary Propagation Failures in Vision-Language Agents\"", "type": "paper", "meta": { "type": "paper", "title": "\"VisualLeakBench: Reproducible Action-Boundary Propagation Failures in Vision-Language Agents\"", "authors": "Youting Wang, Yuan Tang, Yitian Qian, Chen Zhao", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.07595", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-05-29", "updated_at": "2026-05-29", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "memory", "tool-use", "workflow-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CV", "cs.AI", "cs.IR" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "16", "collection_queries": "language-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-07682-swe-marathon-can-agents-autonomously-complete-ultra-long-horizon-software-work.md", "title": "\"SWE-Marathon: Can Agents Autonomously Complete Ultra-Long-Horizon Software Work?\"", "type": "paper", "meta": { "type": "paper", "title": "\"SWE-Marathon: Can Agents Autonomously Complete Ultra-Long-Horizon Software Work?\"", "authors": "Rishi Desai, Jesse Hu, Joan Cabezas, Neel Harsola, Pratyush Shukla, Roey Ben Chaim, Adnan El Assadi, Omkaar Mukund Kamath, et al.", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.07682", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-05", "updated_at": "2026-06-05", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "coding-agent", "memory", "planning", "rag", "reasoning", "workflow-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.SE", "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "17", "collection_queries": "agent-evaluation" } }, { "collection": "papers", "path": "papers/items/2026-2606-07711-rosetta-memory-adaptive-memory-for-cross-llm-agents.md", "title": "\"Rosetta Memory: Adaptive Memory for Cross-LLM Agents\"", "type": "paper", "meta": { "type": "paper", "title": "\"Rosetta Memory: Adaptive Memory for Cross-LLM Agents\"", "authors": "Hao Yang, Shiqi Shen, Haoxuan Li, Zhipeng Wang, Zhi Gong, Xu Chen", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.07711", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-05", "updated_at": "2026-06-05", "status": "queued", "relevance": "high", "topics": [ "coding-agent", "memory", "planning", "reasoning" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.LG", "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "17", "collection_queries": "planning-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-07836-agentic-multi-fidelity-learning-of-quasiparticle-and-excitonic-properties.md", "title": "Agentic multi-fidelity learning of quasiparticle and excitonic properties", "type": "paper", "meta": { "type": "paper", "title": "Agentic multi-fidelity learning of quasiparticle and excitonic properties", "authors": "Arnab Neogi, Aaron Forde, Christopher A. Lane, Sergei Tretiak, Jian-Xin Zhu", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.07836", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-05", "updated_at": "2026-06-05", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "computer-use", "rag", "workflow-agent", "world-model" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cond-mat.mtrl-sci", "cond-mat.stat-mech", "cs.AI", "physics.comp-ph", "quant-ph" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "13", "collection_queries": "agent-evaluation" } }, { "collection": "papers", "path": "papers/items/2026-2606-07867-the-cold-start-safety-gap-in-llm-agents.md", "title": "The Cold-Start Safety Gap in LLM Agents", "type": "paper", "meta": { "type": "paper", "title": "The Cold-Start Safety Gap in LLM Agents", "authors": "Chung-En Sun, Linbo Liu, Tsui-Wei Weng", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.07867", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-05", "updated_at": "2026-06-05", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CL" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "17", "collection_queries": "agent-safety" } }, { "collection": "papers", "path": "papers/items/2026-2606-08162-silent-failure-in-llm-agent-systems-the-entropy-principle-and-the-inevitable-dis.md", "title": "\"Silent Failure in LLM Agent Systems: The Entropy Principle and the Inevitable Disorder of Autonomous Agents\"", "type": "paper", "meta": { "type": "paper", "title": "\"Silent Failure in LLM Agent Systems: The Entropy Principle and the Inevitable Disorder of Autonomous Agents\"", "authors": "Dexing Liu", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.08162", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-06", "updated_at": "2026-06-06", "status": "queued", "relevance": "high", "topics": [ "memory", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.MA" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "13", "collection_queries": "autonomous-agent-llm" } }, { "collection": "papers", "path": "papers/items/2026-2606-08172-the-governance-of-human-llm-interaction-safety-gating-civility-steering-and-affe.md", "title": "\"The Governance of Human-LLM Interaction: Safety Gating, Civility Steering, and Affective Default Lock-In\"", "type": "paper", "meta": { "type": "paper", "title": "\"The Governance of Human-LLM Interaction: Safety Gating, Civility Steering, and Affective Default Lock-In\"", "authors": "Manuele Reani, Hongjian Zhang, Hongyu Tian", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.08172", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-06", "updated_at": "2026-06-06", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "computer-use", "multi-agent", "planning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.HC", "cs.AI", "cs.CY" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "13", "collection_queries": "agent-evaluation" } }, { "collection": "papers", "path": "papers/items/2026-2606-08274-toward-human-centered-multi-agent-systems-integrating-cognition-culture-values-a.md", "title": "\"Toward Human-Centered Multi-Agent Systems: Integrating Cognition, Culture, Values, and Cooperation in AI Agents\"", "type": "paper", "meta": { "type": "paper", "title": "\"Toward Human-Centered Multi-Agent Systems: Integrating Cognition, Culture, Values, and Cooperation in AI Agents\"", "authors": "Safia Baloch, Rahemeen Khan", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.08274", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-06", "updated_at": "2026-06-06", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "computer-use", "multi-agent", "planning", "tool-use", "workflow-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.MA" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "24", "collection_queries": "autonomous-agent-llm, planning-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-08340-benchmarking-open-ended-multi-agent-coordination-in-language-agents.md", "title": "Benchmarking Open-Ended Multi-Agent Coordination in Language Agents", "type": "paper", "meta": { "type": "paper", "title": "Benchmarking Open-Ended Multi-Agent Coordination in Language Agents", "authors": "Kale-ab Abebe Tessera, Andras Szecsenyi, Cameron Barker, Alexander Rutherford, Davide Paglieri, Aidan Scannell, Henry Gouk, Elliot J. Crowley, et al.", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.08340", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-06", "updated_at": "2026-06-06", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "memory", "multi-agent", "planning", "rag", "reasoning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI", "cs.LG", "cs.MA" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "28", "collection_queries": "autonomous-agent-llm, language-agent, planning-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-08531-vesta-a-fully-automated-scenario-generation-and-safety-evaluation-framework-for-.md", "title": "\"VESTA: A Fully Automated Scenario Generation and Safety Evaluation Framework for LLM Agents\"", "type": "paper", "meta": { "type": "paper", "title": "\"VESTA: A Fully Automated Scenario Generation and Safety Evaluation Framework for LLM Agents\"", "authors": "Lu Jia, Haibo Tong, Feifei Zhao, Jindong Li, Dongqi Liang, Ping Wu, Qian Zhang, Yi Zeng", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.08531", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-07", "updated_at": "2026-06-07", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "memory", "rag", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "20", "collection_queries": "agent-safety" } }, { "collection": "papers", "path": "papers/items/2026-2606-08625-from-holistic-evaluation-to-structured-criteria-rubrics-across-the-evolving-llm-.md", "title": "\"From Holistic Evaluation to Structured Criteria: Rubrics Across the Evolving LLM Landscape\"", "type": "paper", "meta": { "type": "paper", "title": "\"From Holistic Evaluation to Structured Criteria: Rubrics Across the Evolving LLM Landscape\"", "authors": "Hao Chen, Ziyu Han, Yukun Yan, Qingfu Zhu, Maosong Sun, Wanxiang Che", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.08625", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-07", "updated_at": "2026-07-01", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "computer-use", "reasoning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CL" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "14", "collection_queries": "autonomous-agent-llm" } }, { "collection": "papers", "path": "papers/items/2026-2606-08790-rails-verification-native-clearing-for-agentic-commerce.md", "title": "\"RAILS: Verification-Native Clearing For Agentic Commerce\"", "type": "paper", "meta": { "type": "paper", "title": "\"RAILS: Verification-Native Clearing For Agentic Commerce\"", "authors": "Adrian de Valois-Franklin, Alex Bogdan", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.08790", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-07", "updated_at": "2026-06-07", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI", "cs.CR", "cs.MA" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "14", "collection_queries": "autonomous-agent-llm" } }, { "collection": "papers", "path": "papers/items/2026-2606-08960-hardening-agent-benchmarks-with-adversarial-hacker-fixer-loops.md", "title": "Hardening Agent Benchmarks with Adversarial Hacker-Fixer Loops", "type": "paper", "meta": { "type": "paper", "title": "Hardening Agent Benchmarks with Adversarial Hacker-Fixer Loops", "authors": "Ziqian Zhong, Ivgeni Segal, Ivan Bercovich, Shashwat Saxena, Kexun Zhang, Aditi Raghunathan", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.08960", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-08", "updated_at": "2026-06-08", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "reasoning" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CR", "cs.AI", "cs.LG", "cs.MA" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "13", "collection_queries": "agent-evaluation" } }, { "collection": "papers", "path": "papers/items/2026-2606-09037-a-multi-agent-system-for-ipmsm-design-optimization-via-an-fea-ai-hybrid-approach.md", "title": "A Multi-Agent System for IPMSM Design Optimization via an FEA-AI Hybrid Approach", "type": "paper", "meta": { "type": "paper", "title": "A Multi-Agent System for IPMSM Design Optimization via an FEA-AI Hybrid Approach", "authors": "Jinseong Han, Sunwoong Yang, Namwoo Kang", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.09037", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-08", "updated_at": "2026-06-08", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "multi-agent", "planning", "rag", "reasoning", "workflow-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI", "cs.MA" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "16", "collection_queries": "rag-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-09071-reflect-intervention-supported-error-attribution-for-silent-failures-in-llm-agen.md", "title": "\"REFLECT: Intervention-Supported Error Attribution for Silent Failures in LLM Agent Traces\"", "type": "paper", "meta": { "type": "paper", "title": "\"REFLECT: Intervention-Supported Error Attribution for Silent Failures in LLM Agent Traces\"", "authors": "Xiaofeng Lin, Yingxu Wang, Tung Sum Thomas Kwok, Daniel Guo, Sahil Arun Nale, Charles Fleming, Guang Cheng", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.09071", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-08", "updated_at": "2026-06-08", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "planning", "reasoning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "15", "collection_queries": "planning-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-09198-mass-deep-research-for-social-sciences-with-memory-augmented-social-simulation.md", "title": "\"MASS: Deep Research for Social Sciences with Memory-Augmented Social Simulation\"", "type": "paper", "meta": { "type": "paper", "title": "\"MASS: Deep Research for Social Sciences with Memory-Augmented Social Simulation\"", "authors": "Yongrui Liu, Deyi Xiong", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.09198", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-08", "updated_at": "2026-06-08", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "computer-use", "memory", "planning", "rag", "world-model" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "14", "collection_queries": "agent-memory" } }, { "collection": "papers", "path": "papers/items/2026-2606-09316-anything2skill-compiling-external-knowledge-into-reusable-skills-for-agents.md", "title": "\"Anything2Skill: Compiling External Knowledge into Reusable Skills for Agents\"", "type": "paper", "meta": { "type": "paper", "title": "\"Anything2Skill: Compiling External Knowledge into Reusable Skills for Agents\"", "authors": "Qianjun Pan, Yutao Yang, Junsong Li, Jie Zhou, Kai Chen, Xin Li, Qin Chen, Liang He", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.09316", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-08", "updated_at": "2026-06-19", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "computer-use", "memory", "planning", "rag", "tool-use", "workflow-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "13", "collection_queries": "rag-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-09399-runagent-superbrowser-a-theory-of-autonomous-web-navigation-grounded-in-human-br.md", "title": "\"RunAgent SuperBrowser: A Theory of Autonomous Web Navigation Grounded in Human Browsing Behaviour\"", "type": "paper", "meta": { "type": "paper", "title": "\"RunAgent SuperBrowser: A Theory of Autonomous Web Navigation Grounded in Human Browsing Behaviour\"", "authors": "Radeen Mostafa, Sawradip Saha", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.09399", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-08", "updated_at": "2026-06-08", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "computer-use", "embodied-agent", "memory", "planning", "reasoning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "15", "collection_queries": "web-gui-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-09426-weavebench-a-long-horizon-real-world-benchmark-for-computer-use-agents-with-hybr.md", "title": "\"WeaveBench: A Long-Horizon, Real-World Benchmark for Computer-Use Agents with Hybrid Interfaces\"", "type": "paper", "meta": { "type": "paper", "title": "\"WeaveBench: A Long-Horizon, Real-World Benchmark for Computer-Use Agents with Hybrid Interfaces\"", "authors": "Wanli Li, Bowen Zhou, Yunyao Yu, Zhou Xu, Yifan Yang, Dongsheng Li, Caihua Shan", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.09426", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-08", "updated_at": "2026-07-06", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "computer-use", "planning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "13", "collection_queries": "web-gui-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-09447-aliyunconsoleagent-training-web-agents-in-real-world-cloud-environments-via-dist.md", "title": "\"AliyunConsoleAgent: Training Web Agents in Real-World Cloud Environments via Distillation and Reinforcement Learning\"", "type": "paper", "meta": { "type": "paper", "title": "\"AliyunConsoleAgent: Training Web Agents in Real-World Cloud Environments via Distillation and Reinforcement Learning\"", "authors": "Bojie Rong, Zheyu Shen, Qiaoping Wang, Pengfei Kang, Yang Xu, Yawen Wei, Hanyu Wu, Zhi Zhao, et al.", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.09447", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-08", "updated_at": "2026-06-08", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "computer-use", "rag", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "13", "collection_queries": "web-gui-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-09483-memory-beyond-recall-a-dual-process-cognitive-memory-system-for-self-evolving-ll.md", "title": "\"Memory Beyond Recall: A Dual-Process Cognitive Memory System for Self-Evolving LLM Agents\"", "type": "paper", "meta": { "type": "paper", "title": "\"Memory Beyond Recall: A Dual-Process Cognitive Memory System for Self-Evolving LLM Agents\"", "authors": "Tianxiang Fei, Mingyang Song, Mao Zheng, Xiang Yu", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.09483", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-08", "updated_at": "2026-06-08", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "memory", "rag", "reasoning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CL", "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "19", "collection_queries": "agent-memory" } }, { "collection": "papers", "path": "papers/items/2026-2606-09549-secureclaw-clawing-back-control-of-llm-agents.md", "title": "\"SecureClaw: Clawing Back Control of LLM Agents\"", "type": "paper", "meta": { "type": "paper", "title": "\"SecureClaw: Clawing Back Control of LLM Agents\"", "authors": "Yuhan Ma, Stefan Schmid", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.09549", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-08", "updated_at": "2026-06-08", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "planning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CR", "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "15", "collection_queries": "agent-safety, planning-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-09738-hdsl-a-hierarchical-domain-specific-language-for-structured-3d-indoor-scene-gene.md", "title": "\"HDSL: A Hierarchical Domain-Specific Language for Structured 3D Indoor Scene Generation and Localized Editing with LLM Agents\"", "type": "paper", "meta": { "type": "paper", "title": "\"HDSL: A Hierarchical Domain-Specific Language for Structured 3D Indoor Scene Generation and Localized Editing with LLM Agents\"", "authors": "Letian Li, Chao Shen, Shuzhao Xie, Chenghao Gu, ZhengXiao He, Yu Meng, Xin Yang, Wenyuan Jiang, et al.", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.09738", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-08", "updated_at": "2026-06-08", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "coding-agent", "planning", "rag" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CV" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "16", "collection_queries": "planning-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-09764-iosworld-a-benchmark-for-personally-intelligent-phone-agents.md", "title": "\"iOSWorld: A Benchmark for Personally Intelligent Phone Agents\"", "type": "paper", "meta": { "type": "paper", "title": "\"iOSWorld: A Benchmark for Personally Intelligent Phone Agents\"", "authors": "Lawrence Keunho Jang, Mareks Woodside, Geronimo Carom, Andrew Keunwoo Jang, Jing Yu Koh, Ruslan Salakhutdinov", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.09764", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-08", "updated_at": "2026-06-08", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "computer-use", "memory", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.LG", "cs.CL" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "15", "collection_queries": "agent-evaluation, web-gui-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-09774-auto-configuring-scientific-simulators-with-lightweight-coding-agent-adapters.md", "title": "Auto-Configuring Scientific Simulators with Lightweight Coding-Agent Adapters", "type": "paper", "meta": { "type": "paper", "title": "Auto-Configuring Scientific Simulators with Lightweight Coding-Agent Adapters", "authors": "Matthew Ho, Brian Liu, Jixuan Chen, Audrey Wang, Lianhui Qin", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.09774", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-08", "updated_at": "2026-06-25", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "coding-agent", "memory", "rag", "tool-use", "world-model" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI", "cs.CL" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "14", "collection_queries": "agent-memory" } }, { "collection": "papers", "path": "papers/items/2026-2606-09863-from-confident-closing-to-silent-failure-characterizing-false-success-in-llm-age.md", "title": "\"From Confident Closing to Silent Failure: Characterizing False Success in LLM Agents\"", "type": "paper", "meta": { "type": "paper", "title": "\"From Confident Closing to Silent Failure: Characterizing False Success in LLM Agents\"", "authors": "Laksh Advani", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.09863", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-01", "updated_at": "2026-06-01", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "coding-agent", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.LG" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "14", "collection_queries": "agent-evaluation" } }, { "collection": "papers", "path": "papers/items/2026-2606-09961-3spo-state-score-supervised-policy-optimization-for-llm-agents.md", "title": "\"3SPO: State-Score-Supervised Policy Optimization for LLM Agents\"", "type": "paper", "meta": { "type": "paper", "title": "\"3SPO: State-Score-Supervised Policy Optimization for LLM Agents\"", "authors": "Yu Han, Kailing Li, Yang Jiao, Yulin Dai, Yuqian Fu, Linhai Zhuo, Tianwen Qian", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.09961", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-08", "updated_at": "2026-06-08", "status": "queued", "relevance": "high", "topics": [ "computer-use", "planning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.LG", "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "13", "collection_queries": "autonomous-agent-llm" } }, { "collection": "papers", "path": "papers/items/2026-2606-10209-less-context-better-agents-efficient-context-engineering-for-long-horizon-tool-u.md", "title": "\"Less Context, Better Agents: Efficient Context Engineering for Long-Horizon Tool-Using LLM Agents\"", "type": "paper", "meta": { "type": "paper", "title": "\"Less Context, Better Agents: Efficient Context Engineering for Long-Horizon Tool-Using LLM Agents\"", "authors": "Abhilasha Lodha, Mahsa Pahlavikhah Varnosfaderani, Abir Chakraborty, Abhinav Mithal", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.10209", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-08", "updated_at": "2026-06-08", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "planning", "rag", "tool-use", "workflow-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI", "cs.LG", "cs.SE" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "16", "collection_queries": "autonomous-agent-llm" } }, { "collection": "papers", "path": "papers/items/2026-2606-10304-mirage-a-polarity-flipping-encoding-subspace-in-llm-agents.md", "title": "\"MIRAGE: A Polarity-Flipping Encoding Subspace in LLM Agents\"", "type": "paper", "meta": { "type": "paper", "title": "\"MIRAGE: A Polarity-Flipping Encoding Subspace in LLM Agents\"", "authors": "Pratibha Revankar, Kargi Chauhan, Jihye Kim, Sadiba Nusrat Nur, Vincent Siu, Chenguang Wang", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.10304", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-09", "updated_at": "2026-06-09", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "coding-agent", "computer-use", "planning", "rag", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CL" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "17", "collection_queries": "planning-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-10316-tabclaw-an-interactive-and-self-evolving-agent-for-spreadsheet-manipulation-and-.md", "title": "\"TabClaw: An Interactive and Self-Evolving Agent for Spreadsheet Manipulation and Table Reasoning\"", "type": "paper", "meta": { "type": "paper", "title": "\"TabClaw: An Interactive and Self-Evolving Agent for Spreadsheet Manipulation and Table Reasoning\"", "authors": "Mingyue Cheng, Shuo Yu, Daoyu Wang, Qingchuan Li, Xiaoyu Tao, Qingyang Mao, Yitong Zhou, Qi Liu", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.10316", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-09", "updated_at": "2026-06-09", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "memory", "planning", "reasoning", "tool-use", "workflow-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CL" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "16", "collection_queries": "planning-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-10381-agentic-hybrid-rag-for-evidence-grounded-muon-collider-analysis.md", "title": "Agentic Hybrid RAG for Evidence-Grounded Muon Collider Analysis", "type": "paper", "meta": { "type": "paper", "title": "Agentic Hybrid RAG for Evidence-Grounded Muon Collider Analysis", "authors": "Ruobing Jiang, Dawei Fu, Cheng Jiang, Tianyi Yang, Zijian Wang, Youpeng Wu, Yong Ban, Yajun Mao, et al.", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.10381", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-09", "updated_at": "2026-06-09", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "rag", "reasoning", "tool-use", "workflow-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "hep-ex", "cs.AI", "cs.CL", "cs.IR", "physics.ins-det" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "15", "collection_queries": "rag-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-10394-stage-claw-automated-state-based-agent-benchmarking-for-realistic-scenarios.md", "title": "\"STAGE-Claw: Automated State-based Agent Benchmarking for Realistic Scenarios\"", "type": "paper", "meta": { "type": "paper", "title": "\"STAGE-Claw: Automated State-based Agent Benchmarking for Realistic Scenarios\"", "authors": "Sirui Liang, Bohan Yu, Peiyu Wang, Shiguang Guo, Wenxing Hu, Pengfei Cao, Jian Zhao, Cao Liu, et al.", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.10394", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-09", "updated_at": "2026-06-09", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "13", "collection_queries": "agent-evaluation" } }, { "collection": "papers", "path": "papers/items/2026-2606-10423-webchallenger-a-reliable-and-efficient-generalist-web-agent.md", "title": "\"WebChallenger: A Reliable and Efficient Generalist Web Agent\"", "type": "paper", "meta": { "type": "paper", "title": "\"WebChallenger: A Reliable and Efficient Generalist Web Agent\"", "authors": "Jayoo Hwang, Xiaowen Zhang, Vedant Padwal", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.10423", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-09", "updated_at": "2026-06-09", "status": "queued", "relevance": "high", "topics": [ "computer-use", "embodied-agent", "memory", "reasoning", "tool-use", "workflow-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CL" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "15", "collection_queries": "web-gui-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-10507-hipif-hierarchical-planning-and-information-folding-for-long-horizon-llm-agent-l.md", "title": "\"HIPIF: Hierarchical Planning and Information Folding for Long-Horizon LLM Agent Learning\"", "type": "paper", "meta": { "type": "paper", "title": "\"HIPIF: Hierarchical Planning and Information Folding for Long-Horizon LLM Agent Learning\"", "authors": "Juncheng Diao, Zhicong Lu, Peiguang Li, Yongwei Zhou, Changyuan Tian, Qingbin Li, Rongxiang Weng, Jingang Wang, et al.", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.10507", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-09", "updated_at": "2026-06-09", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "computer-use", "memory", "planning", "reasoning" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "19", "collection_queries": "agent-evaluation, autonomous-agent-llm, planning-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-10532-activemem-distributed-active-memory-for-long-horizon-llm-reasoning.md", "title": "\"ActiveMem: Distributed Active Memory for Long-Horizon LLM Reasoning\"", "type": "paper", "meta": { "type": "paper", "title": "\"ActiveMem: Distributed Active Memory for Long-Horizon LLM Reasoning\"", "authors": "Yunhan Jiang, Wenbin Duan, Shasha Guo, Liang Pang, Xiaoqian Sun, Huawei Shen", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.10532", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-09", "updated_at": "2026-06-09", "status": "queued", "relevance": "high", "topics": [ "agent-safety", "memory", "planning", "reasoning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "13", "collection_queries": "agent-memory" } }, { "collection": "papers", "path": "papers/items/2026-2606-10577-agenticnav-zero-shot-vision-and-language-navigation-as-a-tool-calling-harness.md", "title": "\"AgenticNav: Zero-Shot Vision-and-Language Navigation as a Tool-Calling Harness\"", "type": "paper", "meta": { "type": "paper", "title": "\"AgenticNav: Zero-Shot Vision-and-Language Navigation as a Tool-Calling Harness\"", "authors": "Yijian Li, Changze Li, Hantian Shi, Jiaying Luo, Jiyuan Cai, Ming Yang, Tong Qin", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.10577", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-09", "updated_at": "2026-06-09", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "embodied-agent", "memory", "rag", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.RO" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "17", "collection_queries": "agent-memory" } }, { "collection": "papers", "path": "papers/items/2026-2606-10616-learning-what-to-remember-observability-safe-memory-retention-via-constrained-op.md", "title": "\"Learning What to Remember: Observability-Safe Memory Retention via Constrained Optimization for Long-Horizon Language Agents\"", "type": "paper", "meta": { "type": "paper", "title": "\"Learning What to Remember: Observability-Safe Memory Retention via Constrained Optimization for Long-Horizon Language Agents\"", "authors": "Qingcan Kang, Liu Mingyang, Shixiong Kai, Kaichao Liang, Tao Zhong, Mingxuan Yuan", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.10616", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-09", "updated_at": "2026-06-29", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "computer-use", "memory", "planning", "reasoning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "17", "collection_queries": "language-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-10677-infini-memory-maintainable-topic-documents-for-long-term-llm-agent-memory.md", "title": "\"Infini Memory: Maintainable Topic Documents for Long-Term LLM Agent Memory\"", "type": "paper", "meta": { "type": "paper", "title": "\"Infini Memory: Maintainable Topic Documents for Long-Term LLM Agent Memory\"", "authors": "Suozhao Ji, Baodong Wu, Zehao Wang, Lei Xia, Qingping Li, Ruisong Wang, Wenbo Ding, Zhenhua Zhu, et al.", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.10677", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-09", "updated_at": "2026-06-09", "status": "skimmed", "relevance": "high", "topics": [ "agent-evaluation", "memory", "rag", "tool-use" ], "methods": [ "topic-documents", "buffered-consolidation", "agentic-retrieval" ], "benchmarks": [ "MemoryAgentBench", "LongMemEval" ], "models": [], "datasets": [ "cs.AI", "cs.CL" ], "related_concepts": [ "semantic-memory", "selective-forgetting" ], "related_jobs": [], "related_experiments": [ "KC-001-agent-memory-pilot" ], "related_projects": [ "learning/agent-memory" ], "collection_score": "18", "collection_queries": "agent-memory" } }, { "collection": "papers", "path": "papers/items/2026-2606-10684-divide-and-cooperate-role-decomposed-multi-agent-llm-training-with-cross-agent-l.md", "title": "\"Divide and Cooperate: Role-Decomposed Multi-Agent LLM Training with Cross-Agent Learning Signals\"", "type": "paper", "meta": { "type": "paper", "title": "\"Divide and Cooperate: Role-Decomposed Multi-Agent LLM Training with Cross-Agent Learning Signals\"", "authors": "Jaewan Park, Solbee Cho, Jay-Yoon Lee", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.10684", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-09", "updated_at": "2026-06-09", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "multi-agent", "reasoning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.LG", "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "17", "collection_queries": "language-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-10742-memvenom-triggered-poisoning-of-multimodal-memories-in-web-agents.md", "title": "\"MemVenom: Triggered Poisoning of Multimodal Memories in Web Agents\"", "type": "paper", "meta": { "type": "paper", "title": "\"MemVenom: Triggered Poisoning of Multimodal Memories in Web Agents\"", "authors": "Yv Zhang, Hao Sun, Hao Fang, Kuofeng Gao, Fan Mo, Bin Chen, Shu-Tao Xia, Yaowei Wang", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.10742", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-09", "updated_at": "2026-06-09", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "computer-use", "memory", "planning", "rag", "reasoning" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CR", "cs.LG" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "14", "collection_queries": "web-gui-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-10749-toward-secure-llm-agents-threat-surfaces-attacks-defenses-and-evaluation.md", "title": "\"Toward Secure LLM Agents: Threat Surfaces, Attacks, Defenses, and Evaluation\"", "type": "paper", "meta": { "type": "paper", "title": "\"Toward Secure LLM Agents: Threat Surfaces, Attacks, Defenses, and Evaluation\"", "authors": "Yuchen Ling, Shengcheng Yu, Zhenyu Chen, Chunrong Fang", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.10749", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-09", "updated_at": "2026-06-09", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "memory", "multi-agent", "planning", "rag", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CR", "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "25", "collection_queries": "agent-safety, planning-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-10921-trace-only-what-you-need-structure-aware-on-demand-hypergraph-memory-for-long-do.md", "title": "\"Trace Only What You Need: Structure-Aware On-Demand Hypergraph Memory for Long-Document Question Answering\"", "type": "paper", "meta": { "type": "paper", "title": "\"Trace Only What You Need: Structure-Aware On-Demand Hypergraph Memory for Long-Document Question Answering\"", "authors": "Xiangjun Zai, Xingyu Tan, Chen Chen, Xiaoyang Wang, Wenjie Zhang", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.10921", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-09", "updated_at": "2026-06-09", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "computer-use", "memory", "multi-agent", "planning", "rag", "reasoning" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CL" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "16", "collection_queries": "rag-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-10933-frontier-coding-agents-use-metaprogramming-to-adapt-to-unfamiliar-programming-la.md", "title": "Frontier Coding Agents Use Metaprogramming to Adapt to Unfamiliar Programming Languages", "type": "paper", "meta": { "type": "paper", "title": "Frontier Coding Agents Use Metaprogramming to Adapt to Unfamiliar Programming Languages", "authors": "Aman Sharma, Sushrut Thorat, Paras Chopra", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.10933", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-09", "updated_at": "2026-06-09", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "coding-agent", "computer-use", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "17", "collection_queries": "agent-evaluation" } }, { "collection": "papers", "path": "papers/items/2026-2606-11042-workflow-gym-towards-long-horizon-evaluation-of-computer-use-agentic-tasks-in-re.md", "title": "\"Workflow-GYM: Towards Long-Horizon Evaluation of Computer-use Agentic tasks in Real-World Professional Fields\"", "type": "paper", "meta": { "type": "paper", "title": "\"Workflow-GYM: Towards Long-Horizon Evaluation of Computer-use Agentic tasks in Real-World Professional Fields\"", "authors": "Liya Zhu, Jingzhe Ding, Jian Zhang, Jianbo Xue, Shihao Liang, Ge Zhang, Yi Zhu, Duju Zeng, et al.", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.11042", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-09", "updated_at": "2026-06-11", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "computer-use", "planning", "tool-use", "workflow-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "19", "collection_queries": "web-gui-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-11078-a-history-aware-visually-grounded-critic-for-computer-use-agents.md", "title": "A History-Aware Visually Grounded Critic for Computer Use Agents", "type": "paper", "meta": { "type": "paper", "title": "A History-Aware Visually Grounded Critic for Computer Use Agents", "authors": "Jaewoo Lee, Zaid Khan, Archiki Prasad, Justin Chih-Yao Chen, Supriyo Chakraborty, Kartik Balasubramaniam, Sambit Sahu, Elias Stengel-Eskin, et al.", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.11078", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-09", "updated_at": "2026-06-09", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "computer-use", "planning", "rag", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI", "cs.CL", "cs.CV" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "15", "collection_queries": "web-gui-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-11079-vista-a-versatile-interactive-user-simulation-toolkit-for-agent-evaluation.md", "title": "\"VISTA: A Versatile Interactive User Simulation Toolkit for Agent Evaluation\"", "type": "paper", "meta": { "type": "paper", "title": "\"VISTA: A Versatile Interactive User Simulation Toolkit for Agent Evaluation\"", "authors": "Yunan Lu, Ryan Shea, Yusen Zhang, Zhou Yu", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.11079", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-09", "updated_at": "2026-06-09", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "rag", "tool-use", "world-model" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CL" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "14", "collection_queries": "agent-evaluation" } }, { "collection": "papers", "path": "papers/items/2026-2606-11119-trace-a-unified-rollout-budget-allocation-framework-for-efficient-agentic-reinfo.md", "title": "\"TRACE: A Unified Rollout Budget Allocation Framework for Efficient Agentic Reinforcement Learning\"", "type": "paper", "meta": { "type": "paper", "title": "\"TRACE: A Unified Rollout Budget Allocation Framework for Efficient Agentic Reinforcement Learning\"", "authors": "Heming Zou, Qi Wang, Yun Qu, Yuhang Jiang, Lizhou Cai, Yixiu Mao, Ru Peng, Xin Xu, et al.", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.11119", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-09", "updated_at": "2026-06-09", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "computer-use", "rag", "reasoning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.LG", "cs.AI", "cs.CL" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "16", "collection_queries": "agent-evaluation" } }, { "collection": "papers", "path": "papers/items/2026-2606-11176-data-journalist-agent-transforming-data-into-verifiable-multimodal-stories.md", "title": "\"Data Journalist Agent: Transforming Data into Verifiable Multimodal Stories\"", "type": "paper", "meta": { "type": "paper", "title": "\"Data Journalist Agent: Transforming Data into Verifiable Multimodal Stories\"", "authors": "Kevin Qinghong Lin, Batu EI, Yuhong Shi, Pan Lu, Philip Torr, James Zou", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.11176", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-09", "updated_at": "2026-06-09", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "coding-agent", "multi-agent", "rag", "reasoning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CV", "cs.CL", "cs.CY", "cs.HC" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "13", "collection_queries": "web-gui-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-11349-knowing-when-to-ask-self-gated-clarification-for-hierarchical-language-agents.md", "title": "\"Knowing When to Ask: Self-Gated Clarification for Hierarchical Language Agents\"", "type": "paper", "meta": { "type": "paper", "title": "\"Knowing When to Ask: Self-Gated Clarification for Hierarchical Language Agents\"", "authors": "Aijing Gao, Yiming Kang, Mengdie Flora Wang, Jae Oh Woo", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.11349", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-09", "updated_at": "2026-06-12", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "embodied-agent", "reasoning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI", "cs.HC" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "15", "collection_queries": "language-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-11354-a-zero-shot-multi-agent-framework-for-human-building-interaction-via-programmati.md", "title": "A Zero-Shot Multi-Agent Framework for Human-Building Interaction via Programmatic Reasoning", "type": "paper", "meta": { "type": "paper", "title": "A Zero-Shot Multi-Agent Framework for Human-Building Interaction via Programmatic Reasoning", "authors": "Yuqi Wang, Gulai Shen, Ali Mehmani", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.11354", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-09", "updated_at": "2026-06-09", "status": "queued", "relevance": "high", "topics": [ "agent-safety", "coding-agent", "multi-agent", "planning", "rag", "reasoning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.ET" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "18", "collection_queries": "rag-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-11680-organize-then-retrieve-hierarchical-memory-navigation-for-efficient-agents.md", "title": "\"Organize then Retrieve: Hierarchical Memory Navigation for Efficient Agents\"", "type": "paper", "meta": { "type": "paper", "title": "\"Organize then Retrieve: Hierarchical Memory Navigation for Efficient Agents\"", "authors": "Hao-Lun Hsu, Nikki Lijing Kuang, Boyi Liu, Zhewei Yao, Yuxiong He", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.11680", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-10", "updated_at": "2026-06-10", "status": "skimmed", "relevance": "high", "topics": [ "agent-evaluation", "computer-use", "embodied-agent", "memory", "planning", "rag", "reasoning" ], "methods": [ "hierarchical-memory-workspace", "memory-skill-evolution", "rl-retrieval-agent" ], "benchmarks": [ "ALFWorld", "LoCoMo", "LongMemEval" ], "models": [], "datasets": [ "cs.AI", "cs.CL", "cs.LG" ], "related_concepts": [ "memory-organization", "agentic-retrieval" ], "related_jobs": [], "related_experiments": [ "KC-001-agent-memory-pilot" ], "related_projects": [ "learning/agent-memory" ], "collection_score": "18", "collection_queries": "agent-memory" } }, { "collection": "papers", "path": "papers/items/2026-2606-11688-goal-autopilot-a-verifiable-anti-fabrication-firewall-for-unattended-long-horizo.md", "title": "\"Goal-Autopilot: A Verifiable Anti-Fabrication Firewall for Unattended Long-Horizon Agents\"", "type": "paper", "meta": { "type": "paper", "title": "\"Goal-Autopilot: A Verifiable Anti-Fabrication Firewall for Unattended Long-Horizon Agents\"", "authors": "Youwang Deng", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.11688", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-10", "updated_at": "2026-06-10", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "coding-agent", "planning", "rag" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CL", "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "13", "collection_queries": "planning-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-11702-medcta-a-benchmark-for-clinical-tool-agents.md", "title": "\"MedCTA: A Benchmark for Clinical Tool Agents\"", "type": "paper", "meta": { "type": "paper", "title": "\"MedCTA: A Benchmark for Clinical Tool Agents\"", "authors": "Tajamul Ashraf, Hyewon Jeong, Fida Mohammad Thoker, Bernard Ghanem", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.11702", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-10", "updated_at": "2026-06-10", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "planning", "rag", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CV", "cs.AI", "cs.CL" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "16", "collection_queries": "tool-use" } }, { "collection": "papers", "path": "papers/items/2026-2606-11869-agents-all-the-way-down-a-methodology-for-building-custom-ai-agents-from-substra.md", "title": "Agents All the Way Down; A Methodology for Building Custom AI Agents from Substrate to Production", "type": "paper", "meta": { "type": "paper", "title": "Agents All the Way Down; A Methodology for Building Custom AI Agents from Substrate to Production", "authors": "Marc Alier Forment, Juanan Pereira, Francisco José García-Peñalvo, María José Casañ Guerrero", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.11869", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-10", "updated_at": "2026-06-10", "status": "queued", "relevance": "high", "topics": [ "agent-safety", "coding-agent", "multi-agent", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.SE", "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "15", "collection_queries": "function-calling" } }, { "collection": "papers", "path": "papers/items/2026-2606-12195-internvideo3-agentify-foundation-models-with-multimodal-contextual-reasoning.md", "title": "\"InternVideo3: Agentify Foundation Models with Multimodal Contextual Reasoning\"", "type": "paper", "meta": { "type": "paper", "title": "\"InternVideo3: Agentify Foundation Models with Multimodal Contextual Reasoning\"", "authors": "Ziang Yan, Sheng Xia, Jiashuo Yu, Yue Wu, Tianxiang Jiang, Songze Li, Kanghui Tian, Yicheng Xu, et al.", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.12195", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-10", "updated_at": "2026-06-10", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "memory", "planning", "rag", "reasoning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CV" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "15", "collection_queries": "tool-use" } }, { "collection": "papers", "path": "papers/items/2026-2606-12320-a-five-plane-reference-architecture-for-runtime-governance-of-production-ai-agen.md", "title": "A Five-Plane Reference Architecture for Runtime Governance of Production AI Agents", "type": "paper", "meta": { "type": "paper", "title": "A Five-Plane Reference Architecture for Runtime Governance of Production AI Agents", "authors": "Krti Tallam", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.12320", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-10", "updated_at": "2026-06-10", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "memory", "planning", "reasoning", "tool-use", "workflow-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI", "cs.CC", "cs.CR", "cs.SE" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "18", "collection_queries": "agent-evaluation" } }, { "collection": "papers", "path": "papers/items/2026-2606-12341-ocelot-inference-leakage-budgets-for-privacy-preserving-llm-agents.md", "title": "\"OCELOT: Inference-Leakage Budgets for Privacy-Preserving LLM Agents\"", "type": "paper", "meta": { "type": "paper", "title": "\"OCELOT: Inference-Leakage Budgets for Privacy-Preserving LLM Agents\"", "authors": "Jin Xie, Songze Li", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.12341", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-10", "updated_at": "2026-06-10", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "reasoning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CR" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "15", "collection_queries": "agent-evaluation" } }, { "collection": "papers", "path": "papers/items/2026-2606-12344-claw-swe-bench-a-benchmark-for-evaluating-openclaw-style-agent-harnesses-on-codi.md", "title": "\"Claw-SWE-Bench: A Benchmark for Evaluating OpenClaw-style Agent Harnesses on Coding Tasks\"", "type": "paper", "meta": { "type": "paper", "title": "\"Claw-SWE-Bench: A Benchmark for Evaluating OpenClaw-style Agent Harnesses on Coding Tasks\"", "authors": "Mengyu Zheng, Kai Han, Boxun Li, Haiyang Xu, Yuchuan Tian, Wei He, Hang Zhou, Jianyuan Guo, et al.", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.12344", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-10", "updated_at": "2026-06-10", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "coding-agent", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.LG", "cs.CL" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "16", "collection_queries": "agent-evaluation" } }, { "collection": "papers", "path": "papers/items/2026-2606-12384-appo-agentic-procedural-policy-optimization.md", "title": "\"APPO: Agentic Procedural Policy Optimization\"", "type": "paper", "meta": { "type": "paper", "title": "\"APPO: Agentic Procedural Policy Optimization\"", "authors": "Xucong Wang, Ziyu Ma, Yong Wang, Yuxiang Ji, Shidong Yang, Guanhua Chen, Pengkun Wang, Xiangxiang Chu", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.12384", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-10", "updated_at": "2026-06-10", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "tool-use", "workflow-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.LG", "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "14", "collection_queries": "tool-use" } }, { "collection": "papers", "path": "papers/items/2026-2606-12563-arbor-tree-search-as-a-cognition-layer-for-autonomous-agents.md", "title": "\"Arbor: Tree Search as a Cognition Layer for Autonomous Agents\"", "type": "paper", "meta": { "type": "paper", "title": "\"Arbor: Tree Search as a Cognition Layer for Autonomous Agents\"", "authors": "Neha Prakriya, Chaojun Hou, Zheng Gong, Huasha Zhao, Xi Zhao, Mou Li, Zhenyu Gu, Emad Barsoum", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.12563", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-10", "updated_at": "2026-06-10", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "memory", "multi-agent", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "14", "collection_queries": "autonomous-agent-llm" } }, { "collection": "papers", "path": "papers/items/2026-2606-12586-beyond-attack-success-rate-examining-trigger-leakage-in-vision-language-agentic-.md", "title": "\"Beyond Attack Success Rate: Examining Trigger Leakage in Vision-Language Agentic Systems\"", "type": "paper", "meta": { "type": "paper", "title": "\"Beyond Attack Success Rate: Examining Trigger Leakage in Vision-Language Agentic Systems\"", "authors": "Jiamin Chang, Salil Kanhere, Piotr Koniusz, Jason, Xue, Hammond Pearce", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.12586", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-10", "updated_at": "2026-06-10", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "embodied-agent", "planning", "tool-use", "workflow-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CR" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "19", "collection_queries": "language-agent, tool-use" } }, { "collection": "papers", "path": "papers/items/2026-2606-12634-keep-policy-gradient-in-charge-sibling-guided-credit-distillation-for-long-horiz.md", "title": "\"Keep Policy Gradient in Charge: Sibling-Guided Credit Distillation for Long-Horizon Tool-Use Agents\"", "type": "paper", "meta": { "type": "paper", "title": "\"Keep Policy Gradient in Charge: Sibling-Guided Credit Distillation for Long-Horizon Tool-Use Agents\"", "authors": "Tianyu Ding, Jianhong Xin, Juan Pablo De la Cruz Weinstein", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.12634", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-10", "updated_at": "2026-06-29", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "computer-use", "planning", "reasoning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.LG", "cs.AI", "cs.CL" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "13", "collection_queries": "tool-use" } }, { "collection": "papers", "path": "papers/items/2026-2606-12657-trajgenagent-a-hierarchical-llm-agent-for-human-mobility-trajectory-generation.md", "title": "\"TrajGenAgent: A Hierarchical LLM Agent for Human Mobility Trajectory Generation\"", "type": "paper", "meta": { "type": "paper", "title": "\"TrajGenAgent: A Hierarchical LLM Agent for Human Mobility Trajectory Generation\"", "authors": "Siyu Li, Toan Tran, Lingyi Zhao, Khurram Shafique, Li Xiong", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.12657", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-10", "updated_at": "2026-06-10", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "planning", "rag", "reasoning", "workflow-agent", "world-model" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI", "cs.DB", "cs.RO" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "18", "collection_queries": "planning-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-12674-evoflux-inference-time-evolution-of-executable-tool-workflows-for-compact-agents.md", "title": "\"Evoflux: Inference-Time Evolution of Executable Tool Workflows for Compact Agents\"", "type": "paper", "meta": { "type": "paper", "title": "\"Evoflux: Inference-Time Evolution of Executable Tool Workflows for Compact Agents\"", "authors": "Kushal Raj Bhandari, Ling Yue, Ching-Yun Ko, Dhaval Patel, Shaowu Pan, Pin-Yu Chen, Jianxi Gao", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.12674", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-10", "updated_at": "2026-06-10", "status": "queued", "relevance": "high", "topics": [ "agent-safety", "computer-use", "planning", "tool-use", "workflow-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "14", "collection_queries": "function-calling, tool-use" } }, { "collection": "papers", "path": "papers/items/2026-2606-12703-smsr-certified-defence-against-runtime-memory-poisoning-in-persistent-llm-agent-.md", "title": "\"SMSR: Certified Defence Against Runtime Memory Poisoning in Persistent LLM Agent Systems\"", "type": "paper", "meta": { "type": "paper", "title": "\"SMSR: Certified Defence Against Runtime Memory Poisoning in Persistent LLM Agent Systems\"", "authors": "Tarun Sharma", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.12703", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-10", "updated_at": "2026-06-10", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "memory", "rag", "workflow-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CR", "cs.AI", "cs.LG" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "17", "collection_queries": "rag-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-12780-proplay-procedural-world-models-for-self-evolving-llm-agents.md", "title": "\"ProPlay: Procedural World Models for Self-Evolving LLM Agents\"", "type": "paper", "meta": { "type": "paper", "title": "\"ProPlay: Procedural World Models for Self-Evolving LLM Agents\"", "authors": "Yijun Ma, Zehong Wang, Yiyang Li, Ziming Li, Xiaoguang Guo, Weixiang Sun, Chuxu Zhang, Yanfang Ye", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.12780", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-11", "updated_at": "2026-06-11", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "computer-use", "memory", "planning", "tool-use", "world-model" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.LG", "cs.CL" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "18", "collection_queries": "planning-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-12837-lohosearch-benchmarking-long-horizon-search-agents-beyond-the-human-difficulty-c.md", "title": "\"LoHoSearch: Benchmarking Long-Horizon Search Agents Beyond the Human Difficulty Ceiling\"", "type": "paper", "meta": { "type": "paper", "title": "\"LoHoSearch: Benchmarking Long-Horizon Search Agents Beyond the Human Difficulty Ceiling\"", "authors": "Jiarui Zhao, Rongzhi Zhang, Lingchuan Liu, Hao Yang, Xunliang Cai, Xi Su", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.12837", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-11", "updated_at": "2026-06-17", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "planning", "reasoning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CL" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "13", "collection_queries": "agent-evaluation" } }, { "collection": "papers", "path": "papers/items/2026-2606-12945-learning-what-to-remember-a-cognitively-grounded-multi-factor-value-model-for-ag.md", "title": "\"Learning What to Remember: A Cognitively Grounded Multi-Factor Value Model for Agentic Memory\"", "type": "paper", "meta": { "type": "paper", "title": "\"Learning What to Remember: A Cognitively Grounded Multi-Factor Value Model for Agentic Memory\"", "authors": "Zhibao Chen, Qian Cheng", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.12945", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-11", "updated_at": "2026-06-20", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "coding-agent", "memory", "planning", "rag", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "22", "collection_queries": "agent-memory" } }, { "collection": "papers", "path": "papers/items/2026-2606-13148-terrabench-can-agents-reason-over-heterogeneous-earth-system-data.md", "title": "\"TerraBench: Can Agents Reason Over Heterogeneous Earth-System Data?\"", "type": "paper", "meta": { "type": "paper", "title": "\"TerraBench: Can Agents Reason Over Heterogeneous Earth-System Data?\"", "authors": "Dat Tien Nguyen, Thao Nguyen, Fadillah Adamsyah Maani, Huy M. Le, Muhammad Umer Sheikh, Numan Saeed, Muhammad Haris Khan, Salman Khan", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.13148", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-11", "updated_at": "2026-07-01", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "planning", "rag", "reasoning", "tool-use", "workflow-agent", "world-model" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "16", "collection_queries": "tool-use" } }, { "collection": "papers", "path": "papers/items/2026-2606-13177-memrefine-llm-guided-compression-for-long-term-agent-memory.md", "title": "\"MemRefine: LLM-Guided Compression for Long-Term Agent Memory\"", "type": "paper", "meta": { "type": "paper", "title": "\"MemRefine: LLM-Guided Compression for Long-Term Agent Memory\"", "authors": "Minjae Kim, Jinheon Baek, Soyeong Jeong, Sung Ju Hwang", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.13177", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-11", "updated_at": "2026-06-11", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "computer-use", "memory", "rag", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CL", "cs.AI", "cs.LG" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "16", "collection_queries": "agent-memory" } }, { "collection": "papers", "path": "papers/items/2026-2606-13192-reasoning-for-mobile-user-experience-with-multimodal-llms-task-benchmark-and-app.md", "title": "\"Reasoning for Mobile User Experience with Multimodal LLMs: Task, Benchmark, and Approach\"", "type": "paper", "meta": { "type": "paper", "title": "\"Reasoning for Mobile User Experience with Multimodal LLMs: Task, Benchmark, and Approach\"", "authors": "Ruichao Mao, Zhou Fang, Teng Guo, Hao Yang, Yaping Li, Shaohua Peng, Maji Huang, Xiaoyu Lin, et al.", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.13192", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-11", "updated_at": "2026-06-11", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "computer-use", "rag", "reasoning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "14", "collection_queries": "web-gui-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-13317-skillcat-contrastive-assessment-and-topology-aware-skill-self-evolution-for-llm-.md", "title": "\"SkillCAT: Contrastive Assessment and Topology-Aware Skill Self-Evolution for LLM Agents\"", "type": "paper", "meta": { "type": "paper", "title": "\"SkillCAT: Contrastive Assessment and Topology-Aware Skill Self-Evolution for LLM Agents\"", "authors": "Kunfeng Chen, Qihuang Zhong, Juhua Liu, Bo Du", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.13317", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-11", "updated_at": "2026-06-11", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "rag", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CL" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "14", "collection_queries": "agent-evaluation" } }, { "collection": "papers", "path": "papers/items/2026-2606-13385-who-pays-the-price-stakeholder-centric-prompt-injection-benchmarking-for-real-wo.md", "title": "Who Pays the Price? Stakeholder-Centric Prompt Injection Benchmarking for Real-world Web Agents", "type": "paper", "meta": { "type": "paper", "title": "Who Pays the Price? Stakeholder-Centric Prompt Injection Benchmarking for Real-world Web Agents", "authors": "Zihao Wang, Yiming Li, Yutong Wu, Zheyu Liu, Kangjie Chen, Fok Kar Wai, Pin-Yu Chen, Vrizlynn L. L. Thing, et al.", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.13385", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-11", "updated_at": "2026-06-11", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "computer-use", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CR", "cs.AI", "cs.CY", "cs.HC", "cs.MM" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "15", "collection_queries": "web-gui-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-13602-epibench-verifiable-evaluation-of-ai-agents-on-epigenomics-analysis.md", "title": "\"EpiBench: Verifiable Evaluation of AI Agents on Epigenomics Analysis\"", "type": "paper", "meta": { "type": "paper", "title": "\"EpiBench: Verifiable Evaluation of AI Agents on Epigenomics Analysis\"", "authors": "Harihara Muralidharan, Reema Baskar, Soo Hee Lee, Tim Proctor, Kenny Workman", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.13602", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-11", "updated_at": "2026-06-11", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "workflow-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "14", "collection_queries": "web-gui-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-13608-agentbeats-agentifying-agent-assessment-for-openness-standardization-and-reprodu.md", "title": "\"AgentBeats: Agentifying Agent Assessment for Openness, Standardization, and Reproducibility\"", "type": "paper", "meta": { "type": "paper", "title": "\"AgentBeats: Agentifying Agent Assessment for Openness, Standardization, and Reproducibility\"", "authors": "Xiaoyuan Liu, Jianhong Tu, Yuqi Chen, Siyuan Xie, Sihan Ren, Tianneng Shi, Gal Gantar, Evan Sandoval, et al.", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.13608", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-11", "updated_at": "2026-06-14", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "coding-agent", "multi-agent", "rag", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI", "cs.LG" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "16", "collection_queries": "agent-evaluation" } }, { "collection": "papers", "path": "papers/items/2026-2606-13643-recursive-agent-harnesses.md", "title": "Recursive Agent Harnesses", "type": "paper", "meta": { "type": "paper", "title": "Recursive Agent Harnesses", "authors": "Elias Lumer, Sahil Sen, Kevin Paul, Vamse Kumar Subbiah", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.13643", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-11", "updated_at": "2026-06-11", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "coding-agent", "planning", "reasoning", "tool-use", "workflow-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CL" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "15", "collection_queries": "function-calling" } }, { "collection": "papers", "path": "papers/items/2026-2606-13663-hypertool-beyond-step-wise-tool-calls-for-tool-augmented-agents.md", "title": "\"HyperTool: Beyond Step-Wise Tool Calls for Tool-Augmented Agents\"", "type": "paper", "meta": { "type": "paper", "title": "\"HyperTool: Beyond Step-Wise Tool Calls for Tool-Augmented Agents\"", "authors": "Yaxin Du, Yifan Zhou, Yujie Ge, Jiajun Wang, Xianghe Pang, Shuo Tang, Tuney Zheng, Bryan Dai, et al.", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.13663", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-11", "updated_at": "2026-06-11", "status": "queued", "relevance": "high", "topics": [ "rag", "reasoning", "tool-use", "workflow-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CL" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "13", "collection_queries": "tool-use" } }, { "collection": "papers", "path": "papers/items/2026-2606-13686-benchmarking-web-agent-safety-under-e-commerce-deceptive-interfaces.md", "title": "Benchmarking Web Agent Safety under E-commerce Deceptive Interfaces", "type": "paper", "meta": { "type": "paper", "title": "Benchmarking Web Agent Safety under E-commerce Deceptive Interfaces", "authors": "Zijing Shi, Meng Fang, Ling Chen", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.13686", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-04-26", "updated_at": "2026-04-26", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "computer-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CL", "cs.CY" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "15", "collection_queries": "agent-safety" } }, { "collection": "papers", "path": "papers/items/2026-2606-13904-sana-what-matters-for-qa-agents-over-massive-data-lakes.md", "title": "\"SANA: What Matters for QA Agents over Massive Data Lakes?\"", "type": "paper", "meta": { "type": "paper", "title": "\"SANA: What Matters for QA Agents over Massive Data Lakes?\"", "authors": "Austin Senna Wijaya, Jiaxiang Liu, Haonan Wang, Eugene Wu", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.13904", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-11", "updated_at": "2026-06-11", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "computer-use", "embodied-agent", "planning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CL", "cs.AI", "cs.DB" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "15", "collection_queries": "planning-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-13994-hidden-in-plain-sight-benchmarking-agent-safety-against-decomposition-attacks-wi.md", "title": "\"Hidden in Plain Sight: Benchmarking Agent Safety Against Decomposition Attacks with DECOMPBENCH\"", "type": "paper", "meta": { "type": "paper", "title": "\"Hidden in Plain Sight: Benchmarking Agent Safety Against Decomposition Attacks with DECOMPBENCH\"", "authors": "Vikhyath Kothamasu, Virginia Smith, Chhavi Yadav", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.13994", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-12", "updated_at": "2026-06-12", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "planning", "tool-use", "workflow-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CR", "cs.AI", "cs.LG" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "21", "collection_queries": "agent-safety, tool-use" } }, { "collection": "papers", "path": "papers/items/2026-2606-14106-naive-visual-memory-is-not-enough-a-failure-mode-study-of-gui-agents.md", "title": "\"Naive Visual Memory is Not Enough: A Failure-Mode Study of GUI Agents\"", "type": "paper", "meta": { "type": "paper", "title": "\"Naive Visual Memory is Not Enough: A Failure-Mode Study of GUI Agents\"", "authors": "Seoyoung Choi, Minseok Ko, Hyunseok Lee, Kunwoong Kim, Woomin Song, Chanseok Jeon, Jinwoo Shin", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.14106", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-12", "updated_at": "2026-06-12", "status": "queued", "relevance": "high", "topics": [ "computer-use", "memory", "rag", "reasoning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.MA", "cs.CV" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "16", "collection_queries": "web-gui-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-14470-gitofthoughts-version-controlled-reasoning-and-agent-memory-you-can-replay-diff-.md", "title": "\"GitOfThoughts: Version-Controlled Reasoning and Agent Memory You Can Replay, Diff, and Merge\"", "type": "paper", "meta": { "type": "paper", "title": "\"GitOfThoughts: Version-Controlled Reasoning and Agent Memory You Can Replay, Diff, and Merge\"", "authors": "Pavan C Shekar, Abhishek H S, Aswanth Krishnan", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.14470", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-12", "updated_at": "2026-06-22", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "coding-agent", "memory", "rag", "reasoning" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI", "cs.CL", "cs.LG" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "16", "collection_queries": "agent-memory" } }, { "collection": "papers", "path": "papers/items/2026-2606-14502-from-chatbot-to-digital-colleague-the-paradigm-shift-toward-persistent-autonomou.md", "title": "\"From Chatbot to Digital Colleague: The Paradigm Shift Toward Persistent Autonomous AI\"", "type": "paper", "meta": { "type": "paper", "title": "\"From Chatbot to Digital Colleague: The Paradigm Shift Toward Persistent Autonomous AI\"", "authors": "Yongheng Zhang, Ziang Liu, Jiaxuan Zhu, Shuai Wang, Xiangqi Chen, Haojing Huang, Jiayi Kuang, Siyu Chen, et al.", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.14502", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-12", "updated_at": "2026-06-12", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "memory", "rag", "reasoning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "14", "collection_queries": "tool-use" } }, { "collection": "papers", "path": "papers/items/2026-2606-14517-from-shield-to-target-denial-of-service-attacks-on-llm-based-agent-guardrails.md", "title": "\"From Shield to Target: Denial-of-Service Attacks on LLM-Based Agent Guardrails\"", "type": "paper", "meta": { "type": "paper", "title": "\"From Shield to Target: Denial-of-Service Attacks on LLM-Based Agent Guardrails\"", "authors": "Yuguang Zhou, Xunguang Wang, Pingchuan Ma, Zhantong Xue, Zhaoyu Wang, Shuai Wang", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.14517", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-12", "updated_at": "2026-06-16", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "computer-use", "multi-agent", "reasoning" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CR", "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "16", "collection_queries": "agent-evaluation, autonomous-agent-llm" } }, { "collection": "papers", "path": "papers/items/2026-2606-14571-streammembench-streaming-evaluation-of-agent-memory-for-future-oriented-assistan.md", "title": "\"StreamMemBench: Streaming Evaluation of Agent Memory for Future-Oriented Assistance\"", "type": "paper", "meta": { "type": "paper", "title": "\"StreamMemBench: Streaming Evaluation of Agent Memory for Future-Oriented Assistance\"", "authors": "Guanming Liu, Yuqi Ren, Hansu Gu, Peng Zhang, Weihang Wang, Jiahao Liu, Ning Gu, Tun Lu", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.14571", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-12", "updated_at": "2026-06-12", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "memory", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "15", "collection_queries": "agent-memory" } }, { "collection": "papers", "path": "papers/items/2026-2606-14574-simmer-benchmarking-latent-failures-in-llm-executable-planning-with-a-world-mode.md", "title": "\"SIMMER: Benchmarking Latent Failures in LLM Executable Planning with a World Model\"", "type": "paper", "meta": { "type": "paper", "title": "\"SIMMER: Benchmarking Latent Failures in LLM Executable Planning with a World Model\"", "authors": "Xiaoxin Lu, Ranran Haoran Zhang, Rui Zhang", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.14574", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-12", "updated_at": "2026-06-12", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "planning", "rag", "reasoning", "tool-use", "world-model" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CL", "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "14", "collection_queries": "autonomous-agent-llm" } }, { "collection": "papers", "path": "papers/items/2026-2606-14790-xflow-an-executable-protocol-programming-system-for-reliable-multi-agent-workflo.md", "title": "\"XFlow: An Executable Protocol Programming System for Reliable Multi-Agent Workflows\"", "type": "paper", "meta": { "type": "paper", "title": "\"XFlow: An Executable Protocol Programming System for Reliable Multi-Agent Workflows\"", "authors": "Hanqi Li, Jing Peng, Zijian Wang, Lu Chen, Kai Yu", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.14790", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-11", "updated_at": "2026-06-11", "status": "queued", "relevance": "high", "topics": [ "coding-agent", "memory", "multi-agent", "planning", "reasoning", "tool-use", "workflow-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.PL", "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "20", "collection_queries": "tool-use" } }, { "collection": "papers", "path": "papers/items/2026-2606-14805-knowledge-based-zero-replay-debugging-of-multi-agent-llm-traces.md", "title": "Knowledge-Based Zero-Replay Debugging of Multi-Agent LLM Traces", "type": "paper", "meta": { "type": "paper", "title": "Knowledge-Based Zero-Replay Debugging of Multi-Agent LLM Traces", "authors": "Dong Ho Kang, Hyeonjeong Cha, Daein Weon", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.14805", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-11", "updated_at": "2026-06-11", "status": "queued", "relevance": "high", "topics": [ "memory", "multi-agent", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.SE", "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "14", "collection_queries": "tool-use" } }, { "collection": "papers", "path": "papers/items/2026-2606-15017-are-online-skill-and-memory-modules-always-worth-their-tokens-a-budget-constrain.md", "title": "Are Online Skill and Memory Modules Always Worth Their Tokens? A Budget-Constrained Study of Web Agents", "type": "paper", "meta": { "type": "paper", "title": "Are Online Skill and Memory Modules Always Worth Their Tokens? A Budget-Constrained Study of Web Agents", "authors": "Sina Hajimiri, Masih Aminbeidokhti, Jose Dolz, Ismail Ben Ayed, Issam H. Laradji, Spandana Gella, Nicolas Gontier", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.15017", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-12", "updated_at": "2026-06-12", "status": "skimmed", "relevance": "high", "topics": [ "agent-evaluation", "computer-use", "memory", "reasoning", "workflow-agent" ], "methods": [ "budget-matched-baseline", "online-augmentation-audit" ], "benchmarks": [ "WebArena", "WorkArena-L1" ], "models": [], "datasets": [ "cs.CL" ], "related_concepts": [ "memory-cost-accounting", "evaluation-hygiene" ], "related_jobs": [], "related_experiments": [ "KC-001-agent-memory-pilot" ], "related_projects": [ "learning/agent-memory" ], "collection_score": "14", "collection_queries": "web-gui-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-15034-osguard-a-benchmark-for-safety-in-computer-use-agents.md", "title": "\"OSGuard: A Benchmark for Safety in Computer-Use Agents\"", "type": "paper", "meta": { "type": "paper", "title": "\"OSGuard: A Benchmark for Safety in Computer-Use Agents\"", "authors": "Mina Mohammadmirzaei, Jeffrey Flanigan", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.15034", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-13", "updated_at": "2026-06-13", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "computer-use", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "13", "collection_queries": "web-gui-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-15079-ling-and-ring-2-6-technical-report-efficient-and-instant-agentic-intelligence-at.md", "title": "\"Ling and Ring 2.6 Technical Report: Efficient and Instant Agentic Intelligence at Trillion-Parameter Scale\"", "type": "paper", "meta": { "type": "paper", "title": "\"Ling and Ring 2.6 Technical Report: Efficient and Instant Agentic Intelligence at Trillion-Parameter Scale\"", "authors": "Ang Li, Ben Liu, Bin Han, Bin Hu, Bin Jing, Binbin Hu, Bing Li, Cai Chen, et al.", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.15079", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-13", "updated_at": "2026-06-13", "status": "queued", "relevance": "high", "topics": [ "agent-safety", "coding-agent", "computer-use", "reasoning", "tool-use", "workflow-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CL", "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "15", "collection_queries": "tool-use" } }, { "collection": "papers", "path": "papers/items/2026-2606-15152-can-agents-read-the-room-benchmarking-visual-social-intelligence-in-multimodal-s.md", "title": "Can Agents Read the Room? Benchmarking Visual Social Intelligence in Multimodal Simulation", "type": "paper", "meta": { "type": "paper", "title": "Can Agents Read the Room? Benchmarking Visual Social Intelligence in Multimodal Simulation", "authors": "Shijun Wan, Xuehai Wu, Jiwen Zhang, Siyuan Wang, Zhongyu Wei", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.15152", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-13", "updated_at": "2026-06-13", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "computer-use", "tool-use", "world-model" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CL" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "14", "collection_queries": "agent-evaluation" } }, { "collection": "papers", "path": "papers/items/2026-2606-15242-benign-in-isolation-harmful-in-composition-security-risks-in-agent-skill-ecosyst.md", "title": "\"Benign in Isolation, Harmful in Composition: Security Risks in Agent Skill Ecosystems\"", "type": "paper", "meta": { "type": "paper", "title": "\"Benign in Isolation, Harmful in Composition: Security Risks in Agent Skill Ecosystems\"", "authors": "Yi Xie, Jiawei Du, Yu Cheng, Jiuan Zhou, Zhaoxia Yin", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.15242", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-13", "updated_at": "2026-06-13", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "planning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CR", "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "14", "collection_queries": "planning-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-15376-coagent-concurrency-control-for-multi-agent-systems.md", "title": "\"CoAgent: Concurrency Control for Multi-Agent Systems\"", "type": "paper", "meta": { "type": "paper", "title": "\"CoAgent: Concurrency Control for Multi-Agent Systems\"", "authors": "Hongtao Lyu, Dingyan Zhang, Mingyu Wu, Xingda Wei, Haibo Chen", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.15376", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-13", "updated_at": "2026-06-13", "status": "queued", "relevance": "high", "topics": [ "coding-agent", "multi-agent", "planning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.DC", "cs.AI", "cs.MA" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "16", "collection_queries": "planning-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-15591-agentic-retrieval-and-reinforcement-learned-equation-chains-a-controlled-generat.md", "title": "\"Agentic Retrieval and Reinforcement Learned Equation Chains: A Controlled Generation Framework for Complex and Novel Physics Word Problems\"", "type": "paper", "meta": { "type": "paper", "title": "\"Agentic Retrieval and Reinforcement Learned Equation Chains: A Controlled Generation Framework for Complex and Novel Physics Word Problems\"", "authors": "Tirthankar Mittra", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.15591", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-14", "updated_at": "2026-06-14", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "computer-use", "rag" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI", "cs.CL", "cs.MA" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "14", "collection_queries": "rag-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-15609-fragfuse-bypassing-access-control-of-large-language-model-agents-via-memory-base.md", "title": "\"FragFuse: Bypassing Access Control of Large Language Model Agents via Memory-Based Query Fragmentation and Fusion\"", "type": "paper", "meta": { "type": "paper", "title": "\"FragFuse: Bypassing Access Control of Large Language Model Agents via Memory-Based Query Fragmentation and Fusion\"", "authors": "Zixin Rao, Wentian Zhu, Chan Aristella Lu, Zhaorun Chen, Wei Niu, Le Guan, Bo Li, Zhen Xiang", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.15609", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-14", "updated_at": "2026-06-14", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "memory", "rag", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CR", "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "15", "collection_queries": "agent-memory" } }, { "collection": "papers", "path": "papers/items/2026-2606-15684-multi-agent-framework-for-time-sensitive-complementary-collaboration-in-minecraf.md", "title": "Multi-agent Framework for Time-Sensitive Complementary Collaboration in Minecraft", "type": "paper", "meta": { "type": "paper", "title": "Multi-agent Framework for Time-Sensitive Complementary Collaboration in Minecraft", "authors": "Juheon Yi, Jinglu Wang, Xiaoyi Zhang, Yan Lu", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.15684", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-14", "updated_at": "2026-06-14", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "multi-agent", "reasoning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "17", "collection_queries": "agent-evaluation" } }, { "collection": "papers", "path": "papers/items/2026-2606-15709-ai-driven-framework-for-adaptive-water-network-management-with-proof-of-concept-.md", "title": "\"AI-Driven Framework for Adaptive Water Network Management with Proof-of-Concept Implementation: Addressing Non-Revenue Water in Jordan\"", "type": "paper", "meta": { "type": "paper", "title": "\"AI-Driven Framework for Adaptive Water Network Management with Proof-of-Concept Implementation: Addressing Non-Revenue Water in Jordan\"", "authors": "Mohammed Fasha, Nahel Al-Maayta, Bilal Sowan, Mohammad Athamneh, Husam Barham", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.15709", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-14", "updated_at": "2026-06-14", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "rag", "tool-use", "workflow-agent", "world-model" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI", "cs.MA" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "13", "collection_queries": "function-calling, rag-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-15862-retailbench-benchmarking-long-horizon-reasoning-and-coherent-decision-making-of-.md", "title": "\"RetailBench: Benchmarking long horizon reasoning and coherent decision making of LLM agents in realistic retail environments\"", "type": "paper", "meta": { "type": "paper", "title": "\"RetailBench: Benchmarking long horizon reasoning and coherent decision making of LLM agents in realistic retail environments\"", "authors": "Linghua Zhang, Jun Wang, Jingtong Wu, Zhisong Zhang", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.15862", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-14", "updated_at": "2026-06-19", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "planning", "reasoning", "tool-use", "world-model" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "19", "collection_queries": "tool-use" } }, { "collection": "papers", "path": "papers/items/2026-2606-15874-llm-as-code-agentic-programming-for-agent-harness.md", "title": "\"LLM-as-Code: Agentic Programming for Agent Harness\"", "type": "paper", "meta": { "type": "paper", "title": "\"LLM-as-Code: Agentic Programming for Agent Harness\"", "authors": "Junjia Qi, Zichuan Fu, Jingtong Gao, Wenlin Zhang, Hanyu Yan, Xian Wu, Xiangyu Zhao", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.15874", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-14", "updated_at": "2026-06-22", "status": "queued", "relevance": "high", "topics": [ "reasoning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI", "cs.SE" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "13", "collection_queries": "web-gui-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-15903-control-plane-placement-shapes-forgetting-an-architectural-study-of-agent-memory.md", "title": "\"Control-Plane Placement Shapes Forgetting: An Architectural Study of Agent Memory Across Thirteen System Configurations\"", "type": "paper", "meta": { "type": "paper", "title": "\"Control-Plane Placement Shapes Forgetting: An Architectural Study of Agent Memory Across Thirteen System Configurations\"", "authors": "Dongxu Yang", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.15903", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-14", "updated_at": "2026-06-16", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "memory", "planning", "rag" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CL", "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "15", "collection_queries": "agent-memory" } }, { "collection": "papers", "path": "papers/items/2026-2606-15906-mage-rag-multigranular-adaptive-graph-evidence-for-agentic-multimodal-rag-in-lon.md", "title": "\"MAGE-RAG: Multigranular Adaptive Graph Evidence for Agentic Multimodal RAG in Long-Document QA\"", "type": "paper", "meta": { "type": "paper", "title": "\"MAGE-RAG: Multigranular Adaptive Graph Evidence for Agentic Multimodal RAG in Long-Document QA\"", "authors": "Yilong Zuo, Xunkai Li, Jing Yuan, Qiangqiang Dai, Hongchao Qin, Ronghua Li", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.15906", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-14", "updated_at": "2026-06-14", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "coding-agent", "rag" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.IR", "cs.AI", "cs.CL", "cs.DB", "cs.MM" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "13", "collection_queries": "rag-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-15931-deeproot-a-kg-coordinated-multi-agent-system-for-therapeutic-reasoning-over-hist.md", "title": "\"DeepRoot: A KG-Coordinated Multi-Agent System for Therapeutic Reasoning over Historical Medical Texts\"", "type": "paper", "meta": { "type": "paper", "title": "\"DeepRoot: A KG-Coordinated Multi-Agent System for Therapeutic Reasoning over Historical Medical Texts\"", "authors": "Zijian Carl Ma, Sean J. Wang, Sijbren Kramer, Li Erran Li", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.15931", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-14", "updated_at": "2026-06-14", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "multi-agent", "rag", "reasoning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.MA", "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "17", "collection_queries": "tool-use" } }, { "collection": "papers", "path": "papers/items/2026-2606-15994-agentic-framework-for-deep-learning-workload-migration-via-in-context-learning.md", "title": "Agentic Framework for Deep Learning workload migration via In-Context Learning", "type": "paper", "meta": { "type": "paper", "title": "Agentic Framework for Deep Learning workload migration via In-Context Learning", "authors": "Qiyue Liang, Steven Ingram, George Vanica, Andi Gavrilescu, Newfel Harrat, Hassan Sipra, Sethuraman Sankaran", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.15994", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-14", "updated_at": "2026-06-14", "status": "queued", "relevance": "high", "topics": [ "agent-safety", "rag", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI", "cs.LG" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "13", "collection_queries": "autonomous-agent-llm" } }, { "collection": "papers", "path": "papers/items/2026-2606-16111-towards-pareto-optimal-tool-integrated-agents-with-pareto-ranking-policy-optimiz.md", "title": "Towards Pareto-Optimal Tool-Integrated Agents with Pareto Ranking Policy Optimization", "type": "paper", "meta": { "type": "paper", "title": "Towards Pareto-Optimal Tool-Integrated Agents with Pareto Ranking Policy Optimization", "authors": "Junyi Li, Xiaowei Qian, Yingyi Zhang, Wenlin Zhang, Guojing Li, Sheng Zhang, Xiao Han, Yichao Wang, et al.", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.16111", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-15", "updated_at": "2026-06-15", "status": "queued", "relevance": "high", "topics": [ "agent-safety", "computer-use", "rag", "reasoning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CL" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "14", "collection_queries": "language-agent, tool-use" } }, { "collection": "papers", "path": "papers/items/2026-2606-16295-visualclaw-a-real-time-personalized-agent-for-the-physical-world.md", "title": "\"VisualClaw: A Real-Time, Personalized Agent for the Physical World\"", "type": "paper", "meta": { "type": "paper", "title": "\"VisualClaw: A Real-Time, Personalized Agent for the Physical World\"", "authors": "Haoqin Tu, Jianwen Chen, Zijun Wang, Siwei Han, Juncheng Wu, Hardy Chen, Haonian Ji, Kaiwen Xiong, et al.", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.16295", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-15", "updated_at": "2026-06-15", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "coding-agent", "computer-use", "rag", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CV", "cs.CL" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "16", "collection_queries": "agent-evaluation, tool-use, web-gui-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-16420-transferable-self-evolving-playbooks-for-agentic-security-auditing.md", "title": "Transferable Self-Evolving Playbooks for Agentic Security Auditing", "type": "paper", "meta": { "type": "paper", "title": "Transferable Self-Evolving Playbooks for Agentic Security Auditing", "authors": "Ziyue Wang, Cheuk Wang Maurice Ng, Chenchen Yu, Strick Sheng, Kaihua Qin, Liyi Zhou", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.16420", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-15", "updated_at": "2026-06-15", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "computer-use", "embodied-agent", "rag", "tool-use", "workflow-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CR" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "20", "collection_queries": "agent-safety, tool-use" } }, { "collection": "papers", "path": "papers/items/2026-2606-16432-accord-action-conditioned-contextual-grounding-for-language-agents.md", "title": "\"ACCORD: Action-Conditioned Contextual Grounding for Language Agents\"", "type": "paper", "meta": { "type": "paper", "title": "\"ACCORD: Action-Conditioned Contextual Grounding for Language Agents\"", "authors": "Lai Jiang, Cheng Qian, Zhenhailong Wang, Pan Lu, Heng Ji, Hao Peng", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.16432", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-15", "updated_at": "2026-06-15", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "embodied-agent", "rag", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CL", "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "15", "collection_queries": "language-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-16481-steering-emotional-dynamics-for-art-therapy-controllable-narrative-script-genera.md", "title": "\"Steering Emotional Dynamics for Art Therapy: Controllable Narrative Script Generation through Hierarchically Guided LLM Agents\"", "type": "paper", "meta": { "type": "paper", "title": "\"Steering Emotional Dynamics for Art Therapy: Controllable Narrative Script Generation through Hierarchically Guided LLM Agents\"", "authors": "Suqing Wang, Qinghai Miao, Chao Guo, Yisheng Lv", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.16481", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-15", "updated_at": "2026-06-15", "status": "queued", "relevance": "high", "topics": [ "computer-use", "planning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "13", "collection_queries": "planning-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-16534-generated-parallel-scalable-a-study-of-agentic-ai-generated-julia-code-on-superc.md", "title": "Generated, Parallel, Scalable? A Study of Agentic AI-Generated Julia Code on Supercomputers", "type": "paper", "meta": { "type": "paper", "title": "Generated, Parallel, Scalable? A Study of Agentic AI-Generated Julia Code on Supercomputers", "authors": "Linus Bantel, Anna-Lena Roth, Jonas Posner, Dirk Pflüger", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.16534", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-15", "updated_at": "2026-06-16", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "memory", "planning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.DC" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "16", "collection_queries": "tool-use" } }, { "collection": "papers", "path": "papers/items/2026-2606-16576-can-llm-agents-infer-world-models-evidence-from-agentic-automata-learning.md", "title": "Can LLM Agents Infer World Models? Evidence from Agentic Automata Learning", "type": "paper", "meta": { "type": "paper", "title": "Can LLM Agents Infer World Models? Evidence from Agentic Automata Learning", "authors": "Reef Menaged, Gili Lior, Shauli Ravfogel, Roee Aharoni, Gabriel Stanovsky", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.16576", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-15", "updated_at": "2026-06-15", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "planning", "reasoning", "tool-use", "world-model" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CL" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "18", "collection_queries": "planning-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-16591-sing-synthetic-intention-graph-for-scalable-active-tool-discovery-in-llm-agents.md", "title": "\"SING: Synthetic Intention Graph for Scalable Active Tool Discovery in LLM Agents\"", "type": "paper", "meta": { "type": "paper", "title": "\"SING: Synthetic Intention Graph for Scalable Active Tool Discovery in LLM Agents\"", "authors": "Qiao Xiao, Haochen Shi, Yisen Gao, Wenbin Hu, Huihao Jing, Tianshi Zheng, Baixuan Xu, Ziheng Zhang, et al.", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.16591", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-15", "updated_at": "2026-06-16", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "multi-agent", "planning", "rag", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CL" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "17", "collection_queries": "tool-use" } }, { "collection": "papers", "path": "papers/items/2026-2606-16613-coffeebench-benchmarking-long-horizon-llm-agents-in-heterogeneous-multi-agent-ec.md", "title": "\"CoffeeBench: Benchmarking Long-Horizon LLM Agents in Heterogeneous Multi-Agent Economies\"", "type": "paper", "meta": { "type": "paper", "title": "\"CoffeeBench: Benchmarking Long-Horizon LLM Agents in Heterogeneous Multi-Agent Economies\"", "authors": "Issa Sugiura, Daichi Hattori, Kazuo Araragi, Keita Ogawa, Shota Onose, Taro Makino, Teppei Usuki, Takashi Ishida", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.16613", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-15", "updated_at": "2026-06-15", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "multi-agent", "planning", "tool-use", "world-model" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "22", "collection_queries": "autonomous-agent-llm, planning-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-16659-fraudsmswalker-benchmarking-agentic-large-language-models-for-sms-to-webpage-fra.md", "title": "\"FraudSMSWalker: Benchmarking Agentic Large Language Models for SMS-to-Webpage Fraud Detection\"", "type": "paper", "meta": { "type": "paper", "title": "\"FraudSMSWalker: Benchmarking Agentic Large Language Models for SMS-to-Webpage Fraud Detection\"", "authors": "Y. H. Zhou, Z. M. Ma, Y. J. Zhou, Y. T. Li, H. X. Xiang, Y. M. Cheng, T. L. Chen, K. J. Zhang, et al.", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.16659", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-15", "updated_at": "2026-06-15", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "computer-use", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CL" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "18", "collection_queries": "web-gui-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-16748-mypcbench-a-benchmark-for-personally-intelligent-computer-use-agents.md", "title": "\"MyPCBench: A Benchmark for Personally Intelligent Computer-Use Agents\"", "type": "paper", "meta": { "type": "paper", "title": "\"MyPCBench: A Benchmark for Personally Intelligent Computer-Use Agents\"", "authors": "Lawrence Keunho Jang, Andrew Keunwoo Jang, Jing Yu Koh, Ruslan Salakhutdinov", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.16748", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-15", "updated_at": "2026-06-15", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "computer-use", "tool-use", "workflow-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.LG", "cs.CL" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "14", "collection_queries": "agent-evaluation, web-gui-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-16774-openclaw-skill-collective-skill-tree-search-for-agentic-large-language-models.md", "title": "\"OpenClaw-Skill: Collective Skill Tree Search for Agentic Large Language Models\"", "type": "paper", "meta": { "type": "paper", "title": "\"OpenClaw-Skill: Collective Skill Tree Search for Agentic Large Language Models\"", "authors": "Tianyi Lin, Chuanyu Sun, Jingyi Zhang, Changxu Wei, Huanjin Yao, Shunyu Liu, Xikun Zhang, Liu Liu, et al.", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.16774", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-15", "updated_at": "2026-06-15", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "planning", "rag", "reasoning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI", "cs.CL" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "19", "collection_queries": "planning-agent, tool-use" } }, { "collection": "papers", "path": "papers/items/2026-2606-16802-labosbench-benchmarking-computer-use-agents-for-scientific-instrument-control.md", "title": "\"LabOSBench: Benchmarking Computer Use Agents for Scientific Instrument Control\"", "type": "paper", "meta": { "type": "paper", "title": "\"LabOSBench: Benchmarking Computer Use Agents for Scientific Instrument Control\"", "authors": "Anqi Zou, Han Deng, Chengyu Zhang, Junquan Hu, Yu Wang, Yuxiang Xing, Aokai Zhang, Hanling Zhang, et al.", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.16802", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-15", "updated_at": "2026-06-15", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "computer-use", "planning", "workflow-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "21", "collection_queries": "web-gui-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-16813-gist-cmtf-goal-state-inference-for-causal-minimal-tool-filtering-in-llm-agents.md", "title": "\"GIST-CMTF: Goal-State Inference for Causal Minimal Tool Filtering in LLM Agents\"", "type": "paper", "meta": { "type": "paper", "title": "\"GIST-CMTF: Goal-State Inference for Causal Minimal Tool Filtering in LLM Agents\"", "authors": "Rahul Suresh Babu, Rohit Shukla", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.16813", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-15", "updated_at": "2026-06-15", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "computer-use", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "13", "collection_queries": "tool-use" } }, { "collection": "papers", "path": "papers/items/2026-2606-16839-towards-llm-accelerated-rapid-reviews-for-software-tool-discovery-case-for-log-a.md", "title": "Towards LLM Accelerated Rapid Reviews for Software Tool Discovery -- Case for Log Anomaly Detection", "type": "paper", "meta": { "type": "paper", "title": "Towards LLM Accelerated Rapid Reviews for Software Tool Discovery -- Case for Log Anomaly Detection", "authors": "Jesse Nyyssölä, Hamza Bin Mazhar, Alexander Bakhtin, Matteo Esposito, Nana Reinikainen, Yuqing Wang, Ying Song, Davide Taibi, et al.", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.16839", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-15", "updated_at": "2026-06-15", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "coding-agent", "planning", "tool-use", "workflow-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.SE" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "13", "collection_queries": "planning-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-16871-human-on-the-bridge-scalable-evaluation-for-ai-agents.md", "title": "\"Human-on-the-Bridge: Scalable Evaluation for AI Agents\"", "type": "paper", "meta": { "type": "paper", "title": "\"Human-on-the-Bridge: Scalable Evaluation for AI Agents\"", "authors": "Fouad Bousetouane", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.16871", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-15", "updated_at": "2026-06-15", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "computer-use", "memory", "rag", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.MA" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "20", "collection_queries": "agent-evaluation" } }, { "collection": "papers", "path": "papers/items/2026-2606-17041-benchmarking-llm-agents-on-meta-analysis-articles-from-nature-portfolio.md", "title": "Benchmarking LLM Agents on Meta-Analysis Articles from Nature Portfolio", "type": "paper", "meta": { "type": "paper", "title": "Benchmarking LLM Agents on Meta-Analysis Articles from Nature Portfolio", "authors": "Anzhe Xie, Weihang Su, Yujia Zhou, Yiqun Liu, Qingyao Ai", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.17041", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-15", "updated_at": "2026-07-01", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "computer-use", "rag", "reasoning", "workflow-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CL", "cs.IR" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "18", "collection_queries": "rag-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-17076-cmip-forge-an-agentic-system-that-retrieves-computes-and-self-reviews-climate-sc.md", "title": "\"CMIP-Forge: An Agentic System that Retrieves, Computes, and Self-Reviews Climate Science\"", "type": "paper", "meta": { "type": "paper", "title": "\"CMIP-Forge: An Agentic System that Retrieves, Computes, and Self-Reviews Climate Science\"", "authors": "Dmitrii Pantiukhin, Boris Shapkin, Ivan Kuznetsov, Thomas Jung, Nikolay Koldunov", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.17076", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-10", "updated_at": "2026-06-10", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "planning", "rag", "tool-use", "workflow-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "physics.ao-ph", "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "15", "collection_queries": "rag-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-17114-an-evaluation-of-data-leakage-risks-in-tool-using-llm-agents-in-realistic-scenar.md", "title": "An Evaluation of Data Leakage Risks in Tool-Using LLM Agents in Realistic Scenarios", "type": "paper", "meta": { "type": "paper", "title": "An Evaluation of Data Leakage Risks in Tool-Using LLM Agents in Realistic Scenarios", "authors": "Hankyul Baek, Jaewon Noh, Sang Seo, Yongsu Kim, Gabriel Waikin Loh Matienzo, Young Il Kim, Ee Wei Seah, Akriti Vij", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.17114", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-15", "updated_at": "2026-06-15", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "tool-use", "workflow-agent", "world-model" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CR", "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "20", "collection_queries": "agent-safety, tool-use" } }, { "collection": "papers", "path": "papers/items/2026-2606-17246-geodisaster-benchmarking-orchestrated-agents-for-operational-disaster-geo-intell.md", "title": "\"GeoDisaster: Benchmarking Orchestrated Agents for Operational Disaster Geo-Intelligence\"", "type": "paper", "meta": { "type": "paper", "title": "\"GeoDisaster: Benchmarking Orchestrated Agents for Operational Disaster Geo-Intelligence\"", "authors": "Maram Hasan, Aman Verma, Savitra Roy, Hariseetharam Gunduboina, Daksh Jain, Muhammad Haris Khan, Subhasis Chaudhuri, Biplab Banerjee", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.17246", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-15", "updated_at": "2026-06-15", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "multi-agent", "rag", "reasoning", "tool-use", "workflow-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CV", "cs.MA" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "20", "collection_queries": "tool-use" } }, { "collection": "papers", "path": "papers/items/2026-2606-17368-distributed-general-purpose-agent-networks-architecture-key-mechanisms-and-proto.md", "title": "\"Distributed General-Purpose Agent Networks: Architecture, Key Mechanisms, and Prototypes\"", "type": "paper", "meta": { "type": "paper", "title": "\"Distributed General-Purpose Agent Networks: Architecture, Key Mechanisms, and Prototypes\"", "authors": "Shengli Zhang, Deen Ma, Zibin Lin, Taotao Wang", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.17368", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-15", "updated_at": "2026-06-15", "status": "queued", "relevance": "high", "topics": [ "computer-use", "multi-agent", "planning", "tool-use", "world-model" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI", "cs.NI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "13", "collection_queries": "autonomous-agent-llm" } }, { "collection": "papers", "path": "papers/items/2026-2606-17383-model-validation-of-agentic-ai-systems-a-pomdp-based-framework-for-belief-state-.md", "title": "\"Model Validation of Agentic AI Systems: A POMDP-Based Framework for Belief-State, Forecast, and Policy Validation\"", "type": "paper", "meta": { "type": "paper", "title": "\"Model Validation of Agentic AI Systems: A POMDP-Based Framework for Belief-State, Forecast, and Policy Validation\"", "authors": "Matthew Francis Dixon", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.17383", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-16", "updated_at": "2026-06-16", "status": "queued", "relevance": "high", "topics": [ "agent-safety", "rag", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "q-fin.RM", "cs.AI", "cs.LG", "stat.ML" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "13", "collection_queries": "autonomous-agent-llm" } }, { "collection": "papers", "path": "papers/items/2026-2606-17449-mode-rag-manifold-outlier-diagnosis-and-energy-based-retrieval-augmented-generat.md", "title": "\"MODE-RAG: Manifold Outlier Diagnosis and Energy-based Retrieval-Augmented Generation Evaluation\"", "type": "paper", "meta": { "type": "paper", "title": "\"MODE-RAG: Manifold Outlier Diagnosis and Energy-based Retrieval-Augmented Generation Evaluation\"", "authors": "Zehang Wei, Jiaxin Dai, Jiamin Yan, Xiang Xiang", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.17449", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-16", "updated_at": "2026-06-16", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "computer-use", "multi-agent", "rag", "reasoning" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CL", "cs.AI", "cs.CV", "cs.LG", "cs.MM" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "15", "collection_queries": "rag-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-17453-mapsatisfybench-benchmarking-satisfaction-aware-map-agents-through-behavior-grou.md", "title": "\"MapSatisfyBench: Benchmarking Satisfaction-Aware Map Agents through Behavior-Grounded Implicit Decision Factors\"", "type": "paper", "meta": { "type": "paper", "title": "\"MapSatisfyBench: Benchmarking Satisfaction-Aware Map Agents through Behavior-Grounded Implicit Decision Factors\"", "authors": "Lubin Bai, Mengyu Cao, Sixue Wang, Zhongwei Wan, Yue Pan, Jiale Hou, Xiang Li, Xiuyuan Zhang", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.17453", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-16", "updated_at": "2026-06-17", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "13", "collection_queries": "agent-evaluation" } }, { "collection": "papers", "path": "papers/items/2026-2606-17459-can-llms-be-ceos-benchmarking-strategic-resource-reallocation-with-multi-role-ag.md", "title": "Can LLMs Be CEOs? Benchmarking Strategic Resource Reallocation with Multi-Role Agent Simulation", "type": "paper", "meta": { "type": "paper", "title": "Can LLMs Be CEOs? Benchmarking Strategic Resource Reallocation with Multi-Role Agent Simulation", "authors": "Yuyang Dai, Xueqing Peng, Lingfei Qian, Zhuohan Xie", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.17459", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-16", "updated_at": "2026-06-16", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "computer-use", "multi-agent", "planning", "rag", "reasoning", "tool-use", "world-model" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "22", "collection_queries": "agent-evaluation, planning-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-17573-cordon-semantic-transactions-for-tool-using-llm-agents.md", "title": "\"Cordon: Semantic Transactions for Tool-Using LLM Agents\"", "type": "paper", "meta": { "type": "paper", "title": "\"Cordon: Semantic Transactions for Tool-Using LLM Agents\"", "authors": "Zheng Chen, Hanqing Liu, Duling Xu, Dong Dong, Jialin Li, Bangzheng Pu, Jidong Zhai", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.17573", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-16", "updated_at": "2026-06-16", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "memory", "tool-use", "workflow-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.OS", "cs.CR" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "16", "collection_queries": "tool-use" } }, { "collection": "papers", "path": "papers/items/2026-2606-17680-envrl-learn-from-environment-dynamics-in-agentic-reinforcement-learning.md", "title": "\"EnvRL: Learn from Environment Dynamics in Agentic Reinforcement Learning\"", "type": "paper", "meta": { "type": "paper", "title": "\"EnvRL: Learn from Environment Dynamics in Agentic Reinforcement Learning\"", "authors": "Zhitong Wang, Songze Li, Hao Peng, Shuzheng Si, Yi Wang, Maosong Sun, Juanzi Li", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.17680", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-16", "updated_at": "2026-06-16", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "planning", "rag", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.LG", "cs.CL" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "15", "collection_queries": "agent-evaluation" } }, { "collection": "papers", "path": "papers/items/2026-2606-18023-loopcoder-v2-only-loop-once-for-efficient-test-time-computation-scaling.md", "title": "\"LoopCoder-v2: Only Loop Once for Efficient Test-Time Computation Scaling\"", "type": "paper", "meta": { "type": "paper", "title": "\"LoopCoder-v2: Only Loop Once for Efficient Test-Time Computation Scaling\"", "authors": "Jian Yang, Shawn Guo, Wei Zhang, Tianyu Zheng, Yaxin Du, Haau-Sing Li, Jiajun Wu, Yue Song, et al.", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.18023", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-16", "updated_at": "2026-06-16", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "coding-agent", "memory", "reasoning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.LG", "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "13", "collection_queries": "tool-use" } }, { "collection": "papers", "path": "papers/items/2026-2606-18037-provenanceguard-source-aware-factuality-verification-for-mcp-based-llm-agents.md", "title": "\"ProvenanceGuard: Source-Aware Factuality Verification for MCP-Based LLM Agents\"", "type": "paper", "meta": { "type": "paper", "title": "\"ProvenanceGuard: Source-Aware Factuality Verification for MCP-Based LLM Agents\"", "authors": "Ander Alvarez, Santhiya Rajan, Samuel Mugel, Román Orús", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.18037", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-16", "updated_at": "2026-06-16", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "rag", "reasoning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI", "cs.CL", "cs.MA" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "16", "collection_queries": "tool-use" } }, { "collection": "papers", "path": "papers/items/2026-2606-18051-compositional-skill-routing-for-llm-agents-decompose-retrieve-and-compose.md", "title": "\"Compositional Skill Routing for LLM Agents: Decompose, Retrieve, and Compose\"", "type": "paper", "meta": { "type": "paper", "title": "\"Compositional Skill Routing for LLM Agents: Decompose, Retrieve, and Compose\"", "authors": "Xueping Gao", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.18051", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-16", "updated_at": "2026-06-16", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "planning", "rag", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CL" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "16", "collection_queries": "planning-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-18068-agentic-ai-based-framework-for-mitigating-premature-diagnostic-handoff-and-silen.md", "title": "Agentic AI-based Framework for Mitigating Premature Diagnostic Handoff and Silent Hallucination in Healthcare Applications", "type": "paper", "meta": { "type": "paper", "title": "Agentic AI-based Framework for Mitigating Premature Diagnostic Handoff and Silent Hallucination in Healthcare Applications", "authors": "Divyansh Srivastava, Shreya Ghosh, Anshul Verma, Rajkumar Buyya", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.18068", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-16", "updated_at": "2026-06-16", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "multi-agent", "reasoning" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "15", "collection_queries": "agentic-ai" } }, { "collection": "papers", "path": "papers/items/2026-2606-18142-your-ai-travel-agent-would-book-you-a-bullfight-an-agentic-benchmark-for-implici.md", "title": "\"Your AI Travel Agent Would Book You a Bullfight: An Agentic Benchmark for Implicit Animal Welfare in Frontier AI Models\"", "type": "paper", "meta": { "type": "paper", "title": "\"Your AI Travel Agent Would Book You a Bullfight: An Agentic Benchmark for Implicit Animal Welfare in Frontier AI Models\"", "authors": "Jasmine Brazilek, Joel Christoph, Maheep Chaudhary, Oliver Tullio, Carol Kline, Miles Tidmarsh, Arturs Kanepajs", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.18142", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-16", "updated_at": "2026-07-06", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI", "cs.CL", "cs.CY" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "17", "collection_queries": "agent-evaluation" } }, { "collection": "papers", "path": "papers/items/2026-2606-18272-mitigating-anchoring-bias-in-llm-based-agents-for-energy-efficient-6g-autonomous.md", "title": "Mitigating Anchoring Bias in LLM-Based Agents for Energy-Efficient 6G Autonomous Networks", "type": "paper", "meta": { "type": "paper", "title": "Mitigating Anchoring Bias in LLM-Based Agents for Energy-Efficient 6G Autonomous Networks", "authors": "Hatim Chergui, Claudia Carballo González, Farhad Rezazadeh, Merouane Debbah", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.18272", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-05", "updated_at": "2026-06-18", "status": "queued", "relevance": "high", "topics": [ "agent-safety", "computer-use", "multi-agent", "reasoning" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.NI", "cs.AI", "eess.SY" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "13", "collection_queries": "autonomous-agent-llm" } }, { "collection": "papers", "path": "papers/items/2026-2606-18356-safeclawbench-separating-semantic-audit-evidence-and-sandbox-harm-in-tool-using-.md", "title": "\"SafeClawBench: Separating Semantic, Audit-Evidence, and Sandbox Harm in Tool-Using LLM Agents\"", "type": "paper", "meta": { "type": "paper", "title": "\"SafeClawBench: Separating Semantic, Audit-Evidence, and Sandbox Harm in Tool-Using LLM Agents\"", "authors": "Yuchuan Tian, Mengyu Zheng, Haocheng Mei, Ye Yuan, Chao Xu, Xinghao Chen, Hanting Chen, Yu Wang", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.18356", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-16", "updated_at": "2026-06-16", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "computer-use", "memory", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CR", "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "19", "collection_queries": "agent-safety, tool-use" } }, { "collection": "papers", "path": "papers/items/2026-2606-18363-guava-an-effective-and-universal-harness-for-embodied-manipulation.md", "title": "\"Guava: An Effective and Universal Harness for Embodied Manipulation\"", "type": "paper", "meta": { "type": "paper", "title": "\"Guava: An Effective and Universal Harness for Embodied Manipulation\"", "authors": "Haowen Liu, Xirui Li, Shaoxiong Yao, Peng Shi, Tianyi Zhou, Jia-Bin Huang, Furong Huang, Jiayuan Mao", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.18363", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-16", "updated_at": "2026-06-16", "status": "queued", "relevance": "high", "topics": [ "embodied-agent", "planning", "reasoning", "tool-use", "workflow-agent", "world-model" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.RO", "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "13", "collection_queries": "agentic-ai, tool-use" } }, { "collection": "papers", "path": "papers/items/2026-2606-18406-coremem-riemannian-retrieval-and-fisher-guided-distillation-for-long-term-memory.md", "title": "\"CoreMem: Riemannian Retrieval and Fisher-Guided Distillation for Long-Term Memory in Dialogue Agents\"", "type": "paper", "meta": { "type": "paper", "title": "\"CoreMem: Riemannian Retrieval and Fisher-Guided Distillation for Long-Term Memory in Dialogue Agents\"", "authors": "Jiaqi Chen, Yongqin Zeng, Shaoshen Chen, Yijian Zhang, Hai-Tao Zheng, Chunxia Ma, XiuTeng Zhou", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.18406", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-16", "updated_at": "2026-06-16", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "computer-use", "memory", "rag", "reasoning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CL" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "15", "collection_queries": "agent-memory" } }, { "collection": "papers", "path": "papers/items/2026-2606-18467-toolchain-crc-conformal-risk-control-for-agentic-ai-under-retrieval-and-tool-use.md", "title": "\"ToolChain-CRC: Conformal Risk Control for Agentic AI Under Retrieval and Tool-Use Drift\"", "type": "paper", "meta": { "type": "paper", "title": "\"ToolChain-CRC: Conformal Risk Control for Agentic AI Under Retrieval and Tool-Use Drift\"", "authors": "Jeffery Opoku, David Banahene", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.18467", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-16", "updated_at": "2026-06-16", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "rag", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "stat.ML", "cs.LG" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "17", "collection_queries": "agent-evaluation, agentic-ai, ai-agent, rag-agent, tool-use" } }, { "collection": "papers", "path": "papers/items/2026-2606-18502-towards-scalable-customization-and-deployment-of-multi-agent-systems-for-enterpr.md", "title": "Towards Scalable Customization and Deployment of Multi-Agent Systems for Enterprise Applications", "type": "paper", "meta": { "type": "paper", "title": "Towards Scalable Customization and Deployment of Multi-Agent Systems for Enterprise Applications", "authors": "Paresh Dashore, Shreyas Kulkarni, Uttam Gurram, Nadia Bathaee, Kartik Balasubramaniam, Genta Indra Winata, Sambit Sahu, Shi-Xiong Zhang", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.18502", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-16", "updated_at": "2026-06-16", "status": "queued", "relevance": "high", "topics": [ "coding-agent", "multi-agent", "reasoning", "tool-use", "workflow-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CL" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "16", "collection_queries": "agentic-ai" } }, { "collection": "papers", "path": "papers/items/2026-2606-18619-code-augur-agentic-vulnerability-detection-via-specification-inference.md", "title": "\"Code-Augur: Agentic Vulnerability Detection via Specification Inference\"", "type": "paper", "meta": { "type": "paper", "title": "\"Code-Augur: Agentic Vulnerability Detection via Specification Inference\"", "authors": "Zhengxiong Luo, Mehtab Zafar, Dylan Wolff, Abhik Roychoudhury", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.18619", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-17", "updated_at": "2026-06-17", "status": "queued", "relevance": "high", "topics": [ "agent-safety", "computer-use", "rag", "reasoning" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CR", "cs.AI", "cs.SE" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "16", "collection_queries": "ai-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-18671-hansel-extracting-breadcrumbs-from-web-agent-trajectories-for-interactive-verifi.md", "title": "\"HANSEL: Extracting Breadcrumbs from Web Agent Trajectories for Interactive Verification\"", "type": "paper", "meta": { "type": "paper", "title": "\"HANSEL: Extracting Breadcrumbs from Web Agent Trajectories for Interactive Verification\"", "authors": "Yujin Zhang, Daye Nam", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.18671", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-17", "updated_at": "2026-06-17", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "computer-use", "embodied-agent", "planning" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.HC" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "14", "collection_queries": "ai-agent, web-gui-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-18789-poweragentbench-ss-a-benchmark-for-agentic-ai-in-power-system-steady-state-studi.md", "title": "\"PowerAgentBench-SS: A Benchmark for Agentic AI in Power System Steady-State Studies\"", "type": "paper", "meta": { "type": "paper", "title": "\"PowerAgentBench-SS: A Benchmark for Agentic AI in Power System Steady-State Studies\"", "authors": "Costas Mylonas, Magda Foti, Andrea Pomarico, Matheus Duarte, Qian Zhang, Emmanouel Varvarigos", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.18789", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-17", "updated_at": "2026-06-17", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "computer-use", "planning", "rag", "tool-use", "workflow-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "eess.SY" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "24", "collection_queries": "agentic-ai, planning-agent, tool-use" } }, { "collection": "papers", "path": "papers/items/2026-2606-18829-gatemem-benchmarking-memory-governance-in-multi-principal-shared-memory-agents.md", "title": "\"GateMem: Benchmarking Memory Governance in Multi-Principal Shared-Memory Agents\"", "type": "paper", "meta": { "type": "paper", "title": "\"GateMem: Benchmarking Memory Governance in Multi-Principal Shared-Memory Agents\"", "authors": "Zhe Ren, Yibo Yang, Yimeng Chen, Zijun Zhao, Benshuo Fu, Zhihao Shu, Bingjie Zhang, Yangyang Xu, et al.", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.18829", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-17", "updated_at": "2026-06-17", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "memory", "planning", "rag", "workflow-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.LG", "cs.CL" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "19", "collection_queries": "agent-memory" } }, { "collection": "papers", "path": "papers/items/2026-2606-18950-rtsgamebench-an-rts-benchmark-for-strategic-reasoning-by-vision-language-models.md", "title": "\"RTSGameBench: An RTS Benchmark for Strategic Reasoning by Vision-Language Models\"", "type": "paper", "meta": { "type": "paper", "title": "\"RTSGameBench: An RTS Benchmark for Strategic Reasoning by Vision-Language Models\"", "authors": "San Kim, Daechul Ahn, Reokyoung Kim, Hyeonbeom Choi, Seungyeon Jwa, Jonghyun Choi", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.18950", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-17", "updated_at": "2026-06-18", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "memory", "multi-agent", "planning", "rag", "reasoning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "18", "collection_queries": "agent-memory" } }, { "collection": "papers", "path": "papers/items/2026-2606-19063-pypiline-malicious-pypi-package-detection-via-suspicious-api-knowledge-and-agent.md", "title": "\"PYPILINE: Malicious PyPI Package Detection via Suspicious API Knowledge and Agent Workflow\"", "type": "paper", "meta": { "type": "paper", "title": "\"PYPILINE: Malicious PyPI Package Detection via Suspicious API Knowledge and Agent Workflow\"", "authors": "Siyuan Pang, Yepeng Yao, Zhengwei Jiang, Zijing Fan, Haozhe Li, Baoxu Liu", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.19063", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-17", "updated_at": "2026-06-26", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "rag", "tool-use", "workflow-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CR" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "14", "collection_queries": "agentic-ai, rag-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-19242-runtime-compliance-verification-for-ai-agents.md", "title": "Runtime Compliance Verification for AI Agents", "type": "paper", "meta": { "type": "paper", "title": "Runtime Compliance Verification for AI Agents", "authors": "Nafiseh Kahani, Masoud Barati, Diana Addae", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.19242", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-17", "updated_at": "2026-06-17", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.SE" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "13", "collection_queries": "ai-agent, function-calling, tool-use" } }, { "collection": "papers", "path": "papers/items/2026-2606-19245-txbench-pp-analyzing-ai-agent-performance-on-small-molecule-preclinical-pharmaco.md", "title": "\"TxBench-PP: Analyzing AI Agent Performance on Small-Molecule Preclinical Pharmacology\"", "type": "paper", "meta": { "type": "paper", "title": "\"TxBench-PP: Analyzing AI Agent Performance on Small-Molecule Preclinical Pharmacology\"", "authors": "Hannah Le, Ramesh Ramasamy, Alex Urrutia, Mahsa Yazdani, Tim Proctor, Kenny Workman", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.19245", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-17", "updated_at": "2026-06-18", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "coding-agent", "reasoning", "tool-use", "workflow-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI", "cs.LG" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "16", "collection_queries": "ai-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-19409-openrath-session-centered-runtime-state-for-agent-systems.md", "title": "\"OpenRath: Session-Centered Runtime State for Agent Systems\"", "type": "paper", "meta": { "type": "paper", "title": "\"OpenRath: Session-Centered Runtime State for Agent Systems\"", "authors": "Fukang Wen, Zhijie Wang, Ruilin Xu", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.19409", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-17", "updated_at": "2026-06-17", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "memory", "multi-agent", "rag", "tool-use", "workflow-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.SE", "cs.PL" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "14", "collection_queries": "agent-memory" } }, { "collection": "papers", "path": "papers/items/2026-2606-19464-deontic-policies-for-runtime-governance-of-agentic-ai-systems.md", "title": "Deontic Policies for Runtime Governance of Agentic AI Systems", "type": "paper", "meta": { "type": "paper", "title": "Deontic Policies for Runtime Governance of Agentic AI Systems", "authors": "Anupam Joshi, Tim Finin, Karuna Pande Joshi, Lalana Kagal", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.19464", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-17", "updated_at": "2026-06-17", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "reasoning", "tool-use", "workflow-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI", "cs.MA" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "16", "collection_queries": "agentic-ai, autonomous-agent-llm" } }, { "collection": "papers", "path": "papers/items/2026-2606-19613-staminabench-stress-testing-coding-agents-over-100-interaction-turns.md", "title": "\"StaminaBench: Stress-Testing Coding Agents over 100 Interaction Turns\"", "type": "paper", "meta": { "type": "paper", "title": "\"StaminaBench: Stress-Testing Coding Agents over 100 Interaction Turns\"", "authors": "Vlad Sobal, Shuo Yang, Yuting Zhang, Wei Xia, Stefano Soatto", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.19613", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-17", "updated_at": "2026-06-17", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "coding-agent", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.SE", "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "14", "collection_queries": "coding-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-19704-beyond-static-leaderboards-predictive-validity-for-the-evaluation-of-llm-agents.md", "title": "\"Beyond Static Leaderboards: Predictive Validity for the Evaluation of LLM Agents\"", "type": "paper", "meta": { "type": "paper", "title": "\"Beyond Static Leaderboards: Predictive Validity for the Evaluation of LLM Agents\"", "authors": "Dhaval C. Patel, Kaoutar El Maghraoui, Shuxin Lin, Yusheng Li, Tianjun Feng, Chun-Yi Tsai, Yihan Sun, Wei Alexander Xin, et al.", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.19704", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-18", "updated_at": "2026-06-18", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "rag", "reasoning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "19", "collection_queries": "agent-evaluation" } }, { "collection": "papers", "path": "papers/items/2026-2606-19787-oragentbench-can-llm-agents-solve-challenging-operations-research-tasks-end-to-e.md", "title": "\"ORAgentBench: Can LLM Agents Solve Challenging Operations Research Tasks End to End?\"", "type": "paper", "meta": { "type": "paper", "title": "\"ORAgentBench: Can LLM Agents Solve Challenging Operations Research Tasks End to End?\"", "authors": "Jiajun Li, Mingshu Cai, Yixuan Li, Yu Ding, Ran Hou, Guanyu Nie, Xiongwei Han, Wanyuan Wang", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.19787", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-18", "updated_at": "2026-06-18", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "rag", "workflow-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "15", "collection_queries": "autonomous-agent-llm" } }, { "collection": "papers", "path": "papers/items/2026-2606-19812-human-on-the-loop-orchestration-for-ai-assisted-legal-discovery.md", "title": "Human-on-the-Loop Orchestration for AI-Assisted Legal Discovery", "type": "paper", "meta": { "type": "paper", "title": "Human-on-the-Loop Orchestration for AI-Assisted Legal Discovery", "authors": "Anushree Sinha, Srivaths Ranganathan, Abhishek Dharmaratnakar, Debanshu Das", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.19812", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-18", "updated_at": "2026-06-18", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "planning", "rag", "reasoning", "workflow-agent", "world-model" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI", "cs.LG" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "14", "collection_queries": "agentic-ai, planning-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-19852-prompt-plan-extract-zero-shot-agentic-llms-workflows-for-lung-pathology-extracti.md", "title": "\"Prompt, Plan, Extract: Zero-Shot Agentic LLMs Workflows for Lung Pathology Extraction from Clinical Narratives\"", "type": "paper", "meta": { "type": "paper", "title": "\"Prompt, Plan, Extract: Zero-Shot Agentic LLMs Workflows for Lung Pathology Extraction from Clinical Narratives\"", "authors": "Aman Pathak, Cheng Peng, Mengxian Lyu, Ziyi Chen, Reema Solan, Sankalp Talankar, Yasir Khan, Hiren Mehta, et al.", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.19852", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-18", "updated_at": "2026-06-25", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "planning", "tool-use", "workflow-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CL", "cs.LG" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "15", "collection_queries": "agentic-ai" } }, { "collection": "papers", "path": "papers/items/2026-2606-19899-measuring-biological-capabilities-and-risks-of-ai-agents.md", "title": "Measuring Biological Capabilities and Risks of AI Agents", "type": "paper", "meta": { "type": "paper", "title": "Measuring Biological Capabilities and Risks of AI Agents", "authors": "Patricia Paskov, Jeffrey Lee, Kyle Brady, Alyssa Worland", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.19899", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-18", "updated_at": "2026-06-18", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "computer-use", "rag", "tool-use", "workflow-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CY", "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "18", "collection_queries": "agent-evaluation, agentic-ai, ai-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-19926-memgui-agent-an-end-to-end-long-horizon-mobile-gui-agent-with-proactive-context-.md", "title": "\"MemGUI-Agent: An End-to-End Long-Horizon Mobile GUI Agent with Proactive Context Management\"", "type": "paper", "meta": { "type": "paper", "title": "\"MemGUI-Agent: An End-to-End Long-Horizon Mobile GUI Agent with Proactive Context Management\"", "authors": "Guangyi Liu, Gao Wu, Congxiao Liu, Pengxiang Zhao, Liang Liu, Mading Li, Qi Zhang, Mengyan Wang, et al.", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.19926", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-18", "updated_at": "2026-06-18", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "computer-use", "planning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.HC" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "15", "collection_queries": "web-gui-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-19930-mobileforge-annotation-free-adaptation-for-mobile-gui-agents-with-hierarchical-f.md", "title": "\"MobileForge: Annotation-Free Adaptation for Mobile GUI Agents with Hierarchical Feedback-Guided Policy Optimization\"", "type": "paper", "meta": { "type": "paper", "title": "\"MobileForge: Annotation-Free Adaptation for Mobile GUI Agents with Hierarchical Feedback-Guided Policy Optimization\"", "authors": "Guangyi Liu, Pengxiang Zhao, Gao Wu, Yiwen Yin, Mading Li, Liang Liu, Congxiao Liu, Zhang Qi, et al.", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.19930", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-18", "updated_at": "2026-06-18", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "computer-use", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.HC" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "14", "collection_queries": "web-gui-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-19980-enpire-agentic-robot-policy-self-improvement-in-the-real-world.md", "title": "\"ENPIRE: Agentic Robot Policy Self-Improvement in the Real World\"", "type": "paper", "meta": { "type": "paper", "title": "\"ENPIRE: Agentic Robot Policy Self-Improvement in the Real World\"", "authors": "\"Wenli Xiao, Jia Xie, Tonghe Zhang, Haotian Lin, Letian \\\"Max\\\" Fu, Haoru Xue, Jalen Lu, Yi Yang, et al.\"", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.19980", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-18", "updated_at": "2026-06-18", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "coding-agent", "embodied-agent", "reasoning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "16", "collection_queries": "coding-agent, tool-use" } }, { "collection": "papers", "path": "papers/items/2026-2606-20023-when-lower-privileges-suffice-investigating-over-privileged-tool-selection-in-ll.md", "title": "\"When Lower Privileges Suffice: Investigating Over-Privileged Tool Selection in LLM Agents\"", "type": "paper", "meta": { "type": "paper", "title": "\"When Lower Privileges Suffice: Investigating Over-Privileged Tool Selection in LLM Agents\"", "authors": "Kaiyue Yang, Yuyan Bu, Jingwei Yi, Yuchi Wang, Biyu Zhou, Juntao Dai, Songlin Hu, Yaodong Yang", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.20023", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-18", "updated_at": "2026-07-07", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.SE", "cs.AI", "cs.CL" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "14", "collection_queries": "tool-use" } }, { "collection": "papers", "path": "papers/items/2026-2606-20041-ai-economist-agent-an-agentic-framework-for-model-grounded-economic-analysis-wit.md", "title": "\"AI Economist Agent: An Agentic Framework for Model-Grounded Economic Analysis with RAG, Knowledge Graphs, and Large Language Models\"", "type": "paper", "meta": { "type": "paper", "title": "\"AI Economist Agent: An Agentic Framework for Model-Grounded Economic Analysis with RAG, Knowledge Graphs, and Large Language Models\"", "authors": "Masahiro Kato", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.20041", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-18", "updated_at": "2026-06-18", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "planning", "rag" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "econ.GN", "cs.AI", "cs.LG", "q-fin.GN" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "15", "collection_queries": "ai-agent, rag-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-20047-pacms-submodular-context-selection-as-a-pluggable-engine-for-llm-agents.md", "title": "\"PACMS: Submodular Context Selection as a Pluggable Engine for LLM Agents\"", "type": "paper", "meta": { "type": "paper", "title": "\"PACMS: Submodular Context Selection as a Pluggable Engine for LLM Agents\"", "authors": "Manu Ghulyani, Arunabh Singh, Karan Bharadwaj, Ankit Nath, Suranjan Goswami", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.20047", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-18", "updated_at": "2026-06-18", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "memory", "rag", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.IR" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "15", "collection_queries": "tool-use" } }, { "collection": "papers", "path": "papers/items/2026-2606-20243-phoenix-safe-github-issue-resolution-via-multi-agent-llms.md", "title": "\"Phoenix: Safe GitHub Issue Resolution via Multi-Agent LLMs\"", "type": "paper", "meta": { "type": "paper", "title": "\"Phoenix: Safe GitHub Issue Resolution via Multi-Agent LLMs\"", "authors": "Kipngeno Koech, Muhammad Adam, Baimam Boukar Jean Jacques, Joao Barros", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.20243", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-18", "updated_at": "2026-06-22", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "coding-agent", "multi-agent", "planning", "rag" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.SE", "cs.MA" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "18", "collection_queries": "coding-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-20401-poweragentbench-dyn-a-benchmark-for-agentic-ai-in-power-system-dynamic-studies.md", "title": "\"PowerAgentBench-Dyn: A Benchmark for Agentic AI in Power System Dynamic Studies\"", "type": "paper", "meta": { "type": "paper", "title": "\"PowerAgentBench-Dyn: A Benchmark for Agentic AI in Power System Dynamic Studies\"", "authors": "Qian Zhang, Andrea Pomarico, Costas Mylonas, Magda Foti, Alberto Berizzi, Le Xie", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.20401", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-18", "updated_at": "2026-06-18", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "coding-agent", "memory", "planning", "rag", "reasoning", "tool-use", "world-model" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "eess.SY" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "24", "collection_queries": "agentic-ai" } }, { "collection": "papers", "path": "papers/items/2026-2606-20470-analyzing-defensive-misdirection-against-model-guided-automated-attacks-on-agent.md", "title": "Analyzing Defensive Misdirection Against Model-Guided Automated Attacks on Agentic AI Systems", "type": "paper", "meta": { "type": "paper", "title": "Analyzing Defensive Misdirection Against Model-Guided Automated Attacks on Agentic AI Systems", "authors": "Reza Soosahabi, Vivek Namsani", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.20470", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-18", "updated_at": "2026-06-26", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "computer-use", "tool-use", "workflow-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CR", "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "14", "collection_queries": "agentic-ai" } }, { "collection": "papers", "path": "papers/items/2026-2606-20479-groundcontrol-anticipating-navigation-failures-in-vision-language-agents-via-tra.md", "title": "\"GroundControl: Anticipating Navigation Failures in Vision-Language Agents via Trajectory-Consistent Uncertainty Estimates\"", "type": "paper", "meta": { "type": "paper", "title": "\"GroundControl: Anticipating Navigation Failures in Vision-Language Agents via Trajectory-Consistent Uncertainty Estimates\"", "authors": "Nastaran Darabi, Divake Kumar, Sina Tayebati, Devashri Naik, Amit Ranjan Trivedi", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.20479", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-18", "updated_at": "2026-06-18", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "embodied-agent", "rag", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.RO" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "15", "collection_queries": "language-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-20510-efficient-and-sound-probabilistic-verification-for-ai-agents.md", "title": "Efficient and Sound Probabilistic Verification for AI Agents", "type": "paper", "meta": { "type": "paper", "title": "Efficient and Sound Probabilistic Verification for AI Agents", "authors": "Alaia Solko-Breslin, Pramod Kaushik Mudrakarta, Mihai Christodorescu, Somesh Jha, Krishnamurthy Dj Dvijotham", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.20510", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-18", "updated_at": "2026-06-18", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "computer-use", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CR", "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "13", "collection_queries": "ai-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-20512-probe-and-refine-tuning-of-repository-guidance-for-coding-agents.md", "title": "Probe-and-Refine Tuning of Repository Guidance for Coding Agents", "type": "paper", "meta": { "type": "paper", "title": "Probe-and-Refine Tuning of Repository Guidance for Coding Agents", "authors": "Asa Shepard, Jeannie Albrecht", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.20512", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-18", "updated_at": "2026-06-19", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "coding-agent", "computer-use", "rag", "tool-use", "workflow-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.SE", "cs.LG" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "19", "collection_queries": "coding-agent, tool-use" } }, { "collection": "papers", "path": "papers/items/2026-2606-20515-s-agent-spatial-tool-use-elicits-reasoning-for-spatial-intelligence.md", "title": "\"S-Agent: Spatial Tool-Use Elicits Reasoning for Spatial Intelligence\"", "type": "paper", "meta": { "type": "paper", "title": "\"S-Agent: Spatial Tool-Use Elicits Reasoning for Spatial Intelligence\"", "authors": "Yalun Dai, Hao Li, Shulin Tian, Runmao Yao, Yuhao Dong, Fangzhou Hong, Zhaoxi Chen, Fangfu Liu, et al.", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.20515", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-18", "updated_at": "2026-06-28", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "memory", "planning", "reasoning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CV" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "14", "collection_queries": "agent-memory, tool-use" } }, { "collection": "papers", "path": "papers/items/2026-2606-20573-aona-a-comprehensive-architecture-and-workflow-design-for-global-agentic-collabo.md", "title": "\"AONA: A Comprehensive Architecture and Workflow Design for Global Agentic Collaboration\"", "type": "paper", "meta": { "type": "paper", "title": "\"AONA: A Comprehensive Architecture and Workflow Design for Global Agentic Collaboration\"", "authors": "Jinliang Xu, Runkai Zhu, Bingqi Li, Fanjie Nie, Jin Li, Jiagui Xie", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.20573", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-04-30", "updated_at": "2026-04-30", "status": "queued", "relevance": "high", "topics": [ "multi-agent", "tool-use", "workflow-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.NI", "cs.MA" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "14", "collection_queries": "autonomous-agent-llm" } }, { "collection": "papers", "path": "papers/items/2026-2606-20629-specialize-roles-mix-deployments-pushing-the-cost-accuracy-frontier-of-llm-agent.md", "title": "\"Specialize Roles, Mix Deployments: Pushing the Cost-Accuracy Frontier of LLM Agent Teams\"", "type": "paper", "meta": { "type": "paper", "title": "\"Specialize Roles, Mix Deployments: Pushing the Cost-Accuracy Frontier of LLM Agent Teams\"", "authors": "Yinsicheng Jiang, Liang Cheng, Yeqi Huang, Yufan Zhao, Zhan Lu, Li Dong, Wenda Li, Edoardo Ponti, et al.", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.20629", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-05-28", "updated_at": "2026-05-28", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "computer-use", "planning", "reasoning", "tool-use", "workflow-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.MA", "cs.AI", "cs.LG" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "19", "collection_queries": "agent-evaluation" } }, { "collection": "papers", "path": "papers/items/2026-2606-20717-mirage-stealthy-visual-prompt-injection-for-vulnerability-detection-in-web-agent.md", "title": "\"MIRAGE: Stealthy Visual Prompt Injection for Vulnerability Detection in Web Agents\"", "type": "paper", "meta": { "type": "paper", "title": "\"MIRAGE: Stealthy Visual Prompt Injection for Vulnerability Detection in Web Agents\"", "authors": "Xuelong Dai, Jianyu Ma, Boyang Ma, Biwei Yan, Yijun Yang, Yue Zhang", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.20717", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-16", "updated_at": "2026-06-16", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "computer-use", "rag", "tool-use", "workflow-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CV", "cs.AI", "cs.CR" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "14", "collection_queries": "web-gui-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-20785-fara-1-5-scalable-learning-environments-for-computer-use-agents.md", "title": "\"Fara-1.5: Scalable Learning Environments for Computer Use Agents\"", "type": "paper", "meta": { "type": "paper", "title": "\"Fara-1.5: Scalable Learning Environments for Computer Use Agents\"", "authors": "Ahmed Awadallah, Sahil Gupta, Yash Lara, Yadong Lu, Hussein Mozannar, Akshay Nambi, Zach Nussbaum, Yash Pandya, et al.", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.20785", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-18", "updated_at": "2026-06-18", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "computer-use", "rag", "reasoning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI", "cs.LG" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "14", "collection_queries": "web-gui-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-20922-think-twice-before-you-act-protecting-llm-agents-against-tool-description-poison.md", "title": "\"Think Twice Before You Act: Protecting LLM Agents Against Tool Description Poisoning via Isolated Planning\"", "type": "paper", "meta": { "type": "paper", "title": "\"Think Twice Before You Act: Protecting LLM Agents Against Tool Description Poisoning via Isolated Planning\"", "authors": "Shanghao Shi, Xiao Wang, Chaoyu Zhang, Hao Li, Wenjing Lou, Thomas Hou, Yevgeniy Vorobeychik, Chongjie Zhang, et al.", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.20922", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-18", "updated_at": "2026-06-18", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "planning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CR" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "15", "collection_queries": "planning-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-20950-power-systems-agent-benchmark-executable-evaluation-of-ai-agents-in-electric-pow.md", "title": "\"Power Systems Agent Benchmark: Executable Evaluation of AI Agents in Electric Power Engineering\"", "type": "paper", "meta": { "type": "paper", "title": "\"Power Systems Agent Benchmark: Executable Evaluation of AI Agents in Electric Power Engineering\"", "authors": "Sergei Trashchenkov", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.20950", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-18", "updated_at": "2026-07-02", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "rag", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI", "eess.SY" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "19", "collection_queries": "agent-evaluation, ai-agent, tool-use" } }, { "collection": "papers", "path": "papers/items/2026-2606-20954-learning-what-not-to-forget-long-horizon-agent-memory-from-a-few-kilobytes-of-le.md", "title": "\"Learning What Not to Forget: Long-Horizon Agent Memory from a Few Kilobytes of Learning\"", "type": "paper", "meta": { "type": "paper", "title": "\"Learning What Not to Forget: Long-Horizon Agent Memory from a Few Kilobytes of Learning\"", "authors": "Nusrat Jahan Lia, Aritra Mazumder", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.20954", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-18", "updated_at": "2026-06-18", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "memory", "planning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CL", "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "16", "collection_queries": "agent-memory" } }, { "collection": "papers", "path": "papers/items/2026-2606-21013-agentic-time-machine-as-an-infrastructure-for-future-event-forecasting.md", "title": "Agentic Time Machine as an Infrastructure for Future-Event Forecasting", "type": "paper", "meta": { "type": "paper", "title": "Agentic Time Machine as an Infrastructure for Future-Event Forecasting", "authors": "Jingyi Chai, Bingyang Zheng, Xiangrui Liu, Hao Lu, Zihang Zhou, Tianchen Wang, Kemeng Zhang, Siheng Chen", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.21013", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-19", "updated_at": "2026-06-19", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "multi-agent", "planning", "rag", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI", "cs.LG" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "18", "collection_queries": "agent-evaluation" } }, { "collection": "papers", "path": "papers/items/2026-2606-21123-a-multi-agent-audit-framework-for-high-stakes-reasoning-evaluation-and-interpret.md", "title": "\"A Multi-Agent Audit Framework for High-Stakes Reasoning: Evaluation and Interpretability in Clinical Mental Health Screening\"", "type": "paper", "meta": { "type": "paper", "title": "\"A Multi-Agent Audit Framework for High-Stakes Reasoning: Evaluation and Interpretability in Clinical Mental Health Screening\"", "authors": "Jingchen Ye, Yanpei Yu, Luyao Zhang", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.21123", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-19", "updated_at": "2026-06-19", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "multi-agent", "rag", "reasoning", "workflow-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CL" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "18", "collection_queries": "rag-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-21129-agenticos-an-intent-oriented-secure-operating-system-architecture-for-autonomous.md", "title": "\"AgenticOS: An Intent-Oriented Secure Operating System Architecture for Autonomous AI Agents\"", "type": "paper", "meta": { "type": "paper", "title": "\"AgenticOS: An Intent-Oriented Secure Operating System Architecture for Autonomous AI Agents\"", "authors": "Zhen Zhao, Yu Zhang, Yanpeng Zhu, Jia Wang, Songqiao Tao, Xin Cheng, Jiexin Gao", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.21129", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-19", "updated_at": "2026-06-19", "status": "queued", "relevance": "high", "topics": [ "agent-safety", "planning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CR", "cs.OS" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "19", "collection_queries": "ai-agent, autonomous-agent-llm, tool-use" } }, { "collection": "papers", "path": "papers/items/2026-2606-21228-sakana-fugu-technical-report.md", "title": "Sakana Fugu Technical Report", "type": "paper", "meta": { "type": "paper", "title": "Sakana Fugu Technical Report", "authors": "Yujin Tang, Edoardo Cetin, Jinglue Xu, Qi Sun, Stefan Nielsen, Vincent Richard, Haruto Goda, Iaroslav Tymchenko, et al.", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.21228", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-19", "updated_at": "2026-06-23", "status": "queued", "relevance": "high", "topics": [ "coding-agent", "multi-agent", "rag", "reasoning" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.LG" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "13", "collection_queries": "coding-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-21401-swarmx-agentic-scheduling-for-low-latency-agentic-systems.md", "title": "\"SwarmX: Agentic Scheduling for Low-Latency Agentic Systems\"", "type": "paper", "meta": { "type": "paper", "title": "\"SwarmX: Agentic Scheduling for Low-Latency Agentic Systems\"", "authors": "Yeqi Huang, Yanwei Ye, Guomin Chen, Wenhao Su, Bin Gong, Jialian Li, Zhan Lu, Yangshen Deng, et al.", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.21401", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-19", "updated_at": "2026-06-28", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "multi-agent", "tool-use", "workflow-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.DC", "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "13", "collection_queries": "agentic-ai" } }, { "collection": "papers", "path": "papers/items/2026-2606-21409-don-t-blindly-trust-it-how-unreliable-feedback-breaks-tool-using-llm-agents.md", "title": "\"Don't Blindly Trust It: How Unreliable Feedback Breaks Tool-Using LLM Agents\"", "type": "paper", "meta": { "type": "paper", "title": "\"Don't Blindly Trust It: How Unreliable Feedback Breaks Tool-Using LLM Agents\"", "authors": "Chubin Zhang, Zhenglin Wan, Xingrui Yu, Pengfei Zhou, Wangbo Zhao, Jingxuan Wu, Yaxin Zhou, Ivor Tsang", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.21409", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-19", "updated_at": "2026-06-19", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "coding-agent", "rag", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "14", "collection_queries": "tool-use" } }, { "collection": "papers", "path": "papers/items/2026-2606-21445-autoras-learning-robust-agentic-systems-with-primitive-representations.md", "title": "\"AutoRAS: Learning Robust Agentic Systems with Primitive Representations\"", "type": "paper", "meta": { "type": "paper", "title": "\"AutoRAS: Learning Robust Agentic Systems with Primitive Representations\"", "authors": "Yang Yue, Xuancheng Zhu, Yuyang Ma, Guoshun Nan, Zihan Dou, Jingru Shan, Congyu Guo, Ji Zhang, et al.", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.21445", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-19", "updated_at": "2026-06-19", "status": "queued", "relevance": "high", "topics": [ "agent-safety", "multi-agent", "reasoning", "tool-use", "workflow-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "16", "collection_queries": "agentic-ai" } }, { "collection": "papers", "path": "papers/items/2026-2606-21553-dissecting-agentic-rag-a-component-ablation-for-multi-hop-qa-with-a-local-7b-mod.md", "title": "\"Dissecting Agentic RAG: A Component Ablation for Multi-Hop QA with a Local 7B Model\"", "type": "paper", "meta": { "type": "paper", "title": "\"Dissecting Agentic RAG: A Component Ablation for Multi-Hop QA with a Local 7B Model\"", "authors": "Sheroz Shaikh", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.21553", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-19", "updated_at": "2026-06-19", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "rag", "reasoning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CL", "cs.IR" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "14", "collection_queries": "rag-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-21565-composing-verifiable-conceptual-models-via-building-blocks-towards-design-time-v.md", "title": "\"Composing Verifiable Conceptual Models via Building Blocks: Towards Design-Time Verification of Agentic AI Workflows\"", "type": "paper", "meta": { "type": "paper", "title": "\"Composing Verifiable Conceptual Models via Building Blocks: Towards Design-Time Verification of Agentic AI Workflows\"", "authors": "Noe Y. Flandre, Alexander C. Nwala, Philippe J. Giabbanelli", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.21565", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-19", "updated_at": "2026-06-19", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "reasoning", "tool-use", "workflow-agent", "world-model" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "15", "collection_queries": "agentic-ai" } }, { "collection": "papers", "path": "papers/items/2026-2606-21627-counsel-a-meta-evaluation-dataset-for-agentic-tasks.md", "title": "\"Counsel: A Meta-Evaluation Dataset for Agentic Tasks\"", "type": "paper", "meta": { "type": "paper", "title": "\"Counsel: A Meta-Evaluation Dataset for Agentic Tasks\"", "authors": "Sashank Pisupati, Henry Broomfield, Eujeong Choi, Antonia Calvi, Charlie Wang, Roman Engeler, Max Bartolo, Patrick Lewis", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.21627", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-19", "updated_at": "2026-06-19", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "coding-agent", "reasoning" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI", "cs.LG" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "20", "collection_queries": "agent-evaluation, coding-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-21649-evoembedding-evolvable-representations-for-long-context-retrieval-and-agentic-me.md", "title": "\"EvoEmbedding: Evolvable Representations for Long-Context Retrieval and Agentic Memory\"", "type": "paper", "meta": { "type": "paper", "title": "\"EvoEmbedding: Evolvable Representations for Long-Context Retrieval and Agentic Memory\"", "authors": "Chang Nie, Chaoyou Fu, Junlan Feng, Caifeng Shan", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.21649", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-19", "updated_at": "2026-06-25", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "coding-agent", "memory", "rag", "workflow-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CL" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "19", "collection_queries": "agent-memory, agentic-ai, rag-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-21710-privacyalign-contextual-privacy-alignment-for-llm-agents.md", "title": "\"PrivacyAlign: Contextual Privacy Alignment for LLM Agents\"", "type": "paper", "meta": { "type": "paper", "title": "\"PrivacyAlign: Contextual Privacy Alignment for LLM Agents\"", "authors": "Manveer Singh Tamber, Abhay Puri, Marc-Etienne Brunet, Perouz Taslakian, Jimmy Lin, Spandana Gella", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.21710", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-19", "updated_at": "2026-06-19", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "planning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CL", "cs.AI", "cs.IR" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "18", "collection_queries": "ai-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-21732-safe-to-check-unsafe-to-use-relinking-at-the-compression-boundary-of-llm-agents.md", "title": "\"Safe to Check, Unsafe to Use: Relinking at the Compression Boundary of LLM Agents\"", "type": "paper", "meta": { "type": "paper", "title": "\"Safe to Check, Unsafe to Use: Relinking at the Compression Boundary of LLM Agents\"", "authors": "Zesen Liu, Zihan Zhang, Dongdong She", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.21732", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-19", "updated_at": "2026-06-19", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "rag", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CR", "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "15", "collection_queries": "agent-evaluation" } }, { "collection": "papers", "path": "papers/items/2026-2606-21740-training-the-orchestrator-a-supervised-approach-to-end-to-end-pddl-planning-with.md", "title": "\"Training the Orchestrator: A Supervised Approach to End-to-End PDDL Planning with LLM Agents\"", "type": "paper", "meta": { "type": "paper", "title": "\"Training the Orchestrator: A Supervised Approach to End-to-End PDDL Planning with LLM Agents\"", "authors": "Rajesh Mangannavar, Zachary Coalson, Pranay Dugar, Prasad Tadepalli", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.21740", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-19", "updated_at": "2026-06-19", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "computer-use", "planning", "reasoning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "17", "collection_queries": "planning-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-21836-agentdse-reasoning-augmented-architectural-design-space-exploration.md", "title": "\"AgentDSE: Reasoning-Augmented Architectural Design Space Exploration\"", "type": "paper", "meta": { "type": "paper", "title": "\"AgentDSE: Reasoning-Augmented Architectural Design Space Exploration\"", "authors": "Chenyu Wang, Jiahe Caroline Shi, David Kong, Duane S. Boning, Zishen Wan, Yilun Du, Vijay Janapa Reddi", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.21836", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-20", "updated_at": "2026-06-20", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "coding-agent", "computer-use", "reasoning" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AR", "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "13", "collection_queries": "coding-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-21842-agent-assisted-side-channel-attacks-on-non-prefix-kv-cache-in-rag.md", "title": "Agent-Assisted Side-Channel Attacks on Non-Prefix KV Cache in RAG", "type": "paper", "meta": { "type": "paper", "title": "Agent-Assisted Side-Channel Attacks on Non-Prefix KV Cache in RAG", "authors": "He Sun, Shinan Liu, Siyuan Ma, Junhao Li, Mingjun Xiao, Wenhao Jiang", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.21842", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-20", "updated_at": "2026-06-20", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "memory", "rag", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CR" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "14", "collection_queries": "rag-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-21877-agentriskbom-a-risk-scoping-security-bill-of-materials-for-agentic-ai-systems.md", "title": "\"AgentRiskBOM: A Risk-Scoping Security Bill of Materials for Agentic AI Systems\"", "type": "paper", "meta": { "type": "paper", "title": "\"AgentRiskBOM: A Risk-Scoping Security Bill of Materials for Agentic AI Systems\"", "authors": "Srimonti Dutta, Akshata Kishore Moharir", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.21877", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-20", "updated_at": "2026-06-20", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "coding-agent", "memory", "multi-agent", "rag", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI", "cs.CR", "cs.SE" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "21", "collection_queries": "agentic-ai, ai-agent, rag-agent, tool-use" } }, { "collection": "papers", "path": "papers/items/2026-2606-21963-holmes-multimodal-agentic-diagnosis-for-mixed-language-mobile-crashes-at-industr.md", "title": "\"Holmes: Multimodal Agentic Diagnosis for Mixed-Language Mobile Crashes at Industrial Scale\"", "type": "paper", "meta": { "type": "paper", "title": "\"Holmes: Multimodal Agentic Diagnosis for Mixed-Language Mobile Crashes at Industrial Scale\"", "authors": "Jia Li, Wenyuan Ma, Ting Peng, Haibin Zheng, Yuetang Deng", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.21963", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-20", "updated_at": "2026-06-20", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "computer-use", "multi-agent", "rag", "workflow-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI", "cs.SE" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "16", "collection_queries": "multi-agent-llm" } }, { "collection": "papers", "path": "papers/items/2026-2606-22030-nous-a-predictive-world-model-for-long-term-agent-memory.md", "title": "\"Nous: A Predictive World Model for Long-Term Agent Memory\"", "type": "paper", "meta": { "type": "paper", "title": "\"Nous: A Predictive World Model for Long-Term Agent Memory\"", "authors": "Pranav Singh", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.22030", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-20", "updated_at": "2026-06-20", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "memory", "rag", "world-model" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI", "cs.CL", "cs.IR", "cs.LG" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "14", "collection_queries": "agent-memory" } }, { "collection": "papers", "path": "papers/items/2026-2606-22082-codeteam-an-llm-powered-multi-agent-framework-for-repository-level-code-generati.md", "title": "\"CodeTeam: An LLM-Powered Multi-Agent Framework for Repository-Level Code Generation\"", "type": "paper", "meta": { "type": "paper", "title": "\"CodeTeam: An LLM-Powered Multi-Agent Framework for Repository-Level Code Generation\"", "authors": "Yifei Wang, Ruiyin Li, Peng Liang, Qiong Feng, Zengyang Li, Mojtaba Shahin, Arif Ali Khan", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.22082", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-20", "updated_at": "2026-06-20", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "coding-agent", "multi-agent", "planning", "rag" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.SE", "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "16", "collection_queries": "multi-agent-llm" } }, { "collection": "papers", "path": "papers/items/2026-2606-22110-traceview-interactive-visualization-of-agentic-program-repair-trajectories.md", "title": "\"TraceView: Interactive Visualization of Agentic Program Repair Trajectories\"", "type": "paper", "meta": { "type": "paper", "title": "\"TraceView: Interactive Visualization of Agentic Program Repair Trajectories\"", "authors": "Amirali Sajadi, Tu Nguyen, Kimmie Huynh, Esteban Parra, Preetha Chatterjee", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.22110", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-20", "updated_at": "2026-06-20", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "coding-agent", "reasoning", "tool-use", "workflow-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.SE", "cs.AI", "cs.HC" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "16", "collection_queries": "tool-use" } }, { "collection": "papers", "path": "papers/items/2026-2606-22151-novelty-aware-agentic-retrieval-comparing-research-contributions-through-structu.md", "title": "\"Novelty-Aware Agentic Retrieval: Comparing Research Contributions Through Structured Multi-Step Reasoning\"", "type": "paper", "meta": { "type": "paper", "title": "\"Novelty-Aware Agentic Retrieval: Comparing Research Contributions Through Structured Multi-Step Reasoning\"", "authors": "Shou-Tzu Han", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.22151", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-20", "updated_at": "2026-06-20", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "computer-use", "rag", "reasoning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.IR" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "14", "collection_queries": "rag-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-22263-revelio-cost-efficient-agentic-memory-safety-vulnerability-detection-for-reposit.md", "title": "\"Revelio: Cost-Efficient Agentic Memory Safety Vulnerability Detection For Repository-Scale Codebases\"", "type": "paper", "meta": { "type": "paper", "title": "\"Revelio: Cost-Efficient Agentic Memory Safety Vulnerability Detection For Repository-Scale Codebases\"", "authors": "Yiwei Hou, Hao Wang, Muxi Lyu, Marius Momeu, Eric Nguyen, Taige Yang, Koushik Sen, Dawn Song, et al.", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.22263", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-20", "updated_at": "2026-06-20", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "coding-agent", "memory" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CR", "cs.AI", "cs.MA", "cs.SE" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "20", "collection_queries": "agent-memory, coding-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-22330-hypothesis-driven-skill-optimization-for-llm-agents.md", "title": "Hypothesis-Driven Skill Optimization for LLM Agents", "type": "paper", "meta": { "type": "paper", "title": "Hypothesis-Driven Skill Optimization for LLM Agents", "authors": "Fangxin Shang, Yehui Yang", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.22330", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-21", "updated_at": "2026-06-21", "status": "queued", "relevance": "high", "topics": [ "agent-safety", "coding-agent", "memory", "planning", "reasoning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI", "cs.SE" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "17", "collection_queries": "planning-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-22388-planbench-xl-evaluating-long-horizon-planning-of-llm-tool-use-agents-in-large-sc.md", "title": "\"PlanBench-XL: Evaluating Long-Horizon Planning of LLM Tool-Use Agents in Large-Scale Tool Ecosystems\"", "type": "paper", "meta": { "type": "paper", "title": "\"PlanBench-XL: Evaluating Long-Horizon Planning of LLM Tool-Use Agents in Large-Scale Tool Ecosystems\"", "authors": "Jiayu Liu, Qihan Lin, Cheng Qian, Rui Wang, Emre Can Acikgoz, Xiaocheng Yang, Jiateng Liu, Zhenhailong Wang, et al.", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.22388", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-21", "updated_at": "2026-06-21", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "planning", "rag", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI", "cs.CL" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "15", "collection_queries": "planning-agent, tool-use" } }, { "collection": "papers", "path": "papers/items/2026-2606-22417-code-isn-t-memory-a-structural-codebase-index-inside-a-coding-agent.md", "title": "\"Code Isn't Memory: A Structural Codebase Index Inside a Coding Agent\"", "type": "paper", "meta": { "type": "paper", "title": "\"Code Isn't Memory: A Structural Codebase Index Inside a Coding Agent\"", "authors": "Ishaan Bhola, Adithyan Krishnan, Sravanth Kurmala, Mukunda NS", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.22417", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-21", "updated_at": "2026-06-21", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "coding-agent", "memory", "rag" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "19", "collection_queries": "coding-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-22484-governed-ai-assisted-engineering-graduated-human-oversight-for-agentic-code-gene.md", "title": "\"Governed AI-Assisted Engineering: Graduated Human Oversight for Agentic Code Generation in Regulated Domains\"", "type": "paper", "meta": { "type": "paper", "title": "\"Governed AI-Assisted Engineering: Graduated Human Oversight for Agentic Code Generation in Regulated Domains\"", "authors": "Richard Kang", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.22484", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-21", "updated_at": "2026-07-04", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "coding-agent", "rag", "workflow-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.HC" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "14", "collection_queries": "agentic-ai" } }, { "collection": "papers", "path": "papers/items/2026-2606-22495-grounded-scaling-why-agentic-ai-needs-deterministic-environments.md", "title": "\"Grounded Scaling: Why Agentic AI Needs Deterministic Environments\"", "type": "paper", "meta": { "type": "paper", "title": "\"Grounded Scaling: Why Agentic AI Needs Deterministic Environments\"", "authors": "Liang Ding, Xintong Wang", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.22495", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-21", "updated_at": "2026-06-21", "status": "queued", "relevance": "high", "topics": [ "agent-safety", "embodied-agent", "multi-agent", "reasoning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "14", "collection_queries": "agentic-ai" } }, { "collection": "papers", "path": "papers/items/2026-2606-22557-macagentbench-benchmarking-ai-agents-on-real-world-macos-desktop.md", "title": "\"MacAgentBench: Benchmarking AI Agents on Real-World macOS Desktop\"", "type": "paper", "meta": { "type": "paper", "title": "\"MacAgentBench: Benchmarking AI Agents on Real-World macOS Desktop\"", "authors": "Yikun Fu, Bowen Fu, Zhenyu Wu, Shuang Cheng, Xiaowei Sun, Bowen Yang, Zehao Li, Yibo Zhao, et al.", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.22557", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-21", "updated_at": "2026-06-21", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "computer-use", "planning", "rag", "tool-use", "workflow-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI", "cs.CL", "cs.HC" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "21", "collection_queries": "agent-evaluation, ai-agent, web-gui-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-22610-paperclaw-harnessing-agents-for-autonomous-research-and-human-in-the-loop-refine.md", "title": "\"PaperClaw: Harnessing Agents for Autonomous Research and Human-in-the-Loop Refinement\"", "type": "paper", "meta": { "type": "paper", "title": "\"PaperClaw: Harnessing Agents for Autonomous Research and Human-in-the-Loop Refinement\"", "authors": "Weiwei Ye, Hangchen Liu, Dongyuan Li, Renhe Jiang", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.22610", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-21", "updated_at": "2026-06-21", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "memory", "multi-agent", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "15", "collection_queries": "multi-agent-llm" } }, { "collection": "papers", "path": "papers/items/2026-2606-22647-raven-agentic-rag-for-automated-vulnerability-repair.md", "title": "\"RAVEN: Agentic RAG for Automated Vulnerability Repair\"", "type": "paper", "meta": { "type": "paper", "title": "\"RAVEN: Agentic RAG for Automated Vulnerability Repair\"", "authors": "Varun Gadey, Zijie Liu, Alexandra Dmitrienko", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.22647", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-21", "updated_at": "2026-06-21", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "coding-agent", "computer-use", "memory", "rag" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CR", "cs.LG", "cs.SE" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "17", "collection_queries": "rag-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-22673-agentlens-interpretable-safety-steering-via-mechanistic-subspaces-for-multi-turn.md", "title": "\"AgentLens: Interpretable Safety Steering via Mechanistic Subspaces for Multi-Turn Coding Agent\"", "type": "paper", "meta": { "type": "paper", "title": "\"AgentLens: Interpretable Safety Steering via Mechanistic Subspaces for Multi-Turn Coding Agent\"", "authors": "Weidi Luo, Qiming Zhang, Yihao Quan, Mingyu Jin, Jie Cai, Chaowei Xiao, Jingcheng Niu, Zhen Xiang", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.22673", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-21", "updated_at": "2026-06-21", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "coding-agent", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI", "cs.SE" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "18", "collection_queries": "agent-safety, coding-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-22678-rigorbench-benchmarking-engineering-process-discipline-in-autonomous-ai-coding-a.md", "title": "\"RigorBench: Benchmarking Engineering Process Discipline in Autonomous AI Coding Agents\"", "type": "paper", "meta": { "type": "paper", "title": "\"RigorBench: Benchmarking Engineering Process Discipline in Autonomous AI Coding Agents\"", "authors": "Meher Bhaskar Madiraju, Meher Sai Preetam Madiraju", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.22678", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-21", "updated_at": "2026-06-29", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "coding-agent", "planning", "rag", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.SE", "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "19", "collection_queries": "coding-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-22737-groundeval-a-deterministic-replacement-for-llm-as-judge-in-stateful-agent-evalua.md", "title": "\"GroundEval: A Deterministic Replacement for LLM-as-Judge in Stateful Agent Evaluation\"", "type": "paper", "meta": { "type": "paper", "title": "\"GroundEval: A Deterministic Replacement for LLM-as-Judge in Stateful Agent Evaluation\"", "authors": "Jeffrey Flynt", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.22737", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-22", "updated_at": "2026-07-02", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "memory", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI", "cs.CL", "cs.SE" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "14", "collection_queries": "agent-evaluation" } }, { "collection": "papers", "path": "papers/items/2026-2606-22741-grade-graph-representation-of-llm-agent-dependency-and-execution.md", "title": "\"GRADE: Graph Representation of LLM Agent Dependency and Execution\"", "type": "paper", "meta": { "type": "paper", "title": "\"GRADE: Graph Representation of LLM Agent Dependency and Execution\"", "authors": "Yue Zhao", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.22741", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-22", "updated_at": "2026-06-22", "status": "queued", "relevance": "high", "topics": [ "coding-agent", "multi-agent", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.LG" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "16", "collection_queries": "multi-agent-llm, tool-use" } }, { "collection": "papers", "path": "papers/items/2026-2606-22844-ramem-contextual-reinstatement-for-long-term-agentic-memory.md", "title": "\"RaMem: Contextual Reinstatement for Long-term Agentic Memory\"", "type": "paper", "meta": { "type": "paper", "title": "\"RaMem: Contextual Reinstatement for Long-term Agentic Memory\"", "authors": "Wei Yang, Bryce Kan, Shixuan Li, Li Li, Yuehan Qin, Jiate Li, Paul Bogdan, Jesse Thomason", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.22844", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-22", "updated_at": "2026-06-22", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "memory", "rag", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI", "cs.MA" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "19", "collection_queries": "agent-memory" } }, { "collection": "papers", "path": "papers/items/2026-2606-22864-when-auc-0-998-is-not-enough-a-candidate-evaluation-protocol-for-hidden-state-pr.md", "title": "\"When AUC 0.998 Is Not Enough: A Candidate Evaluation Protocol for Hidden-State Probes of Indirect Prompt Injection in Multimodal Computer-Use Agents\"", "type": "paper", "meta": { "type": "paper", "title": "\"When AUC 0.998 Is Not Enough: A Candidate Evaluation Protocol for Hidden-State Probes of Indirect Prompt Injection in Multimodal Computer-Use Agents\"", "authors": "Yanhang Li, Zhichao Fan, Zexin Zhuang", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.22864", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-22", "updated_at": "2026-06-22", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.LG" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "14", "collection_queries": "web-gui-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-22948-envs-environment-native-verified-search-for-long-horizon-gui-agents.md", "title": "\"ENVS: Environment-Native Verified Search for Long-Horizon GUI Agents\"", "type": "paper", "meta": { "type": "paper", "title": "\"ENVS: Environment-Native Verified Search for Long-Horizon GUI Agents\"", "authors": "Yincheng Zhou, Athena Zhuoming Zhong, Shijie Zhang, Kevin Zhang, Teresa Xiaotao Shang, Shanghang Zhang", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.22948", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-22", "updated_at": "2026-06-22", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "computer-use", "planning", "reasoning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI", "cs.CV" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "15", "collection_queries": "web-gui-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-22953-plans-don-t-persist-why-context-management-is-load-bearing-for-llm-agents.md", "title": "\"Plans Don't Persist: Why Context Management Is Load Bearing for LLM Agents\"", "type": "paper", "meta": { "type": "paper", "title": "\"Plans Don't Persist: Why Context Management Is Load Bearing for LLM Agents\"", "authors": "Aman Mehta, Anupam Datta", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.22953", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-22", "updated_at": "2026-06-22", "status": "queued", "relevance": "high", "topics": [ "planning", "reasoning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI", "cs.CL" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "13", "collection_queries": "planning-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-23032-ipo-finance-agent-benchmark-of-llm-financial-analysts-beyond-finance-agent-v2-wi.md", "title": "\"IPO Finance Agent: Benchmark of LLM Financial Analysts Beyond Finance Agent v2, with Automated Rubric Generation, on the SpaceX (SPCX) IPO\"", "type": "paper", "meta": { "type": "paper", "title": "\"IPO Finance Agent: Benchmark of LLM Financial Analysts Beyond Finance Agent v2, with Automated Rubric Generation, on the SpaceX (SPCX) IPO\"", "authors": "Mostapha Benhenda", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.23032", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-22", "updated_at": "2026-06-30", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "rag", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI", "q-fin.GN" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "16", "collection_queries": "agent-evaluation" } }, { "collection": "papers", "path": "papers/items/2026-2606-23130-understanding-the-in-security-of-vibe-coded-applications.md", "title": "Understanding the (In)Security of Vibe-Coded Applications", "type": "paper", "meta": { "type": "paper", "title": "Understanding the (In)Security of Vibe-Coded Applications", "authors": "Junquan Deng, Zhiyu Fan, Ruijie Meng", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.23130", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-22", "updated_at": "2026-06-23", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "coding-agent", "memory", "tool-use", "workflow-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CR", "cs.SE" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "13", "collection_queries": "ai-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-23195-memory-contagion-cross-temporal-propagation-of-evaluator-bias-via-agent-memory.md", "title": "\"Memory Contagion: Cross-Temporal Propagation of Evaluator Bias via Agent Memory\"", "type": "paper", "meta": { "type": "paper", "title": "\"Memory Contagion: Cross-Temporal Propagation of Evaluator Bias via Agent Memory\"", "authors": "Zewen Liu", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.23195", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-22", "updated_at": "2026-06-24", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "computer-use", "memory", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.LG", "cs.AI", "cs.CL" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "14", "collection_queries": "agent-memory" } }, { "collection": "papers", "path": "papers/items/2026-2606-23283-towards-root-memories-benchmarking-and-enhancing-implicit-logical-memory-retriev.md", "title": "\"Towards Root Memories: Benchmarking and Enhancing Implicit Logical Memory Retrieval for Personalized LLMs\"", "type": "paper", "meta": { "type": "paper", "title": "\"Towards Root Memories: Benchmarking and Enhancing Implicit Logical Memory Retrieval for Personalized LLMs\"", "authors": "Hongxun Ding, Xiang Yu, Chengbing Wang, Jianfei Xiao, Keqin Bao, Wenjie Wang, Xiangnan He", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.23283", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-22", "updated_at": "2026-06-22", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "memory", "rag" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CL" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "14", "collection_queries": "agent-memory" } }, { "collection": "papers", "path": "papers/items/2026-2606-23327-videoagent-all-in-one-framework-for-video-understanding-and-editing.md", "title": "\"VideoAgent: All-in-One Framework for Video Understanding and Editing\"", "type": "paper", "meta": { "type": "paper", "title": "\"VideoAgent: All-in-One Framework for Video Understanding and Editing\"", "authors": "Hengji Zhou, Lingxuan Huang, Jian Wang, Bing Zhou, Si Wu, Lianghao Xia, Chao Huang", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.23327", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-22", "updated_at": "2026-07-03", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "multi-agent", "planning", "rag", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CV", "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "16", "collection_queries": "multi-agent-llm" } }, { "collection": "papers", "path": "papers/items/2026-2606-23343-group-selection-promotes-prosocial-prompts-in-populations-of-llm-agents.md", "title": "Group Selection Promotes Prosocial Prompts in Populations of LLM Agents", "type": "paper", "meta": { "type": "paper", "title": "Group Selection Promotes Prosocial Prompts in Populations of LLM Agents", "authors": "Luis Celiktemel, Edward Eichhorn, Levin Brinkmann, Robin Schimmelpfennig, Aron Vallinder, Yaomin Jiang, Edward Hughes, Iyad Rahwan", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.23343", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-22", "updated_at": "2026-06-22", "status": "queued", "relevance": "high", "topics": [ "coding-agent", "computer-use", "multi-agent", "world-model" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CY" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "15", "collection_queries": "multi-agent-llm" } }, { "collection": "papers", "path": "papers/items/2026-2606-23565-holoagent-0-a-unified-embodied-agent-framework-with-3d-spatial-memory.md", "title": "\"HoloAgent-0: A Unified Embodied Agent Framework with 3D Spatial Memory\"", "type": "paper", "meta": { "type": "paper", "title": "\"HoloAgent-0: A Unified Embodied Agent Framework with 3D Spatial Memory\"", "authors": "Xiaolin Zhou, Liu Liu, Tingyang Xiao, Wei Feng, Fa Fu, Xinrui Meng, Xinjie Wang, Jialiang Han, et al.", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.23565", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-22", "updated_at": "2026-06-22", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "computer-use", "embodied-agent", "memory", "planning", "rag", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.RO", "cs.CV" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "19", "collection_queries": "planning-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-23664-mas-promptbench-when-does-prompt-optimization-improve-multi-agent-llm-systems.md", "title": "\"MAS-PromptBench: When Does Prompt Optimization Improve Multi-Agent LLM Systems?\"", "type": "paper", "meta": { "type": "paper", "title": "\"MAS-PromptBench: When Does Prompt Optimization Improve Multi-Agent LLM Systems?\"", "authors": "Juyang Bai, Laixi Shi", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.23664", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-22", "updated_at": "2026-06-22", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "multi-agent", "workflow-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.LG", "cs.MA" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "16", "collection_queries": "agentic-ai, multi-agent-llm" } }, { "collection": "papers", "path": "papers/items/2026-2606-23752-esaa-conversational-an-event-sourced-memory-layer-for-continuity-handoff-and-cur.md", "title": "\"ESAA-Conversational: An Event-Sourced Memory Layer for Continuity, Handoff, and Curation Across Heterogeneous LLM Coding Agents\"", "type": "paper", "meta": { "type": "paper", "title": "\"ESAA-Conversational: An Event-Sourced Memory Layer for Continuity, Handoff, and Curation Across Heterogeneous LLM Coding Agents\"", "authors": "Elzo Brito dos Santos Filho", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.23752", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-22", "updated_at": "2026-06-22", "status": "queued", "relevance": "high", "topics": [ "coding-agent", "memory", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.SE" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "16", "collection_queries": "coding-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-23764-emergent-relational-order-in-llm-agent-societies-from-collective-affect-to-autho.md", "title": "\"Emergent Relational Order in LLM Agent Societies: From Collective Affect to Authority Stratification\"", "type": "paper", "meta": { "type": "paper", "title": "\"Emergent Relational Order in LLM Agent Societies: From Collective Affect to Authority Stratification\"", "authors": "Zhiyuan Ji, Xinyu Chen, Ziqi Dai, Shiyun Tang, Chunyu Wei, Yueguo Chen", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.23764", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-22", "updated_at": "2026-06-22", "status": "queued", "relevance": "high", "topics": [ "multi-agent", "planning", "tool-use", "world-model" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.MA", "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "15", "collection_queries": "multi-agent-llm" } }, { "collection": "papers", "path": "papers/items/2026-2606-23927-rift-bench-dynamic-red-teaming-for-agentic-ai-systems.md", "title": "\"RIFT-Bench: Dynamic Red-teaming For Agentic AI Systems\"", "type": "paper", "meta": { "type": "paper", "title": "\"RIFT-Bench: Dynamic Red-teaming For Agentic AI Systems\"", "authors": "Yarin Yerushalmi Levi, Roy Betser, Amit Giloni, Lidor Erez, Itay Gershon, Oren Rachmil, Sindhu Padakandla, Roman Vainshtein", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.23927", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-22", "updated_at": "2026-06-22", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "rag", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "15", "collection_queries": "agentic-ai" } }, { "collection": "papers", "path": "papers/items/2026-2606-23991-critique-of-agent-model.md", "title": "Critique of Agent Model", "type": "paper", "meta": { "type": "paper", "title": "Critique of Agent Model", "authors": "Eric Xing, Mingkai Deng, Jinyu Hou", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.23991", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-22", "updated_at": "2026-06-22", "status": "queued", "relevance": "high", "topics": [ "agent-safety", "coding-agent", "rag", "reasoning", "tool-use", "workflow-agent", "world-model" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI", "cs.LG", "cs.MA", "cs.RO" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "19", "collection_queries": "agentic-ai, ai-agent, coding-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-24193-skychain-intelligence-a-blockchain-secured-multi-agent-drl-framework-for-low-alt.md", "title": "\"SkyChain Intelligence: A Blockchain-Secured Multi-Agent DRL Framework for Low-Altitude Embodied Artificial Intelligence\"", "type": "paper", "meta": { "type": "paper", "title": "\"SkyChain Intelligence: A Blockchain-Secured Multi-Agent DRL Framework for Low-Altitude Embodied Artificial Intelligence\"", "authors": "Haoxiang Luo, Tianqi Jiang, Ruichen Zhang, Yinqiu Liu, Gang Sun, Hongfang Yu, Abbas Jamalipour, Dong In Kim", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.24193", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-23", "updated_at": "2026-06-23", "status": "queued", "relevance": "high", "topics": [ "agent-safety", "embodied-agent", "multi-agent", "tool-use", "world-model" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.NI", "cs.DC" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "15", "collection_queries": "agentic-ai" } }, { "collection": "papers", "path": "papers/items/2026-2606-24235-sp-mind-an-autonomous-reasoning-agent-for-spatial-proteomics-analysis.md", "title": "\"SP-Mind: An Autonomous Reasoning Agent for Spatial Proteomics Analysis\"", "type": "paper", "meta": { "type": "paper", "title": "\"SP-Mind: An Autonomous Reasoning Agent for Spatial Proteomics Analysis\"", "authors": "Yucheng Yuan, Yuanfeng Ji, Zhongxiao Li, Ruijiang Li", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.24235", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-23", "updated_at": "2026-06-23", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "computer-use", "rag", "reasoning", "tool-use", "workflow-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "14", "collection_queries": "ai-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-24322-securing-llm-agent-long-term-memory-against-poisoning-non-malleable-origin-bound.md", "title": "\"Securing LLM-Agent Long-Term Memory Against Poisoning: Non-Malleable, Origin-Bound Authority with Machine-Checked Guarantees\"", "type": "paper", "meta": { "type": "paper", "title": "\"Securing LLM-Agent Long-Term Memory Against Poisoning: Non-Malleable, Origin-Bound Authority with Machine-Checked Guarantees\"", "authors": "Yedidel Louck", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.24322", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-23", "updated_at": "2026-06-23", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "memory", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CR" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "15", "collection_queries": "agent-memory" } }, { "collection": "papers", "path": "papers/items/2026-2606-24402-poisoned-playbooks-demystifying-knowledge-poisoning-effects-on-ai-security-agent.md", "title": "\"Poisoned Playbooks: Demystifying Knowledge Poisoning Effects on AI Security Agents\"", "type": "paper", "meta": { "type": "paper", "title": "\"Poisoned Playbooks: Demystifying Knowledge Poisoning Effects on AI Security Agents\"", "authors": "Juho Park, Hyunmin Choi, Kevin Nam", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.24402", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-23", "updated_at": "2026-06-23", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "rag", "reasoning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CR" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "14", "collection_queries": "ai-agent, rag-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-24416-agentic-ai-for-bilevel-long-term-optimization-of-policy-driven-physical-layer-sy.md", "title": "Agentic AI for Bilevel Long-Term Optimization of Policy-Driven Physical Layer Systems", "type": "paper", "meta": { "type": "paper", "title": "Agentic AI for Bilevel Long-Term Optimization of Policy-Driven Physical Layer Systems", "authors": "Bingnan Xiao, Chenhao Yang, Wei Ni, Xin Wang, Tony Q. S. Quek", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.24416", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-23", "updated_at": "2026-06-23", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "memory", "multi-agent", "rag" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "13", "collection_queries": "agentic-ai" } }, { "collection": "papers", "path": "papers/items/2026-2606-24437-rem-moa-reasoning-memory-sustains-mixture-of-agents-scaling.md", "title": "\"ReM-MoA: Reasoning Memory Sustains Mixture-of-Agents Scaling\"", "type": "paper", "meta": { "type": "paper", "title": "\"ReM-MoA: Reasoning Memory Sustains Mixture-of-Agents Scaling\"", "authors": "Heng Ping, Arijit Bhattacharjee, Peiyu Zhang, Shixuan Li, Wei Yang, Ali Jannesari, Nesreen Ahmed, Paul Bogdan", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.24437", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-23", "updated_at": "2026-06-23", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "memory", "multi-agent", "reasoning" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "17", "collection_queries": "multi-agent-llm" } }, { "collection": "papers", "path": "papers/items/2026-2606-24453-bayesian-control-for-coding-agents.md", "title": "Bayesian control for coding agents", "type": "paper", "meta": { "type": "paper", "title": "Bayesian control for coding agents", "authors": "Theodore Papamarkou, Vladislav Smirnov, Viktor Mazanov, Artem Vazhentsev, Preslav Nakov, Timothy Baldwin, Artem Shelmanov", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.24453", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-23", "updated_at": "2026-06-23", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "coding-agent", "reasoning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI", "cs.CL" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "16", "collection_queries": "coding-agent, tool-use" } }, { "collection": "papers", "path": "papers/items/2026-2606-24515-reinforcement-learning-for-computer-use-agents-with-autonomous-evaluation.md", "title": "Reinforcement Learning for Computer-Use Agents with Autonomous Evaluation", "type": "paper", "meta": { "type": "paper", "title": "Reinforcement Learning for Computer-Use Agents with Autonomous Evaluation", "authors": "Marta Sumyk, Oleksandr Kosovan", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.24515", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-23", "updated_at": "2026-06-23", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "computer-use", "rag" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI", "cs.HC" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "14", "collection_queries": "web-gui-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-24525-viscritic-visual-state-comparison-as-process-reward-for-gui-agents.md", "title": "\"VisCritic: Visual State Comparison as Process Reward for GUI Agents\"", "type": "paper", "meta": { "type": "paper", "title": "\"VisCritic: Visual State Comparison as Process Reward for GUI Agents\"", "authors": "Jiachen Qian", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.24525", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-23", "updated_at": "2026-06-23", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "computer-use", "planning", "reasoning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CV" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "16", "collection_queries": "web-gui-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-24535-governed-shared-memory-for-multi-agent-llm-systems.md", "title": "Governed Shared Memory for Multi-Agent LLM Systems", "type": "paper", "meta": { "type": "paper", "title": "Governed Shared Memory for Multi-Agent LLM Systems", "authors": "Yanki Margalit, Nurit Cohen-Inger, Erni Avram, Ran Taig, Oded Margalit", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.24535", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-23", "updated_at": "2026-06-23", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "memory", "multi-agent", "rag", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "20", "collection_queries": "agent-memory, multi-agent-llm" } }, { "collection": "papers", "path": "papers/items/2026-2606-24551-gui-vs-cli-execution-bottlenecks-in-screen-only-and-skill-mediated-computer-use-.md", "title": "\"GUI vs. CLI: Execution Bottlenecks in Screen-Only and Skill-Mediated Computer-Use Agents\"", "type": "paper", "meta": { "type": "paper", "title": "\"GUI vs. CLI: Execution Bottlenecks in Screen-Only and Skill-Mediated Computer-Use Agents\"", "authors": "Xiao Zhou, Siyue Zhang, Yilun Zhao, Jinbiao Wei, Tingyu Song, Arman Cohan, Chen Zhao", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.24551", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-22", "updated_at": "2026-06-22", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "computer-use", "planning", "rag", "reasoning", "tool-use", "workflow-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "15", "collection_queries": "web-gui-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-24595-memprobe-probing-long-term-agent-memory-via-hidden-user-state-recovery.md", "title": "\"MEMPROBE: Probing Long-Term Agent Memory via Hidden User-State Recovery\"", "type": "paper", "meta": { "type": "paper", "title": "\"MEMPROBE: Probing Long-Term Agent Memory via Hidden User-State Recovery\"", "authors": "Enze Ma, Yufan Zhou, Wei-Chieh Huang, Jie Yang, Huanhuan Ma, Zixuan Wang, Chengze Li, Chunyu Miao, et al.", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.24595", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-23", "updated_at": "2026-06-23", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "memory", "rag", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CL" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "16", "collection_queries": "agent-memory" } }, { "collection": "papers", "path": "papers/items/2026-2606-24597-qwen-agentworld-language-world-models-for-general-agents.md", "title": "\"Qwen-AgentWorld: Language World Models for General Agents\"", "type": "paper", "meta": { "type": "paper", "title": "\"Qwen-AgentWorld: Language World Models for General Agents\"", "authors": "Yuxin Zuo, Zikai Xiao, Li Sheng, Fei Huang, Jianhong Tu, Yuxuan Liu, Tianyi Tang, Xiaomeng Hu, et al.", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.24597", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-23", "updated_at": "2026-06-23", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "planning", "rag", "reasoning", "tool-use", "world-model" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CL" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "15", "collection_queries": "agent-evaluation" } }, { "collection": "papers", "path": "papers/items/2026-2606-24623-privacy-preserving-rag-via-multi-agent-semantic-rewriting-achieving-confidential.md", "title": "\"Privacy-Preserving RAG via Multi-Agent Semantic Rewriting: Achieving Confidentiality Without Compromising Contextual Fidelity\"", "type": "paper", "meta": { "type": "paper", "title": "\"Privacy-Preserving RAG via Multi-Agent Semantic Rewriting: Achieving Confidentiality Without Compromising Contextual Fidelity\"", "authors": "Yuanhe Zhao, Tianyu Zhang, Huafei Xing, Derek F. Wong, Jianbin Li, Tao Fang", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.24623", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-23", "updated_at": "2026-06-23", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "multi-agent", "rag", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CL", "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "16", "collection_queries": "multi-agent-llm, rag-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-24626-safari-scaling-long-horizon-agentic-fault-attribution-via-active-investigation.md", "title": "\"SAFARI: Scaling Long Horizon Agentic Fault Attribution via Active Investigation\"", "type": "paper", "meta": { "type": "paper", "title": "\"SAFARI: Scaling Long Horizon Agentic Fault Attribution via Active Investigation\"", "authors": "Chenyang Zhu, Jiayu Yao, Kushal Chawla, Youbing Yin, Nathan Wolfe, Pengshan Cai, Jingyu Wu, Spencer Hong, et al.", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.24626", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-23", "updated_at": "2026-06-23", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "memory", "multi-agent", "planning", "reasoning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "19", "collection_queries": "autonomous-agent-llm, multi-agent-llm" } }, { "collection": "papers", "path": "papers/items/2026-2606-24649-agentic-collaborative-cognition-for-zero-shot-3d-understanding.md", "title": "Agentic Collaborative Cognition for Zero-Shot 3D Understanding", "type": "paper", "meta": { "type": "paper", "title": "Agentic Collaborative Cognition for Zero-Shot 3D Understanding", "authors": "Wenxin Wang, Bo Zhang, Feng Chen, Zixuan Wang, Wen Li, Changsheng Li, Yinjie Lei", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.24649", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-23", "updated_at": "2026-06-25", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "computer-use", "multi-agent", "planning", "rag" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CV" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "17", "collection_queries": "multi-agent-llm" } }, { "collection": "papers", "path": "papers/items/2026-2606-24689-automated-summarization-of-software-documents-an-llm-based-multi-agent-approach.md", "title": "\"Automated Summarization of Software Documents: An LLM-based Multi-Agent Approach\"", "type": "paper", "meta": { "type": "paper", "title": "\"Automated Summarization of Software Documents: An LLM-based Multi-Agent Approach\"", "authors": "Duc S. H. Nguyen, Minh T. Nguyen, Phuong T. Nguyen, Juri Di Rocco, Davide Di Ruscio", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.24689", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-23", "updated_at": "2026-06-23", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "coding-agent", "multi-agent", "workflow-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.SE" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "16", "collection_queries": "multi-agent-llm" } }, { "collection": "papers", "path": "papers/items/2026-2606-24694-supplynet-supporting-visual-exploratory-learning-in-supply-chain-via-contextual-.md", "title": "\"SupplyNet: Supporting Visual Exploratory Learning in Supply Chain via Contextual Multi-Agent Simulation\"", "type": "paper", "meta": { "type": "paper", "title": "\"SupplyNet: Supporting Visual Exploratory Learning in Supply Chain via Contextual Multi-Agent Simulation\"", "authors": "Yanjia Li, Kelcy Kexin Han, Tianrui Hu, Yi-Fan Cao, Huamin Qu, Sicheng Song", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.24694", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-23", "updated_at": "2026-06-23", "status": "queued", "relevance": "high", "topics": [ "multi-agent", "rag", "reasoning", "world-model" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.HC" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "14", "collection_queries": "multi-agent-llm" } }, { "collection": "papers", "path": "papers/items/2026-2606-24775-are-we-ready-for-an-agent-native-memory-system.md", "title": "Are We Ready For An Agent-Native Memory System?", "type": "paper", "meta": { "type": "paper", "title": "Are We Ready For An Agent-Native Memory System?", "authors": "Wei Zhou, Xuanhe Zhou, Shaokun Han, Hongming Xu, Guoliang Li, Zhiyu Li, Feiyu Xiong, Fan Wu", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.24775", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-23", "updated_at": "2026-06-23", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "memory", "planning", "rag", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CL", "cs.DB", "cs.IR" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "17", "collection_queries": "agent-memory" } }, { "collection": "papers", "path": "papers/items/2026-2606-24779-deepbd-a-grounded-agentic-workflow-for-variant-prioritization-and-diagnosis-of-g.md", "title": "\"DeepBD: A Grounded Agentic Workflow for Variant Prioritization and Diagnosis of Genetic Birth Defects\"", "type": "paper", "meta": { "type": "paper", "title": "\"DeepBD: A Grounded Agentic Workflow for Variant Prioritization and Diagnosis of Genetic Birth Defects\"", "authors": "Shiyu Li, Ziqi Yan, Zhihao Wu, Jielong Lu, Weiran Liao, Jiajun Yu, Genjie Li, Zeyu Chu, et al.", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.24779", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-23", "updated_at": "2026-06-23", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "memory", "tool-use", "workflow-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "q-bio.GN", "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "15", "collection_queries": "agentic-ai" } }, { "collection": "papers", "path": "papers/items/2026-2606-24820-sherloc-structured-diagnostic-localization-for-code-repair-agents.md", "title": "\"SHERLOC: Structured Diagnostic Localization for Code Repair Agents\"", "type": "paper", "meta": { "type": "paper", "title": "\"SHERLOC: Structured Diagnostic Localization for Code Repair Agents\"", "authors": "Hovhannes Tamoyan, Sean Narenthiran, Erik Arakelyan, Mira Mezini, Boris Ginsburg", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.24820", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-23", "updated_at": "2026-06-23", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "coding-agent", "multi-agent", "rag", "reasoning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CL" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "20", "collection_queries": "coding-agent, multi-agent-llm, tool-use" } }, { "collection": "papers", "path": "papers/items/2026-2606-24839-grading-the-grader-lessons-from-evaluating-an-agentic-data-analysis-system.md", "title": "\"Grading the Grader: Lessons from Evaluating an Agentic Data Analysis System\"", "type": "paper", "meta": { "type": "paper", "title": "\"Grading the Grader: Lessons from Evaluating an Agentic Data Analysis System\"", "authors": "Tian Zheng, Kai-Tai Hsu", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.24839", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-23", "updated_at": "2026-06-23", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "computer-use", "multi-agent", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI", "stat.AP" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "15", "collection_queries": "multi-agent-llm" } }, { "collection": "papers", "path": "papers/items/2026-2606-24855-openthoughts-agent-data-recipes-for-agentic-models.md", "title": "\"OpenThoughts-Agent: Data Recipes for Agentic Models\"", "type": "paper", "meta": { "type": "paper", "title": "\"OpenThoughts-Agent: Data Recipes for Agentic Models\"", "authors": "Negin Raoof, Richard Zhuang, Marianna Nezhurina, Etash Guha, Atula Tejaswi, Ryan Marten, Charlie F. Ruan, Tyler Griggs, et al.", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.24855", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-23", "updated_at": "2026-06-23", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "rag" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "13", "collection_queries": "agent-evaluation" } }, { "collection": "papers", "path": "papers/items/2026-2606-24937-the-hitchhiker-s-guide-to-agentic-ai-from-foundations-to-systems.md", "title": "\"The Hitchhiker's Guide to Agentic AI: From Foundations to Systems\"", "type": "paper", "meta": { "type": "paper", "title": "\"The Hitchhiker's Guide to Agentic AI: From Foundations to Systems\"", "authors": "Haggai Roitman", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.24937", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-22", "updated_at": "2026-06-22", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "computer-use", "memory", "multi-agent", "rag", "reasoning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI", "cs.CL", "cs.IR", "cs.LG" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "25", "collection_queries": "agentic-ai, multi-agent-llm, rag-agent, tool-use" } }, { "collection": "papers", "path": "papers/items/2026-2606-24976-diagnosing-and-mitigating-compounding-failures-in-agentic-persuasion-via-taxonom.md", "title": "Diagnosing and Mitigating Compounding Failures in Agentic Persuasion via Taxonomic Strategy Retrieval", "type": "paper", "meta": { "type": "paper", "title": "Diagnosing and Mitigating Compounding Failures in Agentic Persuasion via Taxonomic Strategy Retrieval", "authors": "Sana Ayromlou, Purvi Sehgal, Pradyumna Narayana", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.24976", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-23", "updated_at": "2026-07-04", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "multi-agent", "planning", "rag" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI", "cs.CL", "cs.LG" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "14", "collection_queries": "rag-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-25115-forget-to-improve-on-device-llm-agent-continual-learning-via-budget-curated-memo.md", "title": "\"Forget to Improve: On-Device LLM-Agent Continual Learning via Budget-Curated Memory\"", "type": "paper", "meta": { "type": "paper", "title": "\"Forget to Improve: On-Device LLM-Agent Continual Learning via Budget-Curated Memory\"", "authors": "Beining Wu, Zihao Ding, Jun Huang, Yanxiao Zhao", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.25115", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-23", "updated_at": "2026-06-23", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "embodied-agent", "memory" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.LG", "cs.NI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "15", "collection_queries": "agent-memory" } }, { "collection": "papers", "path": "papers/items/2026-2606-25139-buildrix-an-open-platform-for-sharing-and-benchmarking-agentic-ai-skills-in-buil.md", "title": "\"Buildrix: An Open Platform for Sharing and Benchmarking Agentic AI Skills in Building Engineering\"", "type": "paper", "meta": { "type": "paper", "title": "\"Buildrix: An Open Platform for Sharing and Benchmarking Agentic AI Skills in Building Engineering\"", "authors": "Zixin Jiang, Bing Dong", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.25139", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-23", "updated_at": "2026-06-23", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "tool-use", "workflow-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "eess.SY" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "15", "collection_queries": "agentic-ai" } }, { "collection": "papers", "path": "papers/items/2026-2606-25161-trustmem-learning-trustworthy-memory-consolidation-for-llm-agents-with-long-term.md", "title": "\"TRUSTMEM: Learning Trustworthy Memory Consolidation for LLM Agents with Long-Term Memory\"", "type": "paper", "meta": { "type": "paper", "title": "\"TRUSTMEM: Learning Trustworthy Memory Consolidation for LLM Agents with Long-Term Memory\"", "authors": "Tianyu Yang, Sudipta Paul, Vijay Srinivasan, Vivek Kulkarni, Srinivas Chappidi", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.25161", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-23", "updated_at": "2026-06-23", "status": "skimmed", "relevance": "high", "topics": [ "agent-evaluation", "computer-use", "memory", "rag", "reasoning", "tool-use" ], "methods": [ "memory-transition-verification", "transition-ranked-grpo", "write-revise-prune" ], "benchmarks": [ "MemoryAgentBench", "HaluMem", "Mem-alpha" ], "models": [], "datasets": [ "cs.AI" ], "related_concepts": [ "memory-consolidation", "transition-verification" ], "related_jobs": [], "related_experiments": [ "KC-001-agent-memory-pilot" ], "related_projects": [ "learning/agent-memory" ], "collection_score": "19", "collection_queries": "agent-memory" } }, { "collection": "papers", "path": "papers/items/2026-2606-25189-actplane-programmable-os-level-policy-enforcement-for-agent-harnesses.md", "title": "\"ActPlane: Programmable OS-Level Policy Enforcement for Agent Harnesses\"", "type": "paper", "meta": { "type": "paper", "title": "\"ActPlane: Programmable OS-Level Policy Enforcement for Agent Harnesses\"", "authors": "Yusheng Zheng, Tianyuan Wu, Quanzhi Fu, Tong Yu, Wenan Mao, Tao Ma, Dan Williams, Wei Wang, et al.", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.25189", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-23", "updated_at": "2026-06-30", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "coding-agent", "planning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.OS" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "14", "collection_queries": "ai-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-25191-to-isolate-or-to-score-model-adaptive-assessment-for-cost-efficient-multi-agent-.md", "title": "To Isolate or to Score? Model-Adaptive Assessment for Cost-Efficient Multi-Agent RAG", "type": "paper", "meta": { "type": "paper", "title": "To Isolate or to Score? Model-Adaptive Assessment for Cost-Efficient Multi-Agent RAG", "authors": "Jungseob Lee, Chanjun Park, Heuiseok Lim", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.25191", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-23", "updated_at": "2026-06-23", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "multi-agent", "rag", "reasoning" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI", "cs.CL" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "13", "collection_queries": "rag-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-25195-sok-ai-secure-code-generation-progress-pitfalls-and-paths-forward.md", "title": "\"SoK: AI Secure Code Generation: Progress, Pitfalls, and Paths Forward\"", "type": "paper", "meta": { "type": "paper", "title": "\"SoK: AI Secure Code Generation: Progress, Pitfalls, and Paths Forward\"", "authors": "Rupam Patir, Keyan Guo, Haipeng Cai, Hongxin Hu", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.25195", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-23", "updated_at": "2026-06-23", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "coding-agent", "computer-use", "workflow-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CR", "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "13", "collection_queries": "agentic-ai, coding-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-25206-raven-long-horizon-reasoning-navigation-with-a-visuo-spatio-temporal-memory.md", "title": "\"RAVEN: Long-Horizon Reasoning & Navigation with a Visuo-Spatio-Temporal Memory\"", "type": "paper", "meta": { "type": "paper", "title": "\"RAVEN: Long-Horizon Reasoning & Navigation with a Visuo-Spatio-Temporal Memory\"", "authors": "Yixun Hu, Zhicheng Zheng, Lihan Zha, Chunwei Xing, Rajdeep Singh, Omar Hossain, Antonio Loquercio, Dhruv Shah", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.25206", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-23", "updated_at": "2026-06-23", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "embodied-agent", "memory", "planning", "rag", "reasoning" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.RO", "cs.AI", "cs.CL" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "14", "collection_queries": "agent-memory" } }, { "collection": "papers", "path": "papers/items/2026-2606-25334-bridging-the-post-discharge-gap-a-traceable-multi-agent-framework-for-safe-and-c.md", "title": "\"Bridging the Post-discharge Gap: A Traceable Multi-agent Framework for Safe and Continuous Care\"", "type": "paper", "meta": { "type": "paper", "title": "\"Bridging the Post-discharge Gap: A Traceable Multi-agent Framework for Safe and Continuous Care\"", "authors": "Runwei Guan, Yi Zhou, Heyi Lin, Jinjing Zhu, Mingyuan Hou, Yang Yang, Fang Yuan, Xiaohong Lin, et al.", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.25334", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-24", "updated_at": "2026-06-24", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "memory", "multi-agent", "rag" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.MA" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "18", "collection_queries": "multi-agent-llm" } }, { "collection": "papers", "path": "papers/items/2026-2606-25358-agentic-knowledge-tracing-a-multi-agent-llm-architecture-for-stealth-assessment-.md", "title": "\"Agentic Knowledge Tracing: A Multi-Agent LLM Architecture for Stealth Assessment of Financial Literacy in Serious Games\"", "type": "paper", "meta": { "type": "paper", "title": "\"Agentic Knowledge Tracing: A Multi-Agent LLM Architecture for Stealth Assessment of Financial Literacy in Serious Games\"", "authors": "Gabriel Santos, Rita Julia, Marcelo Nascimento", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.25358", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-24", "updated_at": "2026-06-24", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "multi-agent", "reasoning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI", "cs.MA" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "18", "collection_queries": "multi-agent-llm" } }, { "collection": "papers", "path": "papers/items/2026-2606-25361-memory-makes-the-difference-evaluating-how-different-memory-roles-shape-conversa.md", "title": "\"Memory Makes the Difference: Evaluating How Different Memory Roles Shape Conversational Agents\"", "type": "paper", "meta": { "type": "paper", "title": "\"Memory Makes the Difference: Evaluating How Different Memory Roles Shape Conversational Agents\"", "authors": "Yuxin Wang, Paul Thomas, Zhiwei Yu, Yuan Gao, Saeed Hassanpour, Soroush Vosoughi, Robert Sim, Nick Craswell", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.25361", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-24", "updated_at": "2026-06-24", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "memory", "rag", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CL", "cs.AI", "cs.IR" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "15", "collection_queries": "rag-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-25400-brainagent-a-large-language-model-driven-multi-agent-framework-for-autonomous-br.md", "title": "\"BrainAgent: A Large Language Model-Driven Multi-Agent Framework for Autonomous Brain Signal Understanding\"", "type": "paper", "meta": { "type": "paper", "title": "\"BrainAgent: A Large Language Model-Driven Multi-Agent Framework for Autonomous Brain Signal Understanding\"", "authors": "Yangxuan Zhou, Sha Zhao, Jiquan Wang, Shijian Li, Gang Pan", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.25400", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-24", "updated_at": "2026-06-24", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "multi-agent", "planning", "tool-use", "workflow-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "17", "collection_queries": "multi-agent-llm" } }, { "collection": "papers", "path": "papers/items/2026-2606-25484-from-causal-discovery-to-implementation-an-agentic-ai-framework-for-e-scooter-mo.md", "title": "\"From Causal Discovery to Implementation: An Agentic AI Framework for E-Scooter Mobility Hub Planning Across 29 German Cities\"", "type": "paper", "meta": { "type": "paper", "title": "\"From Causal Discovery to Implementation: An Agentic AI Framework for E-Scooter Mobility Hub Planning Across 29 German Cities\"", "authors": "Meng Jin, Melanie Handrich, Simone Martinenz, Nicholas Hoeser, Ziyue Li", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.25484", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-24", "updated_at": "2026-06-24", "status": "queued", "relevance": "high", "topics": [ "coding-agent", "computer-use", "planning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CY", "econ.GN", "stat.AP" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "14", "collection_queries": "agentic-ai" } }, { "collection": "papers", "path": "papers/items/2026-2606-25514-unlocking-model-potentials-through-adaptive-multi-agent-scaffolding-for-efficien.md", "title": "Unlocking Model Potentials Through Adaptive Multi-Agent Scaffolding for Efficient Issue Resolution", "type": "paper", "meta": { "type": "paper", "title": "Unlocking Model Potentials Through Adaptive Multi-Agent Scaffolding for Efficient Issue Resolution", "authors": "Yang Chen, Aliya Ahmad, Yiheng Zhou, Reyhaneh Jabbarvand", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.25514", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-24", "updated_at": "2026-06-24", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "coding-agent", "multi-agent", "planning", "rag", "tool-use", "workflow-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.SE" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "17", "collection_queries": "coding-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-25519-quantization-inflates-reasoning-token-inflation-as-a-hidden-cost-of-low-bit-reas.md", "title": "\"Quantization Inflates Reasoning: Token Inflation as a Hidden Cost of Low-Bit Reasoning Models\"", "type": "paper", "meta": { "type": "paper", "title": "\"Quantization Inflates Reasoning: Token Inflation as a Hidden Cost of Low-Bit Reasoning Models\"", "authors": "Xinyu Lian, Walid Krichene, Beichen Huang, Masahiro Tanaka, Olatunji Ruwase, Li Zhang, Minjia Zhang", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.25519", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-24", "updated_at": "2026-06-29", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "coding-agent", "rag", "reasoning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI", "cs.LG" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "13", "collection_queries": "tool-use" } }, { "collection": "papers", "path": "papers/items/2026-2606-25588-intenttester-intent-driven-multi-agent-framework-for-cross-library-test-migratio.md", "title": "\"IntentTester: Intent-Driven Multi-agent Framework for Cross-Library Test Migration\"", "type": "paper", "meta": { "type": "paper", "title": "\"IntentTester: Intent-Driven Multi-agent Framework for Cross-Library Test Migration\"", "authors": "Yi Gao, Ziyuan Zhang, Xing Hu, Xiaohu Yang, Xin Xia", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.25588", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-24", "updated_at": "2026-06-24", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "coding-agent", "computer-use", "multi-agent", "reasoning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.SE" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "17", "collection_queries": "multi-agent-llm" } }, { "collection": "papers", "path": "papers/items/2026-2606-25622-probabilistic-agents-in-deterministic-audits-evaluating-multi-agent-systems-for-.md", "title": "\"Probabilistic Agents in Deterministic Audits: Evaluating Multi-Agent Systems for Automated Audits Based on the German IT-Grundschutz\"", "type": "paper", "meta": { "type": "paper", "title": "\"Probabilistic Agents in Deterministic Audits: Evaluating Multi-Agent Systems for Automated Audits Based on the German IT-Grundschutz\"", "authors": "Lea Roxanne Muth, Marian Margraf", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.25622", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-24", "updated_at": "2026-06-24", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "multi-agent", "rag", "reasoning", "tool-use", "workflow-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CR", "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "18", "collection_queries": "multi-agent-llm" } }, { "collection": "papers", "path": "papers/items/2026-2606-25651-medguards-multi-agent-system-for-reliable-medical-error-detection-and-correction.md", "title": "\"MedGuards: Multi-Agent System for Reliable Medical Error Detection and Correction\"", "type": "paper", "meta": { "type": "paper", "title": "\"MedGuards: Multi-Agent System for Reliable Medical Error Detection and Correction\"", "authors": "Congbo Ma, Hu Wang, Yichun Zhang, Farah E. Shamout", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.25651", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-24", "updated_at": "2026-06-26", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "computer-use", "multi-agent", "reasoning" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CL" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "16", "collection_queries": "multi-agent-llm" } }, { "collection": "papers", "path": "papers/items/2026-2606-25656-is-graphrag-needed-from-basic-rag-to-graph-agentic-solutions-with-context-optimi.md", "title": "Is GraphRAG Needed? From Basic RAG to Graph-/Agentic Solutions with Context Optimization", "type": "paper", "meta": { "type": "paper", "title": "Is GraphRAG Needed? From Basic RAG to Graph-/Agentic Solutions with Context Optimization", "authors": "Long Chen, Ryan Razkenari, Yuxuan Zhou, Yuan Tian, Rahul Ghosh, Venkatesh Pappakrishnan, Disha Ahuja, Vidya Sagar Ravipati", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.25656", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-24", "updated_at": "2026-06-24", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "memory", "planning", "rag" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CL", "cs.AI", "cs.IR" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "14", "collection_queries": "rag-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-25705-gui-agent-guided-exploration-of-user-sensitive-screens.md", "title": "\"GUI agent: Guided Exploration of User-Sensitive Screens\"", "type": "paper", "meta": { "type": "paper", "title": "\"GUI agent: Guided Exploration of User-Sensitive Screens\"", "authors": "Aradhana Nayak, Mussadiq Nazeer, Wang Peng, Feng Liu", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.25705", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-24", "updated_at": "2026-06-24", "status": "queued", "relevance": "high", "topics": [ "agent-safety", "computer-use", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "14", "collection_queries": "web-gui-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-25760-uncertainty-quantification-for-computer-use-agents-a-benchmark-across-vision-lan.md", "title": "\"Uncertainty Quantification for Computer-Use Agents: A Benchmark across Vision-Language Models and GUI Grounding Datasets\"", "type": "paper", "meta": { "type": "paper", "title": "\"Uncertainty Quantification for Computer-Use Agents: A Benchmark across Vision-Language Models and GUI Grounding Datasets\"", "authors": "Divake Kumar, Sina Tayebati, Devashri Naik, Amanda Sofie Rios, Nilesh Ahuja, Omesh Tickoo, Ranganath Krishnan, Amit Ranjan Trivedi", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.25760", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-24", "updated_at": "2026-06-24", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "computer-use", "rag" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.LG", "cs.AI", "cs.CL", "cs.CV" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "16", "collection_queries": "agent-evaluation, web-gui-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-25819-beyond-function-calling-benchmarking-tool-using-agents-under-tool-environment-un.md", "title": "\"Beyond Function Calling: Benchmarking Tool-Using Agents under Tool-Environment Unreliability\"", "type": "paper", "meta": { "type": "paper", "title": "\"Beyond Function Calling: Benchmarking Tool-Using Agents under Tool-Environment Unreliability\"", "authors": "Yang Tian, Zhengpeng Shi, Yu Zhou, Bo Zhao", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.25819", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-24", "updated_at": "2026-06-27", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "tool-use", "workflow-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CL", "cs.SE" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "15", "collection_queries": "function-calling, tool-use" } }, { "collection": "papers", "path": "papers/items/2026-2606-25899-manipulation-is-task-dependent-a-multi-axis-multi-environment-evaluation-of-fron.md", "title": "\"Manipulation Is Task-Dependent: A Multi-Axis, Multi-Environment Evaluation of Frontier LLMs\"", "type": "paper", "meta": { "type": "paper", "title": "\"Manipulation Is Task-Dependent: A Multi-Axis, Multi-Environment Evaluation of Frontier LLMs\"", "authors": "Adeeb Zaman, Erik Nordby, Fred Heiding", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.25899", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-24", "updated_at": "2026-06-24", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "rag", "tool-use", "workflow-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.MA" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "14", "collection_queries": "agentic-ai" } }, { "collection": "papers", "path": "papers/items/2026-2606-26057-the-unfireable-safety-kernel-execution-time-ai-alignment-for-ai-agents-and-other.md", "title": "\"The Unfireable Safety Kernel: Execution-Time AI Alignment for AI Agents and Other Escapable AI Systems\"", "type": "paper", "meta": { "type": "paper", "title": "\"The Unfireable Safety Kernel: Execution-Time AI Alignment for AI Agents and Other Escapable AI Systems\"", "authors": "Seth Dobrin, Łukasz Chmiel", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.26057", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-24", "updated_at": "2026-06-24", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "planning", "tool-use", "world-model" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI", "cs.CR", "cs.LG" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "13", "collection_queries": "ai-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-26203-agentic-analysis-for-agentic-infrastructure-an-llm-powered-pipeline-for-comparat.md", "title": "\"Agentic Analysis for Agentic Infrastructure: An LLM-Powered Pipeline for Comparative Governance of DAO and Corporate AI Protocols\"", "type": "paper", "meta": { "type": "paper", "title": "\"Agentic Analysis for Agentic Infrastructure: An LLM-Powered Pipeline for Comparative Governance of DAO and Corporate AI Protocols\"", "authors": "Yutian Wang, Luyao Zhang", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.26203", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-24", "updated_at": "2026-06-24", "status": "queued", "relevance": "high", "topics": [ "agent-safety", "coding-agent", "rag", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI", "cs.CY", "cs.MA", "cs.SI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "16", "collection_queries": "agentic-ai, ai-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-26205-knowledge-augmented-agentic-ai-for-mental-health-medication-information-seeking.md", "title": "Knowledge-augmented Agentic AI for Mental Health Medication Information Seeking", "type": "paper", "meta": { "type": "paper", "title": "Knowledge-augmented Agentic AI for Mental Health Medication Information Seeking", "authors": "Huizi Yu, Jian Liu, Wenkong Wang, Lingyao Li, Jiayan Zhou, Zhaoqian Xue, Xiang Li, Xinxin Lin, et al.", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.26205", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-24", "updated_at": "2026-06-24", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "multi-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "14", "collection_queries": "agentic-ai, multi-agent-llm" } }, { "collection": "papers", "path": "papers/items/2026-2606-26216-cyberchainbench-can-ai-agents-secure-smart-contracts-against-real-world-on-chain.md", "title": "\"CyberChainBench: Can AI Agents Secure Smart Contracts Against Real-World On-Chain Vulnerabilities?\"", "type": "paper", "meta": { "type": "paper", "title": "\"CyberChainBench: Can AI Agents Secure Smart Contracts Against Real-World On-Chain Vulnerabilities?\"", "authors": "Jintao Huang, Fengqing Jiang, Radha Poovendran, Zhiqiang Lin", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.26216", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-24", "updated_at": "2026-06-24", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CR", "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "16", "collection_queries": "agent-safety, ai-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-26300-the-verification-horizon-no-silver-bullet-for-coding-agent-rewards.md", "title": "\"The Verification Horizon: No Silver Bullet for Coding Agent Rewards\"", "type": "paper", "meta": { "type": "paper", "title": "\"The Verification Horizon: No Silver Bullet for Coding Agent Rewards\"", "authors": "Binghai Wang, Chenlong Zhang, Dayiheng Liu, Jiajun Zhang, Jiawei Chen, Mingze Li, Mouxiang Chen, Rongyao Fang, et al.", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.26300", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-24", "updated_at": "2026-06-29", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "coding-agent", "planning", "rag", "reasoning" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI", "cs.CL" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "14", "collection_queries": "coding-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-26346-how-do-tool-augmented-llm-agents-perform-on-real-world-energy-analytics-tasks.md", "title": "How Do Tool-Augmented LLM Agents Perform on Real-World Energy Analytics Tasks?", "type": "paper", "meta": { "type": "paper", "title": "How Do Tool-Augmented LLM Agents Perform on Real-World Energy Analytics Tasks?", "authors": "David Akinpelu, Akintonde Abbas, Rereloluwa Alimi, Ayodeji Lana", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.26346", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-24", "updated_at": "2026-06-24", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "coding-agent", "rag", "reasoning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "19", "collection_queries": "agent-evaluation" } }, { "collection": "papers", "path": "papers/items/2026-2606-26356-instruction-bleed-cross-module-interference-in-prompt-composed-agentic-systems.md", "title": "\"Instruction Bleed: Cross-Module Interference in Prompt-Composed Agentic Systems\"", "type": "paper", "meta": { "type": "paper", "title": "\"Instruction Bleed: Cross-Module Interference in Prompt-Composed Agentic Systems\"", "authors": "Ching-Yu Lin, Yifan Liu", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.26356", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-24", "updated_at": "2026-06-24", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "multi-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI", "cs.IR", "cs.MA" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "13", "collection_queries": "agent-evaluation, agentic-ai" } }, { "collection": "papers", "path": "papers/items/2026-2606-26403-profilefoundry-a-synthetic-person-object-substrate-for-privacy-memory-and-tool-u.md", "title": "\"ProfileFoundry: A Synthetic Person-Object Substrate for Privacy, Memory, and Tool-Use Evaluation in LLM Agent\"", "type": "paper", "meta": { "type": "paper", "title": "\"ProfileFoundry: A Synthetic Person-Object Substrate for Privacy, Memory, and Tool-Use Evaluation in LLM Agent\"", "authors": "Sriram Selvam, Anneswa Ghosh", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.26403", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-24", "updated_at": "2026-06-24", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "memory", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CL" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "19", "collection_queries": "tool-use" } }, { "collection": "papers", "path": "papers/items/2026-2606-26453-optimizing-cuda-like-a-human-micro-profiling-tools-as-expert-surrogates-for-llm-.md", "title": "\"Optimizing CUDA like a Human: Micro-Profiling Tools as Expert Surrogates for LLM-Based GPU Kernel Optimization\"", "type": "paper", "meta": { "type": "paper", "title": "\"Optimizing CUDA like a Human: Micro-Profiling Tools as Expert Surrogates for LLM-Based GPU Kernel Optimization\"", "authors": "Jiading Gai, Shuai Zhang, Kaj Bostrom, Jin Huang, Vihang Patil, Haoyang Fang, Bernie Wang, Huzefa Rangwala, et al.", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.26453", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-24", "updated_at": "2026-06-24", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "coding-agent", "computer-use", "memory", "multi-agent", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.LG" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "15", "collection_queries": "coding-agent, multi-agent-llm" } }, { "collection": "papers", "path": "papers/items/2026-2606-26479-adaptive-evaluation-of-out-of-band-defenses-against-prompt-injection-in-llm-agen.md", "title": "Adaptive Evaluation of Out-of-Band Defenses Against Prompt Injection in LLM Agents", "type": "paper", "meta": { "type": "paper", "title": "Adaptive Evaluation of Out-of-Band Defenses Against Prompt Injection in LLM Agents", "authors": "Praneeth Narisetty, Shiva Nagendra Babu Kore, Uday Kumar Reddy Kattamanchi, Jayaram Kumarapu", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.26479", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-25", "updated_at": "2026-06-25", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "rag", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CR", "cs.AI", "cs.CL", "cs.LG" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "18", "collection_queries": "tool-use" } }, { "collection": "papers", "path": "papers/items/2026-2606-26511-temporal-validity-in-retrieval-memory-eliminating-stale-fact-errors-for-ai-agent.md", "title": "\"Temporal Validity in Retrieval Memory: Eliminating Stale-Fact Errors for AI Agents over Evolving Knowledge\"", "type": "paper", "meta": { "type": "paper", "title": "\"Temporal Validity in Retrieval Memory: Eliminating Stale-Fact Errors for AI Agents over Evolving Knowledge\"", "authors": "Neeraj Yadav", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.26511", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-25", "updated_at": "2026-06-25", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "computer-use", "memory", "rag", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CL", "cs.AI", "cs.ET", "cs.LG" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "21", "collection_queries": "ai-agent, rag-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-26524-vigil-runtime-enforcement-of-behavioral-specifications-in-ai-agent-skills.md", "title": "\"VIGIL: Runtime Enforcement of Behavioral Specifications in AI Agent Skills\"", "type": "paper", "meta": { "type": "paper", "title": "\"VIGIL: Runtime Enforcement of Behavioral Specifications in AI Agent Skills\"", "authors": "Ying Li, Yanju Chen, Hongbo Wen, Bosi Zhang, Hanzhi Liu, Peiran Wang, Yu Feng, Yuan Tian", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.26524", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-25", "updated_at": "2026-06-25", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "tool-use", "workflow-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CR" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "15", "collection_queries": "ai-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-26614-hilsva-design-and-evaluation-of-a-human-in-the-loop-agentic-system-for-scientifi.md", "title": "\"HiLSVA: Design and Evaluation of a Human-in-the-Loop Agentic System for Scientific Visualization\"", "type": "paper", "meta": { "type": "paper", "title": "\"HiLSVA: Design and Evaluation of a Human-in-the-Loop Agentic System for Scientific Visualization\"", "authors": "Kuangshi Ai, Patrick Phuoc Do, Chaoli Wang", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.26614", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-25", "updated_at": "2026-06-25", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "computer-use", "multi-agent", "planning", "rag", "reasoning", "tool-use", "workflow-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.HC", "cs.AI", "cs.GR" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "23", "collection_queries": "multi-agent-llm, planning-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-26627-agents-that-know-too-much-a-data-centric-survey-of-privacy-in-llm-agents.md", "title": "\"Agents That Know Too Much: A Data-Centric Survey of Privacy in LLM Agents\"", "type": "paper", "meta": { "type": "paper", "title": "\"Agents That Know Too Much: A Data-Centric Survey of Privacy in LLM Agents\"", "authors": "Nada Lahjouji, Ashwin Gerard Colaco", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.26627", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-25", "updated_at": "2026-06-25", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "memory", "rag", "tool-use", "workflow-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CR", "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "18", "collection_queries": "agent-memory" } }, { "collection": "papers", "path": "papers/items/2026-2606-26649-autoformalization-of-agent-instructions-into-policy-as-code.md", "title": "Autoformalization of Agent Instructions into Policy-as-Code", "type": "paper", "meta": { "type": "paper", "title": "Autoformalization of Agent Instructions into Policy-as-Code", "authors": "Adam Mondl, Matthew Maisel, John H. Brock", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.26649", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-25", "updated_at": "2026-06-25", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "coding-agent", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI", "cs.CR" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "13", "collection_queries": "agent-safety" } }, { "collection": "papers", "path": "papers/items/2026-2606-26721-knowledge-based-pull-requests-a-trusted-workflow-for-agent-mediated-knowledge-co.md", "title": "\"Knowledge-Based Pull Requests: A Trusted Workflow for Agent-Mediated Knowledge Collaboration\"", "type": "paper", "meta": { "type": "paper", "title": "\"Knowledge-Based Pull Requests: A Trusted Workflow for Agent-Mediated Knowledge Collaboration\"", "authors": "Xinyu Zhang, Weiwei Sun", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.26721", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-25", "updated_at": "2026-06-25", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "coding-agent", "memory", "multi-agent", "planning", "tool-use", "workflow-agent", "world-model" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.SE", "cs.HC" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "16", "collection_queries": "coding-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-26758-egg-an-expert-guided-agent-framework-for-kernel-generation.md", "title": "\"EGG: An Expert-Guided Agent Framework for Kernel Generation\"", "type": "paper", "meta": { "type": "paper", "title": "\"EGG: An Expert-Guided Agent Framework for Kernel Generation\"", "authors": "Yaochen Han, Ke Fan, Hongxu Jiang, Wanqi Xu, Weiyu Xie, Runhua Zhang, Chenhui Zhu, Yixiang Zhang", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.26758", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-25", "updated_at": "2026-06-25", "status": "queued", "relevance": "high", "topics": [ "computer-use", "memory", "multi-agent", "rag", "workflow-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "14", "collection_queries": "multi-agent-llm" } }, { "collection": "papers", "path": "papers/items/2026-2606-26793-mirror-novelty-constrained-memory-guided-mcts-red-teaming-for-agentic-rag.md", "title": "\"MIRROR: Novelty-Constrained Memory-Guided MCTS Red-Teaming for Agentic RAG\"", "type": "paper", "meta": { "type": "paper", "title": "\"MIRROR: Novelty-Constrained Memory-Guided MCTS Red-Teaming for Agentic RAG\"", "authors": "Inderjeet Singh, Andrés Murillo, Motoyoshi Sekiya, Yuki Unno, Junichi Suga", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.26793", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-25", "updated_at": "2026-06-25", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "computer-use", "memory", "rag", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CR", "cs.AI", "cs.LG" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "18", "collection_queries": "rag-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-26806-memory-depth-not-memory-access-selective-parametric-consolidation-for-long-runni.md", "title": "\"Memory Depth, Not Memory Access: Selective Parametric Consolidation for Long-Running Language Agents\"", "type": "paper", "meta": { "type": "paper", "title": "\"Memory Depth, Not Memory Access: Selective Parametric Consolidation for Long-Running Language Agents\"", "authors": "Haoliang Han", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.26806", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-25", "updated_at": "2026-06-25", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "memory", "rag" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI", "cs.LG" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "14", "collection_queries": "language-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-26883-econsimulacra-a-digital-twin-platform-of-socio-economic-systems-powered-by-llm-a.md", "title": "\"EconSimulacra: A Digital Twin Platform of Socio-Economic Systems Powered by LLM Agents\"", "type": "paper", "meta": { "type": "paper", "title": "\"EconSimulacra: A Digital Twin Platform of Socio-Economic Systems Powered by LLM Agents\"", "authors": "Ryuji Hashimoto, Masahiro Kaneko, Kentaro Ueda, Takehiro Takayanagi, Kiyoshi Izumi", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.26883", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-25", "updated_at": "2026-06-30", "status": "queued", "relevance": "high", "topics": [ "memory", "multi-agent", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.DL" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "15", "collection_queries": "multi-agent-llm" } }, { "collection": "papers", "path": "papers/items/2026-2606-26918-diagnosing-task-insensitivity-in-language-agents.md", "title": "Diagnosing Task Insensitivity in Language Agents", "type": "paper", "meta": { "type": "paper", "title": "Diagnosing Task Insensitivity in Language Agents", "authors": "Jingyu Liu, Xiaopeng Wu, Kehan Chen, Chuan Yu, Yong Liu", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.26918", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-25", "updated_at": "2026-06-25", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "planning", "rag", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "15", "collection_queries": "language-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-26924-a-deterministic-control-plane-for-llm-coding-agents.md", "title": "A Deterministic Control Plane for LLM Coding Agents", "type": "paper", "meta": { "type": "paper", "title": "A Deterministic Control Plane for LLM Coding Agents", "authors": "Padmaraj Madatha", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.26924", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-25", "updated_at": "2026-06-25", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "coding-agent", "planning", "tool-use", "workflow-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.SE", "cs.AI", "cs.CR" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "15", "collection_queries": "coding-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-26960-toward-agentic-sysadmin-rethinking-system-administration-with-ai-agents.md", "title": "\"Toward Agentic SysAdmin: Rethinking System Administration with AI Agents\"", "type": "paper", "meta": { "type": "paper", "title": "\"Toward Agentic SysAdmin: Rethinking System Administration with AI Agents\"", "authors": "Gianmaria Frigo, Davide Saladino, Alberto Castagnaro, Francesco Marchiori, Denis Donadel, Luca Pajola, Mauro Conti", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.26960", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-25", "updated_at": "2026-06-25", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "rag", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.NI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "19", "collection_queries": "ai-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-27009-semantic-early-stopping-for-iterative-llm-agent-loops.md", "title": "Semantic Early-Stopping for Iterative LLM Agent Loops", "type": "paper", "meta": { "type": "paper", "title": "Semantic Early-Stopping for Iterative LLM Agent Loops", "authors": "Sahil Shrivastava", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.27009", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-25", "updated_at": "2026-06-25", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "multi-agent", "rag", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI", "cs.LG", "cs.MA" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "16", "collection_queries": "multi-agent-llm" } }, { "collection": "papers", "path": "papers/items/2026-2606-27154-openrca-2-0-from-outcome-labels-to-causal-process-supervision.md", "title": "\"OpenRCA 2.0: From Outcome Labels to Causal Process Supervision\"", "type": "paper", "meta": { "type": "paper", "title": "\"OpenRCA 2.0: From Outcome Labels to Causal Process Supervision\"", "authors": "\"Aoyang Fang, Yifan Yang, Jin'ao Shang, Qisheng Lu, Junjielung Xu, Rui Wang, Songhan Zhang, Yuzhong Zhang, et al.\"", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.27154", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-25", "updated_at": "2026-06-30", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "rag", "reasoning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "14", "collection_queries": "tool-use" } }, { "collection": "papers", "path": "papers/items/2026-2606-27243-nova-a-verification-aware-agent-harness-for-architecture-evolution-in-industrial.md", "title": "\"NOVA: A Verification-Aware Agent Harness for Architecture Evolution in Industrial Recommender Systems\"", "type": "paper", "meta": { "type": "paper", "title": "\"NOVA: A Verification-Aware Agent Harness for Architecture Evolution in Industrial Recommender Systems\"", "authors": "Shaohua Liu, Liang Fang, Yilong Sun, Shudong Huang, Qingsong Luo, Shaoxin Liu, Xiaoyang Chen, Dongqiang Liu, et al.", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.27243", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-25", "updated_at": "2026-06-26", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "coding-agent", "computer-use", "memory", "workflow-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.IR", "cs.SE" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "15", "collection_queries": "coding-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-27330-empowering-gui-agents-via-autonomous-experience-exploration-and-hindsight-experi.md", "title": "Empowering GUI Agents via Autonomous Experience Exploration and Hindsight Experience Utilization for Task Planning", "type": "paper", "meta": { "type": "paper", "title": "Empowering GUI Agents via Autonomous Experience Exploration and Hindsight Experience Utilization for Task Planning", "authors": "Tianyi Men, Zhuoran Jin, Pengfei Cao, Yubo Chen, Kang Liu, Jun Zhao", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.27330", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-25", "updated_at": "2026-06-25", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "computer-use", "planning", "rag", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CL", "cs.AI", "cs.CV", "cs.LG" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "16", "collection_queries": "web-gui-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-27350-chia-an-open-source-framework-for-principled-agentic-ai-driven-hardware-software.md", "title": "\"CHIA: An open-source framework for principled, agentic AI-driven hardware/software co-design research\"", "type": "paper", "meta": { "type": "paper", "title": "\"CHIA: An open-source framework for principled, agentic AI-driven hardware/software co-design research\"", "authors": "Angela Cui, Ferran Hermida-Rivera, Jack Toubes, Raghav Gupta, Jim Fang, Chengyi Lux Zhang, Ella Schwarz, Junha Kim, et al.", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.27350", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-25", "updated_at": "2026-06-27", "status": "queued", "relevance": "high", "topics": [ "agent-safety", "coding-agent", "tool-use", "workflow-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AR" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "16", "collection_queries": "agentic-ai, coding-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-27397-sidconarena-an-environment-evaluating-agents-in-open-ended-positive-sum-bargaini.md", "title": "\"SidConArena: An Environment Evaluating Agents in Open-Ended,Positive-Sum Bargaining Game\"", "type": "paper", "meta": { "type": "paper", "title": "\"SidConArena: An Environment Evaluating Agents in Open-Ended,Positive-Sum Bargaining Game\"", "authors": "Yeqi Feng, Yuxin Chen, Tianxing He", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.27397", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-24", "updated_at": "2026-06-24", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "memory", "planning", "reasoning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.MA", "cs.AI", "cs.GT" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "15", "collection_queries": "planning-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-27406-towards-evaluation-of-implicit-software-world-models-in-coding-llms.md", "title": "Towards Evaluation of Implicit Software World Models in Coding LLMs", "type": "paper", "meta": { "type": "paper", "title": "Towards Evaluation of Implicit Software World Models in Coding LLMs", "authors": "Egor Bogomolov, Yaroslav Zharov", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.27406", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-25", "updated_at": "2026-06-25", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "coding-agent", "memory", "reasoning", "world-model" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.SE", "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "18", "collection_queries": "ai-agent, coding-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-27416-glite-arf-verifier-driven-research-with-parallel-llm-coding-agents.md", "title": "\"Glite ARF: Verifier-Driven Research with Parallel LLM Coding Agents\"", "type": "paper", "meta": { "type": "paper", "title": "\"Glite ARF: Verifier-Driven Research with Parallel LLM Coding Agents\"", "authors": "Vassili Philippov, Pavel Katunin, Dmitry Andreev, Igor Ostanin, Anton Nikolaev", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.27416", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-25", "updated_at": "2026-06-25", "status": "queued", "relevance": "high", "topics": [ "coding-agent", "reasoning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.MA", "cs.SE" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "13", "collection_queries": "coding-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-27472-supersede-diagnosing-and-training-the-memory-update-gap-in-llm-agents.md", "title": "\"Supersede: Diagnosing and Training the Memory-Update Gap in LLM Agents\"", "type": "paper", "meta": { "type": "paper", "title": "\"Supersede: Diagnosing and Training the Memory-Update Gap in LLM Agents\"", "authors": "Vedant Patel", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.27472", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-25", "updated_at": "2026-06-25", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "memory", "planning", "reasoning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CL", "cs.AI", "cs.LG" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "18", "collection_queries": "planning-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-27483-internalizing-the-future-a-unified-agentic-training-paradigm-for-world-model-pla.md", "title": "\"Internalizing the Future: A Unified Agentic Training Paradigm for World Model Planning\"", "type": "paper", "meta": { "type": "paper", "title": "\"Internalizing the Future: A Unified Agentic Training Paradigm for World Model Planning\"", "authors": "Xuan Zhang, Zhijian Zhou, Lingfeng Qiao, Yulei Qin, Ke Li, Xing Sun, Xiaoyu Tan, Chao Qu, et al.", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.27483", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-25", "updated_at": "2026-06-25", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "planning", "rag", "reasoning", "world-model" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "16", "collection_queries": "planning-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-27492-queenbee-planner-skill-evolving-communication-topologies-for-token-efficient-llm.md", "title": "\"QueenBee Planner: Skill-Evolving Communication Topologies for Token-Efficient LLM Multi-Agent Systems\"", "type": "paper", "meta": { "type": "paper", "title": "\"QueenBee Planner: Skill-Evolving Communication Topologies for Token-Efficient LLM Multi-Agent Systems\"", "authors": "Congjia Tian, Yuhang Yao, Jiaming Cui", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.27492", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-25", "updated_at": "2026-06-25", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "multi-agent", "planning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.MA" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "14", "collection_queries": "multi-agent-llm" } }, { "collection": "papers", "path": "papers/items/2026-2606-27499-dmv-bench-diagnosing-long-horizon-multimodal-agents-visual-memory-with-incidenta.md", "title": "\"DMV-Bench: Diagnosing Long-Horizon Multimodal Agents' Visual Memory with Incidental Cue Injection\"", "type": "paper", "meta": { "type": "paper", "title": "\"DMV-Bench: Diagnosing Long-Horizon Multimodal Agents' Visual Memory with Incidental Cue Injection\"", "authors": "Yujin Tang, Chenming Shang, Ruize Xu, Nikhil Singh", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.27499", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-25", "updated_at": "2026-06-25", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "coding-agent", "memory", "planning", "rag", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CV", "cs.AI", "cs.CL" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "15", "collection_queries": "agent-memory" } }, { "collection": "papers", "path": "papers/items/2026-2606-27632-yuvion-llm-an-adversarially-aware-large-language-model-for-content-and-ai-safety.md", "title": "\"Yuvion LLM: An Adversarially-Aware Large Language Model for Content And AI Safety\"", "type": "paper", "meta": { "type": "paper", "title": "\"Yuvion LLM: An Adversarially-Aware Large Language Model for Content And AI Safety\"", "authors": "Ting Ma, Xiufeng Huang, Benlei Cui, Xiaowen Xu, Shikai Qiu, Ruijie Jian, Hongxing Li, Guanghui Wang, et al.", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.27632", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-26", "updated_at": "2026-06-26", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "planning", "reasoning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CL" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "14", "collection_queries": "tool-use" } }, { "collection": "papers", "path": "papers/items/2026-2606-27806-agent-vs-parametric-world-models-hybrid-planning-for-reliable-language-agents.md", "title": "\"Agent vs. Parametric World Models: Hybrid Planning for Reliable Language Agents\"", "type": "paper", "meta": { "type": "paper", "title": "\"Agent vs. Parametric World Models: Hybrid Planning for Reliable Language Agents\"", "authors": "Xinyuan Song, Zekun Cai", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.27806", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-26", "updated_at": "2026-07-05", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "planning", "rag", "reasoning", "tool-use", "world-model" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "19", "collection_queries": "language-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-27929-when-multi-robot-systems-meet-agentic-ai-towards-embodied-collective-intelligenc.md", "title": "\"When Multi-Robot Systems Meet Agentic AI:Towards Embodied Collective Intelligence\"", "type": "paper", "meta": { "type": "paper", "title": "\"When Multi-Robot Systems Meet Agentic AI:Towards Embodied Collective Intelligence\"", "authors": "Yuxuan Yan, Yuanyuan Jia, Qianqian Yang", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.27929", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-26", "updated_at": "2026-06-26", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "embodied-agent", "memory", "multi-agent", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.RO" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "17", "collection_queries": "ai-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-27990-advancedshellm-a-stateful-multi-agent-llm-honeypot-for-ssh-deception.md", "title": "\"AdvancedShelLM: A Stateful Multi-Agent LLM Honeypot for SSH Deception\"", "type": "paper", "meta": { "type": "paper", "title": "\"AdvancedShelLM: A Stateful Multi-Agent LLM Honeypot for SSH Deception\"", "authors": "Muris Sladić, Eman Alibalić, Veronica Valeros, Carlos Catania, Sebastian Garcia", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.27990", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-26", "updated_at": "2026-06-26", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "memory", "multi-agent", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CR" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "17", "collection_queries": "llm-agent, multi-agent-llm" } }, { "collection": "papers", "path": "papers/items/2026-2606-28011-from-detection-to-action-using-llm-agents-for-fault-tolerant-control.md", "title": "\"From Detection to Action: Using LLM Agents for Fault-Tolerant Control\"", "type": "paper", "meta": { "type": "paper", "title": "\"From Detection to Action: Using LLM Agents for Fault-Tolerant Control\"", "authors": "Javal Vyas, Milapji Singh Gill, Artan Markaj, Felix Gehlhoff, Mehmet Mercangöz", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.28011", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-26", "updated_at": "2026-06-26", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "multi-agent", "planning", "rag", "tool-use", "workflow-agent", "world-model" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "eess.SY", "cs.LG" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "24", "collection_queries": "agentic-ai, llm-agent, multi-agent-llm, planning-agent, rag-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-28061-toolprivacybench-benchmarking-purpose-bound-privacy-in-tool-using-llm-agents.md", "title": "\"ToolPrivacyBench: Benchmarking Purpose-Bound Privacy in Tool-Using LLM Agents\"", "type": "paper", "meta": { "type": "paper", "title": "\"ToolPrivacyBench: Benchmarking Purpose-Bound Privacy in Tool-Using LLM Agents\"", "authors": "Shijing Hu, Liang Liu, Zhu Meng, Zhicheng Zhao", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.28061", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-26", "updated_at": "2026-06-26", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "rag", "tool-use", "workflow-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CR", "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "22", "collection_queries": "agent-evaluation, function-calling, llm-agent, tool-use" } }, { "collection": "papers", "path": "papers/items/2026-2606-28182-llawco-learning-laws-of-cooperation-for-modeling-embodied-multi-agent-behavior.md", "title": "\"LLawCo: Learning Laws of Cooperation for Modeling Embodied Multi-Agent Behavior\"", "type": "paper", "meta": { "type": "paper", "title": "\"LLawCo: Learning Laws of Cooperation for Modeling Embodied Multi-Agent Behavior\"", "authors": "Qinhong Zhou, Chuang Gan, Anoop Cherian", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.28182", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-26", "updated_at": "2026-06-26", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "embodied-agent", "multi-agent", "planning", "rag", "reasoning" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.LG", "cs.AI", "cs.CV", "cs.RO" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "17", "collection_queries": "multi-agent-llm" } }, { "collection": "papers", "path": "papers/items/2026-2606-28187-gbc-gradient-based-connections-for-optimizing-multi-agent-systems.md", "title": "\"GBC: Gradient-Based Connections for Optimizing Multi-Agent Systems\"", "type": "paper", "meta": { "type": "paper", "title": "\"GBC: Gradient-Based Connections for Optimizing Multi-Agent Systems\"", "authors": "Xiaocheng Yang, Abdulrahman Alrabah, Dilek Hakkani-Tür, Gokhan Tur", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.28187", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-26", "updated_at": "2026-06-26", "status": "queued", "relevance": "high", "topics": [ "multi-agent", "rag", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.MA" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "13", "collection_queries": "multi-agent-llm" } }, { "collection": "papers", "path": "papers/items/2026-2606-28270-agent-native-immune-system-architecture-taxonomy-and-engineering.md", "title": "\"Agent-Native Immune System: Architecture, Taxonomy, and Engineering\"", "type": "paper", "meta": { "type": "paper", "title": "\"Agent-Native Immune System: Architecture, Taxonomy, and Engineering\"", "authors": "Bo Shen, Lifeng Chang, Tianyuan Wei, Yunpeng Li, Feng Shi, Yichen Han, Peijie Gao, Shiyi Kuang, et al.", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.28270", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-26", "updated_at": "2026-06-26", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "memory", "multi-agent", "reasoning", "tool-use", "workflow-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI", "cs.MA" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "15", "collection_queries": "tool-use" } }, { "collection": "papers", "path": "papers/items/2026-2606-28279-agentic-hardware-design-as-repository-level-code-evolution.md", "title": "Agentic Hardware Design as Repository-Level Code Evolution", "type": "paper", "meta": { "type": "paper", "title": "Agentic Hardware Design as Repository-Level Code Evolution", "authors": "Cunxi Yu, Chenhui Deng, Nathaniel Pinckney, Brucek Khailany", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.28279", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-26", "updated_at": "2026-06-26", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "coding-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AR", "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "13", "collection_queries": "agentic-ai" } }, { "collection": "papers", "path": "papers/items/2026-2606-28349-hmars-a-hierarchical-multi-agent-memory-system-for-long-context-reasoning.md", "title": "\"HMARS: A Hierarchical Multi-Agent Memory System for Long-Context Reasoning\"", "type": "paper", "meta": { "type": "paper", "title": "\"HMARS: A Hierarchical Multi-Agent Memory System for Long-Context Reasoning\"", "authors": "Zeju Li, Ziyang Zheng, Yizhou Zhou, Qiang Xu", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.28349", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-03", "updated_at": "2026-06-03", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "memory", "multi-agent", "rag", "reasoning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.IR", "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "19", "collection_queries": "agent-memory" } }, { "collection": "papers", "path": "papers/items/2026-2606-28360-carolina-guide-a-multi-agent-rag-system-with-institutional-guardrails-for-academ.md", "title": "\"Carolina Guide: A Multi-Agent RAG System with Institutional Guardrails for Academic Policy Assistance\"", "type": "paper", "meta": { "type": "paper", "title": "\"Carolina Guide: A Multi-Agent RAG System with Institutional Guardrails for Academic Policy Assistance\"", "authors": "Ben Torsion, Jun Zhou", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.28360", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-11", "updated_at": "2026-06-11", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "computer-use", "multi-agent", "rag" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.IR", "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "16", "collection_queries": "rag-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-28374-recursive-self-evolving-agents-via-held-out-selection.md", "title": "Recursive Self-Evolving Agents via Held-Out Selection", "type": "paper", "meta": { "type": "paper", "title": "Recursive Self-Evolving Agents via Held-Out Selection", "authors": "Michael Nguyen, Quoc Nguyen, Paul Vuong", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.28374", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-17", "updated_at": "2026-06-17", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "reasoning", "tool-use", "workflow-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "13", "collection_queries": "tool-use" } }, { "collection": "papers", "path": "papers/items/2026-2606-28409-evidence-driven-llm-agent-for-c-to-synthesizable-c-conversion-and-verification.md", "title": "Evidence-Driven LLM Agent for C-to-Synthesizable-C Conversion and Verification", "type": "paper", "meta": { "type": "paper", "title": "Evidence-Driven LLM Agent for C-to-Synthesizable-C Conversion and Verification", "authors": "Zhe Zhao, Hongbing Lang, Zhihan Xiao, Luke Ztz Hu, John Imoleayo Adebisi, Songping Mai", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.28409", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-25", "updated_at": "2026-06-25", "status": "queued", "relevance": "high", "topics": [ "rag", "reasoning", "tool-use", "workflow-agent", "world-model" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AR", "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "15", "collection_queries": "rag-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-28425-tool-use-enables-undetectable-steganography-in-multi-agent-llm-systems.md", "title": "Tool Use Enables Undetectable Steganography in Multi-Agent LLM Systems", "type": "paper", "meta": { "type": "paper", "title": "Tool Use Enables Undetectable Steganography in Multi-Agent LLM Systems", "authors": "Jimmy Laurence Rippin, Simon C. Marshall, David Demitri Africa, Christian Schroeder de Witt", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.28425", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-25", "updated_at": "2026-06-25", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "coding-agent", "computer-use", "multi-agent", "rag", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CR", "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "25", "collection_queries": "agentic-ai, ai-agent, autonomous-agent-llm, multi-agent-llm, tool-use" } }, { "collection": "papers", "path": "papers/items/2026-2606-28430-building-to-the-test-coding-agents-deliver-what-you-check-not-what-you-requested.md", "title": "\"Building to the Test: Coding Agents Deliver What You Check, Not What You Requested\"", "type": "paper", "meta": { "type": "paper", "title": "\"Building to the Test: Coding Agents Deliver What You Check, Not What You Requested\"", "authors": "Yanuo Ma, Ben Kereopa-Yorke, Ben Schultz", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.28430", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-26", "updated_at": "2026-06-26", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "coding-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.SE", "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "13", "collection_queries": "coding-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-28434-swe-mem-learning-adaptive-memory-management-for-long-horizon-coding-agents.md", "title": "\"SWE-MeM: Learning Adaptive Memory Management for Long-Horizon Coding Agents\"", "type": "paper", "meta": { "type": "paper", "title": "\"SWE-MeM: Learning Adaptive Memory Management for Long-Horizon Coding Agents\"", "authors": "Shuzheng Gao, Wenhao Zeng, Zhaojian Yu, Jianqiao Wangni, Chaozheng Wang, Kai Cai, Shilin He, Michael R. Lyu", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.28434", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-26", "updated_at": "2026-06-26", "status": "queued", "relevance": "high", "topics": [ "coding-agent", "memory", "planning", "tool-use", "workflow-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.SE", "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "17", "collection_queries": "coding-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-28436-dockerless-environment-free-program-verifier-for-coding-agents.md", "title": "\"Dockerless: Environment-Free Program Verifier for Coding Agents\"", "type": "paper", "meta": { "type": "paper", "title": "\"Dockerless: Environment-Free Program Verifier for Coding Agents\"", "authors": "Wenhao Zeng, Yuling Shi, Xiaodong Gu, Chao Hu, Chaofan Wang, Yuhao Cui, Hongting Zhou, Mengnan Qi, et al.", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.28436", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-26", "updated_at": "2026-06-26", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "coding-agent", "reasoning" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.SE", "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "15", "collection_queries": "coding-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-28450-llm-agents-security-duality-a-comprehensive-survey-of-self-security-and-empowere.md", "title": "\"LLM agents security duality: a comprehensive survey of self-security and empowered cybersecurity\"", "type": "paper", "meta": { "type": "paper", "title": "\"LLM agents security duality: a comprehensive survey of self-security and empowered cybersecurity\"", "authors": "Yiwei Xu, Yong Zhuang, Xuanming Liu, Tian Zhang, Bowen Xiao, Xiaoyang Xu, Delong Jiang, Juan Wang, et al.", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.28450", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-26", "updated_at": "2026-06-26", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CR", "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "16", "collection_queries": "agent-safety, llm-agent, tool-use" } }, { "collection": "papers", "path": "papers/items/2026-2606-28456-is-lying-an-emergent-behaviour-in-llms-evidence-from-gaslighting-ai-agents-in-a-.md", "title": "Is Lying an Emergent Behaviour in LLMs? Evidence from Gaslighting AI agents in a Sustainability Game", "type": "paper", "meta": { "type": "paper", "title": "Is Lying an Emergent Behaviour in LLMs? Evidence from Gaslighting AI agents in a Sustainability Game", "authors": "Subhendu Bhandary, Federico Carucci, Christos Charalambous, Francesca Dilisante, Ksenia Dvorkina, Anna Garbo, Jiaqi Liang, Riccardo Vasellini, et al.", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.28456", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-26", "updated_at": "2026-06-26", "status": "queued", "relevance": "high", "topics": [ "agent-safety", "memory", "multi-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.MA", "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "18", "collection_queries": "ai-agent, llm-agent, multi-agent-llm" } }, { "collection": "papers", "path": "papers/items/2026-2606-28467-an-agentic-ai-pipeline-for-appliance-level-energy-anomaly-detection-and-llm-driv.md", "title": "An Agentic AI Pipeline for Appliance-Level Energy Anomaly Detection and LLM-Driven Recommendations", "type": "paper", "meta": { "type": "paper", "title": "An Agentic AI Pipeline for Appliance-Level Energy Anomaly Detection and LLM-Driven Recommendations", "authors": "Dihia Falouz, Aida Douaibia, Amine Bechar, Youssef Elmir, Abbes Amira, Adel Oulefki", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.28467", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-26", "updated_at": "2026-06-26", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "memory", "rag", "reasoning", "tool-use", "workflow-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.LG", "cs.AI", "cs.CL" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "19", "collection_queries": "agentic-ai, rag-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-28480-tua-bench-a-benchmark-for-general-purpose-terminal-use-agents.md", "title": "\"TUA-Bench: A Benchmark for General-Purpose Terminal-Use Agents\"", "type": "paper", "meta": { "type": "paper", "title": "\"TUA-Bench: A Benchmark for General-Purpose Terminal-Use Agents\"", "authors": "Shoufa Chen, Luyuan Wang, Xuan Yang, Zhiheng Liu, Yuren Cong, Yuanfeng Ji, Feiyan Zhou, Xiaohui Zhang, et al.", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.28480", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-26", "updated_at": "2026-06-26", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "coding-agent", "computer-use", "reasoning", "workflow-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.SE", "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "16", "collection_queries": "web-gui-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-28570-digitizing-coaching-intelligence-an-agentic-framework-for-holistic-athlete-profi.md", "title": "\"Digitizing Coaching Intelligence: An Agentic Framework for Holistic Athlete Profiling using VLM and RAG\"", "type": "paper", "meta": { "type": "paper", "title": "\"Digitizing Coaching Intelligence: An Agentic Framework for Holistic Athlete Profiling using VLM and RAG\"", "authors": "Deep Ghosal, Ishani Sen, Wazib Ansar, Amlan Chakrabarti", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.28570", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-26", "updated_at": "2026-06-26", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "multi-agent", "rag", "reasoning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CV", "cs.AI", "cs.MA" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "17", "collection_queries": "multi-agent-llm, rag-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-28666-why-trust-your-agent-empirical-security-gains-from-trism-guided-agentic-workflow.md", "title": "Why Trust Your Agent? Empirical Security Gains from TRiSM-Guided Agentic Workflows in Healthcare", "type": "paper", "meta": { "type": "paper", "title": "Why Trust Your Agent? Empirical Security Gains from TRiSM-Guided Agentic Workflows in Healthcare", "authors": "Liam Kearns", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.28666", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-27", "updated_at": "2026-06-27", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "computer-use", "rag", "tool-use", "workflow-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CR", "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "17", "collection_queries": "agentic-ai, rag-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-28679-capability-gates-are-not-authorization-confused-deputy-failures-in-llm-agent-fra.md", "title": "\"Capability Gates Are Not Authorization: Confused-Deputy Failures in LLM Agent Frameworks\"", "type": "paper", "meta": { "type": "paper", "title": "\"Capability Gates Are Not Authorization: Confused-Deputy Failures in LLM Agent Frameworks\"", "authors": "David Mellafe Zuvic", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.28679", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-27", "updated_at": "2026-06-27", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CR", "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "14", "collection_queries": "llm-agent, tool-use" } }, { "collection": "papers", "path": "papers/items/2026-2606-28692-an-ai-agent-for-treatment-reasoning-over-a-biomedical-tool-universe.md", "title": "An AI agent for treatment reasoning over a biomedical tool universe", "type": "paper", "meta": { "type": "paper", "title": "An AI agent for treatment reasoning over a biomedical tool universe", "authors": "Shanghua Gao, Ayush Noori, Richard Zhu, Curtis Ginder, Zhenglun Kong, Xiaorui Su, Justin Kauffman, Benjamin S. Glicksberg, et al.", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.28692", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-27", "updated_at": "2026-06-27", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "multi-agent", "reasoning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "19", "collection_queries": "ai-agent, tool-use" } }, { "collection": "papers", "path": "papers/items/2026-2606-28733-agentic-abstention-do-agents-know-when-to-stop-instead-of-act.md", "title": "\"Agentic Abstention: Do Agents Know When to Stop Instead of Act?\"", "type": "paper", "meta": { "type": "paper", "title": "\"Agentic Abstention: Do Agents Know When to Stop Instead of Act?\"", "authors": "Han Luo, Bingbing Wen, Lucy Lu Wang", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.28733", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-27", "updated_at": "2026-06-27", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "reasoning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "14", "collection_queries": "llm-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-28739-agent-safety-is-action-alignment.md", "title": "Agent Safety Is Action Alignment", "type": "paper", "meta": { "type": "paper", "title": "Agent Safety Is Action Alignment", "authors": "Shawn Li, Yue Zhao", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.28739", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-27", "updated_at": "2026-06-27", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "14", "collection_queries": "agent-safety" } }, { "collection": "papers", "path": "papers/items/2026-2606-28781-hyphaedb-a-living-knowledge-topology-for-agent-first-memory.md", "title": "\"HyphaeDB: A Living Knowledge Topology for Agent-First Memory\"", "type": "paper", "meta": { "type": "paper", "title": "\"HyphaeDB: A Living Knowledge Topology for Agent-First Memory\"", "authors": "Krishna Halaharvi", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.28781", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-27", "updated_at": "2026-06-27", "status": "queued", "relevance": "high", "topics": [ "coding-agent", "memory", "multi-agent", "rag", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI", "cs.MA" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "15", "collection_queries": "agent-memory, agentic-ai" } }, { "collection": "papers", "path": "papers/items/2026-2606-28791-from-determinism-to-delegation-ai-native-software-engineering-and-the-evolution-.md", "title": "\"From Determinism to Delegation: AI-Native Software Engineering and the Evolution of the Agentic Engineer\"", "type": "paper", "meta": { "type": "paper", "title": "\"From Determinism to Delegation: AI-Native Software Engineering and the Evolution of the Agentic Engineer\"", "authors": "Mamdouh Alenezi", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.28791", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-27", "updated_at": "2026-06-27", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "coding-agent", "memory", "multi-agent", "reasoning", "tool-use", "workflow-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.SE" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "23", "collection_queries": "agentic-ai, autonomous-agent-llm, tool-use" } }, { "collection": "papers", "path": "papers/items/2026-2606-28839-the-contagion-tensor-a-framework-for-measuring-output-distribution-coupling-in-m.md", "title": "\"The Contagion Tensor: A Framework for Measuring Output-Distribution Coupling in Multi-Agent LLM Systems -- and Auditing the Claims It Enables\"", "type": "paper", "meta": { "type": "paper", "title": "\"The Contagion Tensor: A Framework for Measuring Output-Distribution Coupling in Multi-Agent LLM Systems -- and Auditing the Claims It Enables\"", "authors": "Zewen Liu", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.28839", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-27", "updated_at": "2026-06-27", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "computer-use", "multi-agent", "tool-use", "world-model" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.LG" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "15", "collection_queries": "multi-agent-llm" } }, { "collection": "papers", "path": "papers/items/2026-2606-28841-lamp-lean-based-agentic-framework-with-mcp-and-proof-repair.md", "title": "\"LAMP: Lean-based Agentic framework with MCP and Proof Repair\"", "type": "paper", "meta": { "type": "paper", "title": "\"LAMP: Lean-based Agentic framework with MCP and Proof Repair\"", "authors": "Santhana Srinivasan R, Maithilee Patawar", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.28841", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-27", "updated_at": "2026-06-27", "status": "queued", "relevance": "high", "topics": [ "coding-agent", "multi-agent", "planning", "reasoning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.LO", "cs.AI", "cs.CL" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "16", "collection_queries": "multi-agent-llm" } }, { "collection": "papers", "path": "papers/items/2026-2606-28896-a-task-driven-and-quality-assured-agent-framework-for-sar-data-generation.md", "title": "A Task-Driven and Quality-Assured Agent Framework for SAR Data Generation", "type": "paper", "meta": { "type": "paper", "title": "A Task-Driven and Quality-Assured Agent Framework for SAR Data Generation", "authors": "Xuanting Wu, Fan Zhanga, Fei Ma, Ling Guan, Guochun Ma, Yongsheng Zhou", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.28896", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-27", "updated_at": "2026-06-27", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "planning", "rag", "workflow-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "eess.IV", "cs.AI", "cs.LG" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "13", "collection_queries": "agent-evaluation" } }, { "collection": "papers", "path": "papers/items/2026-2606-28925-multi-agent-routing-as-set-valued-prediction-a-wildchat-benchmark-and-cost-aware.md", "title": "\"Multi-Agent Routing as Set-Valued Prediction: A WildChat Benchmark and Cost-Aware Evaluation\"", "type": "paper", "meta": { "type": "paper", "title": "\"Multi-Agent Routing as Set-Valued Prediction: A WildChat Benchmark and Cost-Aware Evaluation\"", "authors": "Ananto Nayan Bala, Faisal Muhammad Shah", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.28925", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-27", "updated_at": "2026-06-27", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "multi-agent", "rag", "tool-use", "world-model" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.LG", "cs.AI", "cs.IR", "cs.MA" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "21", "collection_queries": "multi-agent-llm" } }, { "collection": "papers", "path": "papers/items/2026-2606-28958-when-latent-agents-lie-kv-cache-integrity-in-multi-agent-llm-collaboration.md", "title": "\"When Latent Agents Lie: KV-Cache Integrity in Multi-Agent LLM Collaboration\"", "type": "paper", "meta": { "type": "paper", "title": "\"When Latent Agents Lie: KV-Cache Integrity in Multi-Agent LLM Collaboration\"", "authors": "Luís Brito, Carlos Baquero", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.28958", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-27", "updated_at": "2026-06-27", "status": "queued", "relevance": "high", "topics": [ "agent-safety", "memory", "multi-agent", "reasoning" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.MA" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "17", "collection_queries": "llm-agent, multi-agent-llm" } }, { "collection": "papers", "path": "papers/items/2026-2606-29014-customized-generative-ai-agent-for-transportation-engineering-practice-a-develop.md", "title": "\"Customized Generative AI Agent for Transportation Engineering Practice: A Development and Continued Pre-training Guideline\"", "type": "paper", "meta": { "type": "paper", "title": "\"Customized Generative AI Agent for Transportation Engineering Practice: A Development and Continued Pre-training Guideline\"", "authors": "Dianwei Chen, Yuan-Zheng Lei, Zifan Zhang, Yuchen Liu, Xianfeng Yang", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.29014", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-27", "updated_at": "2026-07-03", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "computer-use", "planning", "reasoning" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI", "cs.DL" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "15", "collection_queries": "ai-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-29026-preventing-error-propagation-in-multi-agent-ai-through-runtime-monitoring.md", "title": "Preventing Error Propagation in Multi-Agent AI through Runtime Monitoring", "type": "paper", "meta": { "type": "paper", "title": "Preventing Error Propagation in Multi-Agent AI through Runtime Monitoring", "authors": "Shahnewaz Karim Sakib, Anindya Bijoy Das", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.29026", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-27", "updated_at": "2026-06-27", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "multi-agent", "reasoning" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI", "cs.ET" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "14", "collection_queries": "agentic-ai" } }, { "collection": "papers", "path": "papers/items/2026-2606-29030-memory-as-an-attack-surface-in-llm-agents-a-study-on-multiple-choice-question-an.md", "title": "\"Memory as an Attack Surface in LLM Agents: A Study on Multiple-Choice Question Answering\"", "type": "paper", "meta": { "type": "paper", "title": "\"Memory as an Attack Surface in LLM Agents: A Study on Multiple-Choice Question Answering\"", "authors": "Shahnewaz Karim Sakib, Anindya Bijoy Das", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.29030", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-27", "updated_at": "2026-06-27", "status": "queued", "relevance": "high", "topics": [ "memory", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI", "cs.ET" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "19", "collection_queries": "ai-agent, llm-agent, tool-use" } }, { "collection": "papers", "path": "papers/items/2026-2606-29116-characterizing-large-language-model-agentic-workflows-a-study-on-n8n-ecosystem.md", "title": "\"Characterizing Large Language Model Agentic Workflows: A Study on N8n Ecosystem\"", "type": "paper", "meta": { "type": "paper", "title": "\"Characterizing Large Language Model Agentic Workflows: A Study on N8n Ecosystem\"", "authors": "Yutian Tang, Yuming Zhou, Huaming Chen", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.29116", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-27", "updated_at": "2026-06-27", "status": "queued", "relevance": "high", "topics": [ "agent-safety", "planning", "rag", "tool-use", "workflow-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "20", "collection_queries": "agentic-ai, llm-agent, planning-agent, tool-use" } }, { "collection": "papers", "path": "papers/items/2026-2606-29142-agent-security-meets-regulatory-reality-a-practitioner-systematization-of-autono.md", "title": "Agent Security Meets Regulatory Reality -- A Practitioner Systematization of Autonomous-Agent Threats and Controls in Regulated Financial Systems", "type": "paper", "meta": { "type": "paper", "title": "Agent Security Meets Regulatory Reality -- A Practitioner Systematization of Autonomous-Agent Threats and Controls in Regulated Financial Systems", "authors": "Krishna Mohan, Guda Nagavenkata Srinivasa", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.29142", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-28", "updated_at": "2026-06-28", "status": "queued", "relevance": "high", "topics": [ "agent-safety", "computer-use", "rag", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CY", "cs.SE" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "14", "collection_queries": "agent-safety, autonomous-agent-llm, rag-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-29178-selective-memory-retention-for-long-horizon-llm-agents.md", "title": "Selective Memory Retention for Long-Horizon LLM Agents", "type": "paper", "meta": { "type": "paper", "title": "Selective Memory Retention for Long-Horizon LLM Agents", "authors": "Pranath Reddy", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.29178", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-28", "updated_at": "2026-06-28", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "memory", "planning" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI", "cs.CL", "cs.LG" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "18", "collection_queries": "llm-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-29193-a-multi-dataset-benchmark-for-evaluating-llm-agents-in-microservice-failure-diag.md", "title": "A Multi-Dataset Benchmark for Evaluating LLM Agents in Microservice Failure Diagnosis", "type": "paper", "meta": { "type": "paper", "title": "A Multi-Dataset Benchmark for Evaluating LLM Agents in Microservice Failure Diagnosis", "authors": "Yuanhong Cai, Xiaohui Nie, Kanglin Yin, Changhua Pei, Yongqian Sun, Shenglin Zhang, Haibin Liu, Guiyang Liu, et al.", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.29193", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-28", "updated_at": "2026-06-28", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "rag", "reasoning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.SE", "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "19", "collection_queries": "llm-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-29225-policyguard-a-dialogue-grounded-sub-agent-verifier-for-policy-adherence-in-llm-a.md", "title": "\"PolicyGuard: A Dialogue-Grounded Sub-Agent Verifier for Policy Adherence in LLM Agents\"", "type": "paper", "meta": { "type": "paper", "title": "\"PolicyGuard: A Dialogue-Grounded Sub-Agent Verifier for Policy Adherence in LLM Agents\"", "authors": "Seongjae Kang, Taehyung Yu, Sung Ju Hwang", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.29225", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-28", "updated_at": "2026-06-28", "status": "queued", "relevance": "high", "topics": [ "computer-use", "reasoning", "tool-use", "workflow-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI", "cs.CL" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "14", "collection_queries": "llm-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-29270-minority-sentinel-when-to-overturn-majority-voting-in-multi-agent-llm-debates.md", "title": "\"Minority Sentinel: When to Overturn Majority Voting in Multi-Agent LLM Debates\"", "type": "paper", "meta": { "type": "paper", "title": "\"Minority Sentinel: When to Overturn Majority Voting in Multi-Agent LLM Debates\"", "authors": "Chuan He, Zebin Chen, Zhengyi Yang, Shaobo Qiao, Mingchen Ju, Jiate Liu, Dong Wen, Guanfeng Liu", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.29270", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-28", "updated_at": "2026-06-28", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "multi-agent", "reasoning" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.MA" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "17", "collection_queries": "llm-agent, multi-agent-llm" } }, { "collection": "papers", "path": "papers/items/2026-2606-29315-hierarchical-experimentalist-agents.md", "title": "Hierarchical Experimentalist Agents", "type": "paper", "meta": { "type": "paper", "title": "Hierarchical Experimentalist Agents", "authors": "Abhranil Chandra, Sankaran Vaidyanathan, Utsav Dhanuka, Varun Gandhi, Scott Niekum", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.29315", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-28", "updated_at": "2026-06-28", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "planning", "rag", "reasoning", "tool-use", "world-model" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI", "cs.LG" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "16", "collection_queries": "llm-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-29354-when-llms-develop-languages-symbolic-communication-for-efficient-multi-agent-rea.md", "title": "\"When LLMs Develop Languages: Symbolic Communication for Efficient Multi-Agent Reasoning\"", "type": "paper", "meta": { "type": "paper", "title": "\"When LLMs Develop Languages: Symbolic Communication for Efficient Multi-Agent Reasoning\"", "authors": "Zhengqi Pei, Qingming Huang, Shuhui Wang", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.29354", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-28", "updated_at": "2026-06-28", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "multi-agent", "reasoning" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI", "cs.NE" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "16", "collection_queries": "llm-agent, multi-agent-llm" } }, { "collection": "papers", "path": "papers/items/2026-2606-29445-bridging-videoqa-and-video-guided-agentic-tasks-via-generalized-keyframe-extract.md", "title": "Bridging VideoQA and Video-Guided Agentic Tasks via Generalized Keyframe Extraction", "type": "paper", "meta": { "type": "paper", "title": "Bridging VideoQA and Video-Guided Agentic Tasks via Generalized Keyframe Extraction", "authors": "Sunqi Fan, Qingle Liu, Runqi Yin, Meng-Hao Guo, Shuojin Yang", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.29445", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-28", "updated_at": "2026-06-28", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "computer-use", "planning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CV", "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "16", "collection_queries": "web-gui-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-29495-cognitive-world-models-for-process-level-social-influence-evaluation.md", "title": "Cognitive World Models for Process-Level Social Influence Evaluation", "type": "paper", "meta": { "type": "paper", "title": "Cognitive World Models for Process-Level Social Influence Evaluation", "authors": "Minghui Ma, Bin Guo, Han Wang, Mengqi Chen, Jingqi Liu, Yan Liu, Zhiwen Yu", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.29495", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-28", "updated_at": "2026-06-28", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "computer-use", "multi-agent", "world-model" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "13", "collection_queries": "multi-agent-llm" } }, { "collection": "papers", "path": "papers/items/2026-2606-29537-osworld2-0-benchmarking-computer-use-agents-on-long-horizon-real-world-tasks.md", "title": "\"OSWorld2.0: Benchmarking Computer Use Agents on Long-Horizon Real-World Tasks\"", "type": "paper", "meta": { "type": "paper", "title": "\"OSWorld2.0: Benchmarking Computer Use Agents on Long-Horizon Real-World Tasks\"", "authors": "Mengqi Yuan, Zilong Zhou, Xinzhuang Xiong, Weiming Wu, Jiayang Sun, Jiamin Song, Kaiqian Cui, Bowen Wang, et al.", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.29537", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-28", "updated_at": "2026-06-28", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "coding-agent", "computer-use", "memory", "planning", "rag", "reasoning", "tool-use", "workflow-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "21", "collection_queries": "web-gui-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-29654-budgeted-act-or-defer-multi-agent-llm-deliberation-with-local-reliability-bounds.md", "title": "Budgeted Act-or-Defer Multi-Agent LLM Deliberation with Local Reliability Bounds", "type": "paper", "meta": { "type": "paper", "title": "Budgeted Act-or-Defer Multi-Agent LLM Deliberation with Local Reliability Bounds", "authors": "Mengdie Flora Wang, Haochen Xie, Guanghui Wang, Devin Zhang, Jae Oh Woo", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.29654", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-28", "updated_at": "2026-06-28", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "multi-agent", "reasoning", "tool-use", "workflow-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI", "cs.MA" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "17", "collection_queries": "multi-agent-llm" } }, { "collection": "papers", "path": "papers/items/2026-2606-29719-a-diagnostic-framework-and-multi-evaluator-audit-of-evaluator-driven-preference-.md", "title": "A Diagnostic Framework and Multi-Evaluator Audit of Evaluator-Driven Preference Dynamics in Self-Adapting LLM Agents", "type": "paper", "meta": { "type": "paper", "title": "A Diagnostic Framework and Multi-Evaluator Audit of Evaluator-Driven Preference Dynamics in Self-Adapting LLM Agents", "authors": "Liu Zewen", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.29719", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-29", "updated_at": "2026-06-29", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "rag" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.LG", "cs.CL" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "13", "collection_queries": "llm-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-29742-microagent-context-augmented-multi-agent-framework-for-automatic-microservice-de.md", "title": "\"MicroAgent: Context-Augmented Multi-Agent Framework for Automatic Microservice Decomposition\"", "type": "paper", "meta": { "type": "paper", "title": "\"MicroAgent: Context-Augmented Multi-Agent Framework for Automatic Microservice Decomposition\"", "authors": "Zishan Su, Junjie Huang, Shiwen Shan, Xingyan Chen, Hui Zeng, Yuxin Su, Yanlin Wang, Michael R. Lyu", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.29742", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-29", "updated_at": "2026-06-29", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "coding-agent", "computer-use", "multi-agent", "rag", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.SE" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "17", "collection_queries": "multi-agent-llm" } }, { "collection": "papers", "path": "papers/items/2026-2606-29745-echo-learning-epistemically-adaptive-language-agents-with-turn-level-credit.md", "title": "\"ECHO: Learning Epistemically Adaptive Language Agents with Turn-Level Credit\"", "type": "paper", "meta": { "type": "paper", "title": "\"ECHO: Learning Epistemically Adaptive Language Agents with Turn-Level Credit\"", "authors": "Abhijnan Nath, Nikhil Krishnaswamy", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.29745", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-29", "updated_at": "2026-06-29", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "rag", "reasoning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.MA" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "13", "collection_queries": "language-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-29746-deepmed-search-an-open-source-agentic-platform-for-medical-deep-research-with-in.md", "title": "\"DEEPMED Search: An Open-Source Agentic Platform for Medical Deep Research with Introspective Verification\"", "type": "paper", "meta": { "type": "paper", "title": "\"DEEPMED Search: An Open-Source Agentic Platform for Medical Deep Research with Introspective Verification\"", "authors": "Maolin Liu, Fanyu Xu, Ruoqing Xu, Jiahang Zhang, Hao Wang, Rui Wang", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.29746", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-29", "updated_at": "2026-06-29", "status": "queued", "relevance": "high", "topics": [ "computer-use", "multi-agent", "rag", "reasoning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI", "cs.HC" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "14", "collection_queries": "rag-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-29762-do-recommendation-algorithms-work-when-users-are-llm-agents-a-case-study-on-molt.md", "title": "Do Recommendation Algorithms Work When Users Are LLM Agents? A Case Study on Moltbook", "type": "paper", "meta": { "type": "paper", "title": "Do Recommendation Algorithms Work When Users Are LLM Agents? A Case Study on Moltbook", "authors": "Daming Li, Simeng Han, Jialu Zhang", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.29762", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-29", "updated_at": "2026-06-29", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "rag" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.IR" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "14", "collection_queries": "ai-agent, llm-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-29771-clqt-a-closed-loop-cost-aware-strategy-consistent-benchmark-for-diagnostic-evalu.md", "title": "\"CLQT: A Closed-Loop, Cost-Aware, Strategy-Consistent Benchmark for Diagnostic Evaluation of LLM Portfolio-Management Agents\"", "type": "paper", "meta": { "type": "paper", "title": "\"CLQT: A Closed-Loop, Cost-Aware, Strategy-Consistent Benchmark for Diagnostic Evaluation of LLM Portfolio-Management Agents\"", "authors": "Bo Qu, Mingguang Chen", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.29771", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-29", "updated_at": "2026-06-29", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "memory", "reasoning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI", "cs.LG", "q-fin.CP", "q-fin.PM" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "19", "collection_queries": "llm-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-29774-analytic-concept-centric-memory-for-agentic-embodied-manipulation.md", "title": "Analytic Concept-Centric Memory for Agentic Embodied Manipulation", "type": "paper", "meta": { "type": "paper", "title": "Analytic Concept-Centric Memory for Agentic Embodied Manipulation", "authors": "Mingyang Sun, Xiujian Liang, Jiude Wei, Qichen He, Donglin Wang, Cewu Lu, Jianhua Sun", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.29774", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-29", "updated_at": "2026-06-29", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "embodied-agent", "memory", "planning", "rag", "reasoning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.RO" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "19", "collection_queries": "agent-memory" } }, { "collection": "papers", "path": "papers/items/2026-2606-29778-mandol-an-agglomerative-agent-memory-system-for-long-term-conversations.md", "title": "\"Mandol: An Agglomerative Agent Memory System for Long-Term Conversations\"", "type": "paper", "meta": { "type": "paper", "title": "\"Mandol: An Agglomerative Agent Memory System for Long-Term Conversations\"", "authors": "Yuhan Zhang, Zhiyuan Guo, Ziheng Zeng, Wei Wang, Wentao Wu, Lijie Xu", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.29778", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-29", "updated_at": "2026-06-29", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "memory", "rag" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.DB", "cs.AI", "cs.CL", "cs.IR" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "15", "collection_queries": "agent-memory, rag-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-29788-memleak-diagnosing-information-leaks-in-multimodal-agent-memory.md", "title": "\"MemLeak: Diagnosing Information Leaks in Multimodal Agent Memory\"", "type": "paper", "meta": { "type": "paper", "title": "\"MemLeak: Diagnosing Information Leaks in Multimodal Agent Memory\"", "authors": "Kuan Wang, Chao Zhang", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.29788", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-29", "updated_at": "2026-06-29", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "memory" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.LG" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "13", "collection_queries": "agent-memory, ai-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-29824-neural-procedural-memory-empowering-llm-agents-with-implicit-activation-steering.md", "title": "\"Neural Procedural Memory: Empowering LLM Agents with Implicit Activation Steering\"", "type": "paper", "meta": { "type": "paper", "title": "\"Neural Procedural Memory: Empowering LLM Agents with Implicit Activation Steering\"", "authors": "Chengfeng Zhao, Yuqiao Tan, Shizhu He, Yequan Wang, Jun Zhao, Kang Liu", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.29824", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-29", "updated_at": "2026-06-29", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "computer-use", "memory", "rag", "tool-use", "workflow-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CL", "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "24", "collection_queries": "agent-evaluation, agent-memory, autonomous-agent-llm, llm-agent, rag-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-29894-saber-math-automated-benchmark-for-information-retrieval-evaluation-in-mathemati.md", "title": "\"SABER-Math: Automated Benchmark for Information Retrieval Evaluation in Mathematics\"", "type": "paper", "meta": { "type": "paper", "title": "\"SABER-Math: Automated Benchmark for Information Retrieval Evaluation in Mathematics\"", "authors": "Nikolay Georgiev, Maria Drencheva, Kseniia Ibragimova, Ivo Petrov, Dimitar I. Dimitrov, Martin Vechev", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.29894", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-29", "updated_at": "2026-06-29", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "rag" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.IR", "cs.AI", "cs.CL", "cs.LG" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "14", "collection_queries": "agentic-ai" } }, { "collection": "papers", "path": "papers/items/2026-2606-29914-memdelta-controlled-baselines-and-hidden-confounds-in-agent-memory-evaluation.md", "title": "\"MemDelta: Controlled Baselines and Hidden Confounds in Agent Memory Evaluation\"", "type": "paper", "meta": { "type": "paper", "title": "\"MemDelta: Controlled Baselines and Hidden Confounds in Agent Memory Evaluation\"", "authors": "Kuan Wang", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.29914", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-29", "updated_at": "2026-06-29", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "memory", "rag" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CL", "cs.LG" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "17", "collection_queries": "agent-memory, rag-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-29932-saga-scene-aware-goal-evolving-agents-for-long-horizon-civrealm-strategy-plannin.md", "title": "\"SAGA: Scene-Aware, Goal-Evolving Agents for Long-Horizon CivRealm Strategy Planning\"", "type": "paper", "meta": { "type": "paper", "title": "\"SAGA: Scene-Aware, Goal-Evolving Agents for Long-Horizon CivRealm Strategy Planning\"", "authors": "Tianyu Jin, Shuo Chen, Yida Wang, Liuyu Xiang, Yingzhuo Liu, Zhiyao Jiang, Yexin Li, Zhaofeng He", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.29932", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-29", "updated_at": "2026-07-02", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "coding-agent", "multi-agent", "planning", "reasoning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "14", "collection_queries": "multi-agent-llm" } }, { "collection": "papers", "path": "papers/items/2026-2606-29957-swe-together-evaluating-coding-agents-in-interactive-user-sessions.md", "title": "\"SWE-Together: Evaluating Coding Agents in Interactive User Sessions\"", "type": "paper", "meta": { "type": "paper", "title": "\"SWE-Together: Evaluating Coding Agents in Interactive User Sessions\"", "authors": "Yifan Wu, Zhuokai Zhao, Songlin Li, Ho Hin Lee, Jiacheng Zhu, Shirley Wu, Tianhe Yu, Serena Li, et al.", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.29957", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-29", "updated_at": "2026-06-29", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "coding-agent", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.SE", "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "15", "collection_queries": "agent-evaluation, coding-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-29961-duomem-towards-capable-on-device-memory-agents-via-dual-space-distillation.md", "title": "\"DuoMem: Towards Capable On-Device Memory Agents via Dual-Space Distillation\"", "type": "paper", "meta": { "type": "paper", "title": "\"DuoMem: Towards Capable On-Device Memory Agents via Dual-Space Distillation\"", "authors": "Peyman Hosseini, Ondrej Bohdal, Ahmed Alajrami, Andrea Maracani, Ignacio Castro, Matthew Purver, Mete Ozay, Savas Ozkan, et al.", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.29961", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-29", "updated_at": "2026-06-29", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "embodied-agent", "memory" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.LG", "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "14", "collection_queries": "agent-memory" } }, { "collection": "papers", "path": "papers/items/2026-2606-30005-llm-agents-are-latent-context-managers-eliciting-self-managed-context-via-a-prop.md", "title": "\"LLM Agents Are Latent Context Managers: Eliciting Self-Managed Context via a Proprioceptive Dashboard\"", "type": "paper", "meta": { "type": "paper", "title": "\"LLM Agents Are Latent Context Managers: Eliciting Self-Managed Context via a Proprioceptive Dashboard\"", "authors": "Binyan Xu, Haitao Li, Kehuan Zhang", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.30005", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-29", "updated_at": "2026-07-05", "status": "queued", "relevance": "high", "topics": [ "memory", "planning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CL" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "14", "collection_queries": "llm-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-30111-automating-the-design-of-embodied-agent-architectures.md", "title": "Automating the Design of Embodied Agent Architectures", "type": "paper", "meta": { "type": "paper", "title": "Automating the Design of Embodied Agent Architectures", "authors": "Jian Zhou, Sihao Lin, Jin Li, Shuai Fu, Gengze Zhou, Qi Wu", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.30111", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-29", "updated_at": "2026-07-03", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "coding-agent", "embodied-agent", "memory", "planning", "reasoning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.RO", "cs.AI", "cs.LG" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "13", "collection_queries": "coding-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-30119-on-the-internet-nobody-knows-you-re-an-llm-bot-unmasking-web-agents-with-multi-l.md", "title": "\"On the Internet, Nobody Knows You're an LLM Bot: Unmasking Web Agents with Multi-Layer Fingerprinting\"", "type": "paper", "meta": { "type": "paper", "title": "\"On the Internet, Nobody Knows You're an LLM Bot: Unmasking Web Agents with Multi-Layer Fingerprinting\"", "authors": "Iliana Fayolle, Sihem Bouhenniche, Samuel Pélissier, Pierre Laperdrix, Clémentine Maurice, Walter Rudametkin", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.30119", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-29", "updated_at": "2026-06-29", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "computer-use", "embodied-agent", "rag", "tool-use", "workflow-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CR" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "14", "collection_queries": "web-gui-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-30185-dynamo-dynamic-skill-tool-evolution-for-vision-language-agents.md", "title": "\"Dynamo: Dynamic Skill-Tool Evolution for Vision-Language Agents\"", "type": "paper", "meta": { "type": "paper", "title": "\"Dynamo: Dynamic Skill-Tool Evolution for Vision-Language Agents\"", "authors": "Yutao Sun, Yanting Miao, Hao-Xuan Ma, Mengyu Zhou, Mingshuai Chen, Tiancheng Zhao, Dexin Wang, Lei Lv, et al.", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.30185", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-29", "updated_at": "2026-06-29", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "reasoning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "14", "collection_queries": "language-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-30251-taco-tool-augmented-credit-optimization-for-agentic-tool-use.md", "title": "\"TACO: Tool-Augmented Credit Optimization for Agentic Tool Use\"", "type": "paper", "meta": { "type": "paper", "title": "\"TACO: Tool-Augmented Credit Optimization for Agentic Tool Use\"", "authors": "Mingkuan Feng, Jinyang Wu, Hao Gu, Fangrui Lv, Ruihan Jin, Chuyuan Zhang, Zhengqi Wen, Jianhua Tao", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.30251", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-29", "updated_at": "2026-06-29", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "computer-use", "reasoning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.MA" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "16", "collection_queries": "tool-use" } }, { "collection": "papers", "path": "papers/items/2026-2606-30259-multi-agentic-system-leveraging-open-source-llms-to-mitigate-disinformation-thre.md", "title": "Multi-Agentic System Leveraging Open-Source LLMs to Mitigate Disinformation Threats", "type": "paper", "meta": { "type": "paper", "title": "Multi-Agentic System Leveraging Open-Source LLMs to Mitigate Disinformation Threats", "authors": "Sebastian Kula, Martin Tamajka", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.30259", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-29", "updated_at": "2026-06-29", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "multi-agent", "rag" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CL" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "17", "collection_queries": "multi-agent-llm" } }, { "collection": "papers", "path": "papers/items/2026-2606-30266-towards-continual-motion-language-agents-lora-variants-for-incremental-motion-un.md", "title": "\"Towards Continual Motion-Language Agents: LoRA Variants for Incremental Motion Understanding and Generation\"", "type": "paper", "meta": { "type": "paper", "title": "\"Towards Continual Motion-Language Agents: LoRA Variants for Incremental Motion Understanding and Generation\"", "authors": "Bertram Taetz, Hugo Albuquerque Cosme da Silva, Gabriele Bleser-Taetz", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.30266", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-29", "updated_at": "2026-06-29", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.LG", "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "14", "collection_queries": "autonomous-agent-llm, language-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-30294-rehearsed-multi-agent-live-product-demonstrations-with-real-time-voice-question-.md", "title": "Rehearsed Multi-Agent Live Product Demonstrations with Real-Time Voice Question Answering", "type": "paper", "meta": { "type": "paper", "title": "Rehearsed Multi-Agent Live Product Demonstrations with Real-Time Voice Question Answering", "authors": "Rahul Khedar, Mayank Malhotra, Avinash Karn, Mouli V, Prakhar Mehrotra", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.30294", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-29", "updated_at": "2026-06-29", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "coding-agent", "computer-use", "multi-agent", "rag", "tool-use", "workflow-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI", "cs.HC", "cs.SE" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "16", "collection_queries": "web-gui-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-30383-whose-side-is-your-agent-on-multi-party-principal-loyalty-in-llm-agents.md", "title": "Whose Side Is Your Agent On? Multi-Party Principal Loyalty in LLM Agents", "type": "paper", "meta": { "type": "paper", "title": "Whose Side Is Your Agent On? Multi-Party Principal Loyalty in LLM Agents", "authors": "Bojie Li, Noah Shi", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.30383", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-29", "updated_at": "2026-06-29", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "15", "collection_queries": "llm-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-30454-collective-cooperation-without-individual-fidelity-in-llm-agents.md", "title": "Collective cooperation without individual fidelity in LLM agents", "type": "paper", "meta": { "type": "paper", "title": "Collective cooperation without individual fidelity in LLM agents", "authors": "Henrique Ferraz de Arruda, Carlos Gracia Lázaro, Alberto Aleta, Yamir Moreno", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.30454", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-29", "updated_at": "2026-06-29", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "tool-use", "world-model" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "physics.soc-ph", "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "14", "collection_queries": "llm-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-30524-the-illusion-of-agentic-complexity-in-readme-md-generation-evaluating-single-age.md", "title": "\"The Illusion of Agentic Complexity in README.md Generation: Evaluating Single-Agent vs. Multi-Agent RAG Systems\"", "type": "paper", "meta": { "type": "paper", "title": "\"The Illusion of Agentic Complexity in README.md Generation: Evaluating Single-Agent vs. Multi-Agent RAG Systems\"", "authors": "Abu Saleh, Tesfay Welegebreal Tesfay, Phuong T. Nguyen, Juri Di Rocco, Muhammad Umar Zeshan, Davide Di Ruscio", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.30524", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-29", "updated_at": "2026-07-01", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "coding-agent", "computer-use", "multi-agent", "planning", "rag" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.SE" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "21", "collection_queries": "multi-agent-llm, rag-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-30546-mas-lab-a-specification-driven-validation-framework-for-reliable-multi-agent-sys.md", "title": "\"MAS-Lab: A Specification-Driven Validation Framework for Reliable Multi-Agent Systems\"", "type": "paper", "meta": { "type": "paper", "title": "\"MAS-Lab: A Specification-Driven Validation Framework for Reliable Multi-Agent Systems\"", "authors": "Jordan Augé, Giovanna Carofiglio, Giulio Grassi, Jacques Samain", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.30546", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-29", "updated_at": "2026-06-29", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "memory", "multi-agent", "tool-use", "workflow-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.MA" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "17", "collection_queries": "multi-agent-llm" } }, { "collection": "papers", "path": "papers/items/2026-2606-30555-linguistic-firewall-geometry-as-defense-in-multi-agent-systems-routing.md", "title": "\"Linguistic Firewall: Geometry as Defense in Multi-Agent Systems Routing\"", "type": "paper", "meta": { "type": "paper", "title": "\"Linguistic Firewall: Geometry as Defense in Multi-Agent Systems Routing\"", "authors": "Dvir Alsheich, Adar Peleg, Ben Hagag, Rom Himelstein, Amit Levi, Avi Mendelson", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.30555", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-29", "updated_at": "2026-07-05", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "computer-use", "multi-agent", "tool-use", "workflow-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI", "cs.MA" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "17", "collection_queries": "multi-agent-llm" } }, { "collection": "papers", "path": "papers/items/2026-2606-30560-tracelab-characterizing-coding-agent-workloads-for-llm-serving.md", "title": "\"TraceLab: Characterizing Coding Agent Workloads for LLM Serving\"", "type": "paper", "meta": { "type": "paper", "title": "\"TraceLab: Characterizing Coding Agent Workloads for LLM Serving\"", "authors": "Kan Zhu, Mathew Jacob, Chenxi Ma, Yi Pan, Stephanie Wang, Arvind Krishnamurthy, Baris Kasikci", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.30560", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-29", "updated_at": "2026-06-30", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "coding-agent", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.LG", "cs.AI", "cs.PF" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "15", "collection_queries": "coding-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-30566-forensic-trajectory-signatures-for-agent-memory-poisoning-detection.md", "title": "Forensic Trajectory Signatures for Agent Memory Poisoning Detection", "type": "paper", "meta": { "type": "paper", "title": "Forensic Trajectory Signatures for Agent Memory Poisoning Detection", "authors": "Jun Wen Leong", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.30566", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-29", "updated_at": "2026-06-29", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "computer-use", "memory", "rag", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CR", "cs.LG" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "17", "collection_queries": "agent-memory, llm-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-30573-swe-interact-reimagining-swe-benchmarks-as-user-driven-long-horizon-coding-sessi.md", "title": "\"SWE-INTERACT: Reimagining SWE Benchmarks as User-Driven Long-Horizon Coding Sessions\"", "type": "paper", "meta": { "type": "paper", "title": "\"SWE-INTERACT: Reimagining SWE Benchmarks as User-Driven Long-Horizon Coding Sessions\"", "authors": "Mohit Raghavendra, Anisha Gunjal, Aakash Sabharwal, Yunzhong He", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.30573", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-29", "updated_at": "2026-06-29", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "coding-agent", "computer-use", "planning", "tool-use", "workflow-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.LG" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "15", "collection_queries": "coding-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-30602-mesa-prioritizing-vulnerable-communication-channels-for-securing-multi-agent-sys.md", "title": "\"MESA: Prioritizing Vulnerable Communication Channels for Securing Multi-Agent Systems\"", "type": "paper", "meta": { "type": "paper", "title": "\"MESA: Prioritizing Vulnerable Communication Channels for Securing Multi-Agent Systems\"", "authors": "Kunyang Li, Kyle Domico, Jonathan Gregory, Patrick McDaniel", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.30602", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-29", "updated_at": "2026-06-29", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "multi-agent", "workflow-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CR", "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "14", "collection_queries": "multi-agent-llm" } }, { "collection": "papers", "path": "papers/items/2026-2606-30616-scaling-the-horizon-not-the-parameters-reaching-trillion-parameter-performance-w.md", "title": "\"Scaling the Horizon, Not the Parameters: Reaching Trillion-Parameter Performance with a 35B Agent\"", "type": "paper", "meta": { "type": "paper", "title": "\"Scaling the Horizon, Not the Parameters: Reaching Trillion-Parameter Performance with a 35B Agent\"", "authors": "Lei Bai, Zongsheng Cao, Yang Chen, Zhiyao Cui, Shangheng Du, Yue Fan, Shiyang Feng, Zijie Guo, et al.", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.30616", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-29", "updated_at": "2026-06-29", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "planning", "rag", "reasoning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CL" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "13", "collection_queries": "agent-evaluation" } }, { "collection": "papers", "path": "papers/items/2026-2606-30639-self-evolving-world-models-for-llm-agent-planning.md", "title": "Self-Evolving World Models for LLM Agent Planning", "type": "paper", "meta": { "type": "paper", "title": "Self-Evolving World Models for LLM Agent Planning", "authors": "Xuan Zhang, Wenxuan Zhang, See-Kiong Ng, Yang Deng", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.30639", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-29", "updated_at": "2026-06-29", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "memory", "planning", "rag", "reasoning", "tool-use", "world-model" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI", "cs.CL" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "19", "collection_queries": "llm-agent, planning-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-30697-lumos-a-semantic-operating-system-layer-for-accessibility-grounded-ai-agents.md", "title": "\"LUMOS: A Semantic Operating-System Layer for Accessibility-Grounded AI Agents\"", "type": "paper", "meta": { "type": "paper", "title": "\"LUMOS: A Semantic Operating-System Layer for Accessibility-Grounded AI Agents\"", "authors": "Yogeswar Reddy Thota", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.30697", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-29", "updated_at": "2026-06-29", "status": "queued", "relevance": "high", "topics": [ "computer-use", "rag", "tool-use", "workflow-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.OS", "cs.AI", "cs.CV" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "15", "collection_queries": "ai-agent, web-gui-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-30755-understanding-and-evaluating-claw-like-agent-security-through-a-computer-systems.md", "title": "Understanding and Evaluating Claw-like Agent Security Through a Computer-Systems Lens", "type": "paper", "meta": { "type": "paper", "title": "Understanding and Evaluating Claw-like Agent Security Through a Computer-Systems Lens", "authors": "Peizhi Niu, Wenjie Qu, Shangding Gu, Tianneng Shi, Yuankai Li, Ahmad Tawaha, Hend Alzahrani, Vincent Siu, et al.", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.30755", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-29", "updated_at": "2026-06-29", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CR", "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "14", "collection_queries": "agent-safety, ai-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-30840-contrastive-reflection-for-iterative-prompt-optimization.md", "title": "Contrastive Reflection for Iterative Prompt Optimization", "type": "paper", "meta": { "type": "paper", "title": "Contrastive Reflection for Iterative Prompt Optimization", "authors": "Derek Koh, Jinghui Mo, Benjamin H. Le, Jiening Zhan, Baofen Zheng, Kevin Bevis, Nathaniel C. Owen, Lauren Elizabeth Charney, et al.", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.30840", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-29", "updated_at": "2026-06-29", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "computer-use", "rag", "reasoning", "workflow-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "14", "collection_queries": "ai-agent, llm-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-30877-a-systematic-approach-to-multi-agent-ai-from-advanced-regulatory-control-theory-.md", "title": "\"A Systematic Approach to Multi-Agent AI from Advanced Regulatory Control Theory: Safe and Auditable LLM Operator Agents for Process Control\"", "type": "paper", "meta": { "type": "paper", "title": "\"A Systematic Approach to Multi-Agent AI from Advanced Regulatory Control Theory: Safe and Auditable LLM Operator Agents for Process Control\"", "authors": "Idelfonso B. R. Nogueira, Sigurd Skogestad", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.30877", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-29", "updated_at": "2026-06-29", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "multi-agent", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "eess.SY", "cs.LG" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "15", "collection_queries": "agentic-ai, multi-agent-llm" } }, { "collection": "papers", "path": "papers/items/2026-2606-30887-training-therapeutic-judges-and-multi-agent-systems-for-human-aligned-mental-hea.md", "title": "Training Therapeutic Judges and Multi-Agent Systems for Human-Aligned Mental Health Support", "type": "paper", "meta": { "type": "paper", "title": "Training Therapeutic Judges and Multi-Agent Systems for Human-Aligned Mental Health Support", "authors": "Mizanur Rahman, Abeer Badawi, Elahe Rahimi, Laleh Seyyed-Kalantari, Frank Rudzicz, Enamul Hoque, Elham Dolatabadi", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.30887", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-29", "updated_at": "2026-06-29", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "multi-agent", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CL", "cs.AI", "cs.MA" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "15", "collection_queries": "multi-agent-llm" } }, { "collection": "papers", "path": "papers/items/2026-2606-30906-investigating-multi-agent-deliberation-in-law.md", "title": "Investigating Multi-Agent Deliberation in Law", "type": "paper", "meta": { "type": "paper", "title": "Investigating Multi-Agent Deliberation in Law", "authors": "Cor Steging, Ludi van Leeuwen, Tadeusz Zbiegień", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.30906", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-29", "updated_at": "2026-06-29", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "multi-agent", "reasoning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "20", "collection_queries": "agentic-ai, ai-agent, multi-agent-llm" } }, { "collection": "papers", "path": "papers/items/2026-2606-30949-agrefactor-self-evolving-agentic-workflow-for-hls-compatibility-and-performance.md", "title": "\"AgRefactor: Self-Evolving Agentic Workflow for HLS Compatibility and Performance\"", "type": "paper", "meta": { "type": "paper", "title": "\"AgRefactor: Self-Evolving Agentic Workflow for HLS Compatibility and Performance\"", "authors": "Yang Zou, Zijian Ding, Yizhou Sun, Jason Cong", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.30949", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-29", "updated_at": "2026-06-29", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "memory", "multi-agent", "rag", "tool-use", "workflow-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI", "cs.AR" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "20", "collection_queries": "agentic-ai, multi-agent-llm" } }, { "collection": "papers", "path": "papers/items/2026-2606-30970-behavioral-governance-for-autonomous-ai-agents-the-agentbound-framework.md", "title": "\"Behavioral Governance for Autonomous AI Agents: The AgentBound Framework\"", "type": "paper", "meta": { "type": "paper", "title": "\"Behavioral Governance for Autonomous AI Agents: The AgentBound Framework\"", "authors": "Anuj Kaul, Qianlong Lan, Pranay Gupta", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.30970", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-29", "updated_at": "2026-07-01", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "tool-use", "workflow-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "13", "collection_queries": "ai-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-30986-the-organizational-behavior-of-agentic-ai-collective-intelligence-in-human-agent.md", "title": "\"The Organizational Behavior of Agentic AI: Collective Intelligence in Human-Agent Workflows\"", "type": "paper", "meta": { "type": "paper", "title": "\"The Organizational Behavior of Agentic AI: Collective Intelligence in Human-Agent Workflows\"", "authors": "Canhui Liu", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.30986", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-29", "updated_at": "2026-06-29", "status": "queued", "relevance": "high", "topics": [ "memory", "planning", "tool-use", "workflow-agent", "world-model" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CY", "cs.HC", "cs.MA", "econ.GN" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "19", "collection_queries": "agentic-ai, llm-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-31046-openlife-toward-open-world-artificial-life-with-autonomous-llm-agents.md", "title": "\"OpenLife: Toward Open-World Artificial Life with Autonomous LLM Agents\"", "type": "paper", "meta": { "type": "paper", "title": "\"OpenLife: Toward Open-World Artificial Life with Autonomous LLM Agents\"", "authors": "Atsushi Masumori, Itsuki Doi, Norihiro Maruyama, Ryosuke Takata, Takashi Ikegami", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.31046", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-30", "updated_at": "2026-06-30", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "memory", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "17", "collection_queries": "llm-agent, tool-use" } }, { "collection": "papers", "path": "papers/items/2026-2606-31073-multiuav-plat-an-llm-oriented-platform-benchmark-and-framework-for-multi-uav-col.md", "title": "\"MultiUAV-Plat: An LLM-Oriented Platform, Benchmark and Framework for Multi-UAV Collaborative Task Planning\"", "type": "paper", "meta": { "type": "paper", "title": "\"MultiUAV-Plat: An LLM-Oriented Platform, Benchmark and Framework for Multi-UAV Collaborative Task Planning\"", "authors": "Sheng Zhang, Qinglin Li, Yuechao Zang, Xueqin Huang, Yijia Fu, Cheng Zhu", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.31073", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-30", "updated_at": "2026-06-30", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "embodied-agent", "memory", "multi-agent", "planning", "rag", "tool-use", "world-model" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI", "cs.MA", "cs.RO" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "20", "collection_queries": "agent-evaluation, llm-agent, planning-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-31085-ddiagents-mechanism-conditioned-context-flow-for-drug-drug-interaction-predictio.md", "title": "\"DDIAgents: Mechanism-Conditioned Context Flow for Drug-Drug Interaction Prediction\"", "type": "paper", "meta": { "type": "paper", "title": "\"DDIAgents: Mechanism-Conditioned Context Flow for Drug-Drug Interaction Prediction\"", "authors": "Zhenqian Shen, Yu Liu, Xiaoyi Fu, Quanming Yao", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.31085", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-30", "updated_at": "2026-06-30", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "multi-agent", "planning", "reasoning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "15", "collection_queries": "multi-agent-llm" } }, { "collection": "papers", "path": "papers/items/2026-2606-31134-beyond-the-library-an-agentic-framework-for-autoformalizing-research-mathematics.md", "title": "\"Beyond the Library: An Agentic Framework for Autoformalizing Research Mathematics\"", "type": "paper", "meta": { "type": "paper", "title": "\"Beyond the Library: An Agentic Framework for Autoformalizing Research Mathematics\"", "authors": "Arshia Soltani Moakhar, Iman Gholami, Max Springer, Mahdi JafariRaviz, MohammadTaghi Hajiaghayi", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.31134", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-30", "updated_at": "2026-07-01", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "coding-agent", "multi-agent", "rag", "reasoning" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "16", "collection_queries": "multi-agent-llm" } }, { "collection": "papers", "path": "papers/items/2026-2606-31174-clawarena-team-benchmarking-subagent-orchestration-and-dynamic-workflows-in-lang.md", "title": "\"ClawArena-Team: Benchmarking Subagent Orchestration and Dynamic Workflows in Language-Model Agents\"", "type": "paper", "meta": { "type": "paper", "title": "\"ClawArena-Team: Benchmarking Subagent Orchestration and Dynamic Workflows in Language-Model Agents\"", "authors": "Kaiwen Xiong, Haonian Ji, Shi Qiu, Zeyu Zheng, Cihang Xie, Xinyu Ye, Huaxiu Yao", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.31174", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-30", "updated_at": "2026-07-02", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "multi-agent", "tool-use", "workflow-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "17", "collection_queries": "llm-agent, multi-agent-llm" } }, { "collection": "papers", "path": "papers/items/2026-2606-31179-healthagentbench-a-unified-benchmark-suite-of-realistic-agentic-healthcare-envir.md", "title": "\"HealthAgentBench: A Unified Benchmark Suite of Realistic Agentic Healthcare Environments for Challenging Frontier AI Agents\"", "type": "paper", "meta": { "type": "paper", "title": "\"HealthAgentBench: A Unified Benchmark Suite of Realistic Agentic Healthcare Environments for Challenging Frontier AI Agents\"", "authors": "Qianchu Liu, Sheng Zhang, Guanghui Qin, Jeya Maria Jose Valanarasu, Maximilian Rokuss, Mingyu Lu, Timothy Ossowski, Juan Manuel Zambrano Chaves, et al.", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.31179", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-30", "updated_at": "2026-06-30", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "planning", "reasoning", "workflow-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI", "cs.CL", "cs.CV" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "20", "collection_queries": "agent-evaluation, ai-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-31200-agentic-rag-vlm-affordance-aware-retrieval-augmented-generation-with-self-reflec.md", "title": "\"Agentic RAG-VLM: Affordance-Aware Retrieval-Augmented Generation with Self-Reflective Planning for Robotic Grasping\"", "type": "paper", "meta": { "type": "paper", "title": "\"Agentic RAG-VLM: Affordance-Aware Retrieval-Augmented Generation with Self-Reflective Planning for Robotic Grasping\"", "authors": "Tao Chen, Lizheng Liu, Jiaxu Wang, Ziyue Jiang, Ruiqi Tian, JiGuang Huo, Zhongxue Gan", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.31200", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-30", "updated_at": "2026-06-30", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "embodied-agent", "planning", "rag", "reasoning" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "16", "collection_queries": "rag-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-31209-long-term-traffic-simulation-via-structured-autoregressive-modeling.md", "title": "Long-term Traffic Simulation via Structured Autoregressive Modeling", "type": "paper", "meta": { "type": "paper", "title": "Long-term Traffic Simulation via Structured Autoregressive Modeling", "authors": "Lingyu Xiao, Zexin Feng, Xintao Yan", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.31209", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-30", "updated_at": "2026-06-30", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "memory", "multi-agent", "planning", "rag", "tool-use", "world-model" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI", "cs.RO" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "15", "collection_queries": "multi-agent-llm" } }, { "collection": "papers", "path": "papers/items/2026-2606-31227-securing-the-ai-agent-a-unified-framework-for-multi-layer-agent-red-teaming.md", "title": "\"Securing the AI Agent: A Unified Framework for Multi-Layer Agent Red Teaming\"", "type": "paper", "meta": { "type": "paper", "title": "\"Securing the AI Agent: A Unified Framework for Multi-Layer Agent Red Teaming\"", "authors": "Yong Yang, Xing Zheng, Huiyu Wu, Huangsheng Cheng, Xiaorong Shi, Jing Guo, Bo Yang, Yi Zhou, et al.", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.31227", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-30", "updated_at": "2026-06-30", "status": "queued", "relevance": "high", "topics": [ "agent-safety", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CR" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "14", "collection_queries": "agent-safety, ai-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-31229-agentic-ideation-sample-efficient-agentic-trajectories-synthesis-for-scientific-.md", "title": "\"Agentic-Ideation: Sample Efficient Agentic Trajectories Synthesis for Scientific Ideation Agents\"", "type": "paper", "meta": { "type": "paper", "title": "\"Agentic-Ideation: Sample Efficient Agentic Trajectories Synthesis for Scientific Ideation Agents\"", "authors": "Keyu Zhao, Lingyan Kong, Fengli Xu, Yong Li", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.31229", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-30", "updated_at": "2026-06-30", "status": "queued", "relevance": "high", "topics": [ "computer-use", "multi-agent", "rag", "reasoning", "tool-use", "workflow-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "18", "collection_queries": "agentic-ai, multi-agent-llm" } }, { "collection": "papers", "path": "papers/items/2026-2606-31252-embodied-cad-solver-grounded-llm-agents-for-parametric-b-rep-assembly-modeling.md", "title": "\"Embodied CAD: Solver-Grounded LLM Agents for Parametric B-Rep Assembly Modeling\"", "type": "paper", "meta": { "type": "paper", "title": "\"Embodied CAD: Solver-Grounded LLM Agents for Parametric B-Rep Assembly Modeling\"", "authors": "Fumin Liu, Haoyu Zhou, Fei Hao, Lin Yang", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.31252", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-30", "updated_at": "2026-06-30", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "embodied-agent", "planning", "tool-use", "workflow-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "17", "collection_queries": "llm-agent, planning-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-31314-a-novel-method-for-differential-algebraic-dynamic-model-discovery-in-power-syste.md", "title": "\"A Novel Method for Differential-Algebraic Dynamic Model Discovery in Power Systems: An LLM-Based Multi-Agent Collaborative Framework\"", "type": "paper", "meta": { "type": "paper", "title": "\"A Novel Method for Differential-Algebraic Dynamic Model Discovery in Power Systems: An LLM-Based Multi-Agent Collaborative Framework\"", "authors": "Xinming Wang, Fan Tang, Yingli Wei, Yakun He, Zhe Liu, Ping Jiang, Haoyu Wu, Zihan Guo, et al.", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.31314", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-30", "updated_at": "2026-06-30", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "computer-use", "multi-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "eess.SY" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "14", "collection_queries": "multi-agent-llm" } }, { "collection": "papers", "path": "papers/items/2026-2606-31339-verification-gated-agentic-mission-state-governance-for-intelligent-industrial-m.md", "title": "Verification-Gated Agentic Mission-State Governance for Intelligent Industrial Multi-Robot Systems", "type": "paper", "meta": { "type": "paper", "title": "Verification-Gated Agentic Mission-State Governance for Intelligent Industrial Multi-Robot Systems", "authors": "Guoqin Tang, Qingxuan Jia, Yichen Tan, Zeyuan Huang, Ning Ji, Gang Chen", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.31339", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-30", "updated_at": "2026-06-30", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "embodied-agent", "planning", "rag", "reasoning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.RO" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "16", "collection_queries": "agentic-ai" } }, { "collection": "papers", "path": "papers/items/2026-2606-31410-xiaomi-gui-0-technical-report.md", "title": "Xiaomi-GUI-0 Technical Report", "type": "paper", "meta": { "type": "paper", "title": "Xiaomi-GUI-0 Technical Report", "authors": "Wanxia Cao, Chengzhen Duan, Pei Fu, Pengzhi Gao, Niu Lian, Fazhan Liu, Hui Liu, Heng Qu, et al.", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.31410", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-30", "updated_at": "2026-07-01", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "computer-use", "embodied-agent", "memory", "planning", "reasoning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "18", "collection_queries": "web-gui-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-31471-think-while-you-map-asynchronous-vision-language-agents-for-incremental-3d-scene.md", "title": "\"Think While You Map: Asynchronous Vision-Language Agents for Incremental 3D Scene Graphs\"", "type": "paper", "meta": { "type": "paper", "title": "\"Think While You Map: Asynchronous Vision-Language Agents for Incremental 3D Scene Graphs\"", "authors": "Deniz Bickici, Michael Pabst, Shohei Mori, Dieter Schmalstieg", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.31471", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-30", "updated_at": "2026-06-30", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "rag" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CV" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "13", "collection_queries": "language-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-31612-what-memory-do-gui-agents-really-need-from-passive-records-to-active-task-drivin.md", "title": "What Memory Do GUI Agents Really Need? From Passive Records to Active Task-Driving States", "type": "paper", "meta": { "type": "paper", "title": "What Memory Do GUI Agents Really Need? From Passive Records to Active Task-Driving States", "authors": "Chen Liu, Ling Chen, Hanzhang Zhou, Xu Zhang, Quyu Kong, Panrong Tong, Wenhao Wang, Xin Yu, et al.", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.31612", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-30", "updated_at": "2026-07-02", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "computer-use", "memory", "planning", "rag", "tool-use", "workflow-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CV" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "20", "collection_queries": "agent-memory, web-gui-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-31635-a-tutorial-on-autonomous-fault-tolerant-control-using-knowledge-grounded-llm-age.md", "title": "A Tutorial on Autonomous Fault-Tolerant Control Using Knowledge-Grounded LLM Agents", "type": "paper", "meta": { "type": "paper", "title": "A Tutorial on Autonomous Fault-Tolerant Control Using Knowledge-Grounded LLM Agents", "authors": "Javal Vyas, Milapji Singh Gill, Artan Markaj, Felix Gehlhoff, Mehmet Mercangöz", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.31635", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-30", "updated_at": "2026-06-30", "status": "queued", "relevance": "high", "topics": [ "agent-safety", "planning", "tool-use", "world-model" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "eess.SY", "cs.AI", "cs.MA" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "14", "collection_queries": "llm-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-31639-a-lifecycle-and-application-stack-survey-of-large-language-model-vulnerabilities.md", "title": "\"A Lifecycle and Application-Stack Survey of Large Language Model Vulnerabilities: Attacks, Risks, Defenses, and Open Problems\"", "type": "paper", "meta": { "type": "paper", "title": "\"A Lifecycle and Application-Stack Survey of Large Language Model Vulnerabilities: Attacks, Risks, Defenses, and Open Problems\"", "authors": "Seyed Bagher Hashemi Natanzi, Bo Tang", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.31639", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-30", "updated_at": "2026-06-30", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "coding-agent", "embodied-agent", "memory", "planning", "rag", "tool-use", "workflow-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CR", "cs.AI", "cs.GT", "cs.LO" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "17", "collection_queries": "agent-evaluation, autonomous-agent-llm" } }, { "collection": "papers", "path": "papers/items/2026-2606-31648-think-in-english-answer-in-korean-efficient-adaptation-of-multilingual-tool-usin.md", "title": "\"Think in English, Answer in Korean: Efficient Adaptation of Multilingual Tool-Using Agents\"", "type": "paper", "meta": { "type": "paper", "title": "\"Think in English, Answer in Korean: Efficient Adaptation of Multilingual Tool-Using Agents\"", "authors": "Utsav Garg, Sungjin Hong, Jason Jung, Justin Lee, Shaan Desai, Joon Hee Kim, Anirudh Shrinivason, Edmond Wen, et al.", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.31648", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-30", "updated_at": "2026-06-30", "status": "queued", "relevance": "high", "topics": [ "memory", "multi-agent", "reasoning", "tool-use", "workflow-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI", "cs.LG" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "14", "collection_queries": "agentic-ai, function-calling, tool-use" } }, { "collection": "papers", "path": "papers/items/2026-2606-31650-echo-prune-to-act-trace-to-learn-with-selective-turn-memory-in-agentic-rl.md", "title": "\"ECHO: Prune to act, trace to learn with selective turn memory in agentic RL\"", "type": "paper", "meta": { "type": "paper", "title": "\"ECHO: Prune to act, trace to learn with selective turn memory in agentic RL\"", "authors": "Zijun Xie, Binbin Zheng, Enlei Gong, Jihua Liu, Yuyang You, Lingfeng Liu, Jiayao Tang, Guanqun Zhao, et al.", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.31650", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-30", "updated_at": "2026-06-30", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "memory", "planning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.LG", "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "17", "collection_queries": "language-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-31665-forecastagentsearch-towards-a-multi-expert-agent-search-system-for-geopolitical-.md", "title": "\"ForecastAgentSearch: Towards a Multi-Expert Agent Search System for Geopolitical Event Forecasting\"", "type": "paper", "meta": { "type": "paper", "title": "\"ForecastAgentSearch: Towards a Multi-Expert Agent Search System for Geopolitical Event Forecasting\"", "authors": "Miaomiao Cai, He Chang, Yunshan Ma, See-kiong Ng", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.31665", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-30", "updated_at": "2026-06-30", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "multi-agent", "planning", "rag" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.MA" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "13", "collection_queries": "multi-agent-llm" } }, { "collection": "papers", "path": "papers/items/2026-2606-31693-shopx-a-foundation-model-for-intent-to-item-fulfillment-in-agentic-shopping.md", "title": "\"ShopX: A Foundation Model for Intent-to-Item Fulfillment in Agentic Shopping\"", "type": "paper", "meta": { "type": "paper", "title": "\"ShopX: A Foundation Model for Intent-to-Item Fulfillment in Agentic Shopping\"", "authors": "Jiacheng Chen, Tao Zhang, Manxi Lin, Dunxian Huang, Teng Shi, Honghao Fu, Mengyan Li, Xinming Zhang, et al.", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.31693", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-30", "updated_at": "2026-06-30", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "planning", "rag", "tool-use", "workflow-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.IR", "cs.AI", "cs.CL" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "17", "collection_queries": "llm-agent, planning-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-31744-a-conversational-agentic-interface-to-physics-based-household-digital-twins-for-.md", "title": "A Conversational Agentic Interface to Physics-Based Household Digital Twins for Residential Energy Decision Support", "type": "paper", "meta": { "type": "paper", "title": "A Conversational Agentic Interface to Physics-Based Household Digital Twins for Residential Energy Decision Support", "authors": "Costas Mylonas, Titos Georgoulakis, Magda Foti", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.31744", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-30", "updated_at": "2026-06-30", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "planning", "rag", "tool-use", "world-model" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "eess.SY" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "15", "collection_queries": "llm-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-31767-jeto-bench-a-reproducible-benchmark-for-execution-time-improvement-patches-in-ja.md", "title": "\"JETO-Bench: A Reproducible Benchmark for Execution Time Improvement Patches in Java\"", "type": "paper", "meta": { "type": "paper", "title": "\"JETO-Bench: A Reproducible Benchmark for Execution Time Improvement Patches in Java\"", "authors": "Khashayar Etemadi, Zhendong Su", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.31767", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-30", "updated_at": "2026-06-30", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "coding-agent", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.SE" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "13", "collection_queries": "coding-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-31831-an-agentic-ai-framework-to-accelerate-scientific-discovery-in-plant-phenotyping.md", "title": "An Agentic AI Framework to Accelerate Scientific Discovery in Plant Phenotyping", "type": "paper", "meta": { "type": "paper", "title": "An Agentic AI Framework to Accelerate Scientific Discovery in Plant Phenotyping", "authors": "Renan Souza, Daniel Rosendo, Kelsey Carter, John Lagergren, Frédéric Suter, Shelaine L. Curd, Gerald A. Tuskan, Rafael Ferreira da Silva, et al.", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.31831", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-30", "updated_at": "2026-06-30", "status": "queued", "relevance": "high", "topics": [ "agent-safety", "planning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "13", "collection_queries": "agentic-ai, ai-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-31916-theory-of-mind-and-persuasion-beyond-conversation-assessing-the-capacity-of-llms.md", "title": "\"Theory of Mind and Persuasion Beyond Conversation: Assessing the Capacity of LLMs to Induce Belief States via Planning and Action\"", "type": "paper", "meta": { "type": "paper", "title": "\"Theory of Mind and Persuasion Beyond Conversation: Assessing the Capacity of LLMs to Induce Belief States via Planning and Action\"", "authors": "Ben Slater, Matteo G. Mecattaf, Lucy G. Cheke, John Burden, Winnie Street", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.31916", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-30", "updated_at": "2026-06-30", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "planning", "reasoning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CL" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "13", "collection_queries": "agent-evaluation" } }, { "collection": "papers", "path": "papers/items/2026-2606-31980-digitalcoach-communication-and-grounding-gaps-in-human-and-agentic-computer-use-.md", "title": "\"DigitalCoach: Communication and Grounding Gaps in Human and Agentic Computer Use Coaching\"", "type": "paper", "meta": { "type": "paper", "title": "\"DigitalCoach: Communication and Grounding Gaps in Human and Agentic Computer Use Coaching\"", "authors": "Meng Chen, Anya Ji, Tsung-Han Wu, Tobias Maringgele, David M. Chan, Alane Suhr, Amy Pavel", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.31980", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-30", "updated_at": "2026-06-30", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "computer-use", "planning", "rag" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CL", "cs.AI", "cs.HC" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "16", "collection_queries": "web-gui-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-32025-generative-skill-composition-for-llm-agents.md", "title": "Generative Skill Composition for LLM Agents", "type": "paper", "meta": { "type": "paper", "title": "Generative Skill Composition for LLM Agents", "authors": "Xinyu Zhao, Zhen Tan, Vaishnav Tadiparthi, Nakul Agarwal, Kwonjoon Lee, Ehsan Moradi Pari, Hossein Nourkhiz Mahjoub, Tianlong Chen", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.32025", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-30", "updated_at": "2026-06-30", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "coding-agent", "planning", "rag", "reasoning" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CL" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "18", "collection_queries": "coding-agent, llm-agent, planning-agent" } }, { "collection": "papers", "path": "papers/items/2026-2606-32034-qval-cheaply-evaluating-dense-supervision-signals-for-long-horizon-llm-agents.md", "title": "\"QVal: Cheaply Evaluating Dense Supervision Signals for Long-Horizon LLM Agents\"", "type": "paper", "meta": { "type": "paper", "title": "\"QVal: Cheaply Evaluating Dense Supervision Signals for Long-Horizon LLM Agents\"", "authors": "Sergio Hernández-Gutiérrez, Matteo Merler, Ilze Amanda Auzina, Joschka Strüber, Ameya Prabhu, Matthias Bethge", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2606.32034", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-30", "updated_at": "2026-06-30", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "computer-use", "planning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.LG", "cs.AI", "cs.CL" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "16", "collection_queries": "llm-agent" } }, { "collection": "papers", "path": "papers/items/2026-2607-00038-stop-hand-holding-your-coding-agent-engineering-the-loops-that-replace-step-by-s.md", "title": "\"Stop Hand-Holding Your Coding Agent: Engineering the Loops that Replace Step-by-Step Prompting\"", "type": "paper", "meta": { "type": "paper", "title": "\"Stop Hand-Holding Your Coding Agent: Engineering the Loops that Replace Step-by-Step Prompting\"", "authors": "Sandeco Macedo", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2607.00038", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-28", "updated_at": "2026-06-28", "status": "queued", "relevance": "high", "topics": [ "agent-safety", "coding-agent", "computer-use", "memory", "rag", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.SE" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "15", "collection_queries": "coding-agent" } }, { "collection": "papers", "path": "papers/items/2026-2607-00041-atm-cid-brokered-pre-write-admission-for-multi-agent-code-co-synthesis.md", "title": "\"ATM: CID-Brokered Pre-Write Admission for Multi-Agent Code Co-Synthesis\"", "type": "paper", "meta": { "type": "paper", "title": "\"ATM: CID-Brokered Pre-Write Admission for Multi-Agent Code Co-Synthesis\"", "authors": "Eagl Huang", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2607.00041", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-29", "updated_at": "2026-06-29", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "coding-agent", "computer-use", "multi-agent", "planning", "rag" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.SE", "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "19", "collection_queries": "agent-evaluation, multi-agent-llm" } }, { "collection": "papers", "path": "papers/items/2026-2607-00233-from-signals-to-structure-how-memory-architecture-drives-language-emergence-in-l.md", "title": "\"From Signals to Structure: How Memory Architecture Drives Language Emergence in LLM Agents\"", "type": "paper", "meta": { "type": "paper", "title": "\"From Signals to Structure: How Memory Architecture Drives Language Emergence in LLM Agents\"", "authors": "Yashar Talebirad, Eden Redman, Ali Parsaee, Osmar R. Zaiane", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2607.00233", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-30", "updated_at": "2026-06-30", "status": "queued", "relevance": "high", "topics": [ "memory", "rag", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI", "cs.CL", "cs.IT", "cs.MA" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "16", "collection_queries": "llm-agent" } }, { "collection": "papers", "path": "papers/items/2026-2607-00255-slm-llm-or-agentic-ai-toward-intelligent-uav-enabled-wpt-systems-in-low-altitude.md", "title": "SLM, LLM or Agentic AI? Toward Intelligent UAV-Enabled WPT Systems in Low-Altitude Economy Networks", "type": "paper", "meta": { "type": "paper", "title": "SLM, LLM or Agentic AI? Toward Intelligent UAV-Enabled WPT Systems in Low-Altitude Economy Networks", "authors": "Feibo Jiang, Li Dong, Lei Mao, Kezhi Wang, Xianbin Wang, Abbas Jamalipour", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2607.00255", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-30", "updated_at": "2026-06-30", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "reasoning", "world-model" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.IT" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "14", "collection_queries": "agentic-ai" } }, { "collection": "papers", "path": "papers/items/2026-2607-00297-epc-a-standardized-protocol-for-measuring-evaluator-preference-dynamics-in-llm-a.md", "title": "\"EPC: A Standardized Protocol for Measuring Evaluator Preference Dynamics in LLM Agent Systems\"", "type": "paper", "meta": { "type": "paper", "title": "\"EPC: A Standardized Protocol for Measuring Evaluator Preference Dynamics in LLM Agent Systems\"", "authors": "Zewen Liu", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2607.00297", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-07-01", "updated_at": "2026-07-01", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "coding-agent", "computer-use", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.LG", "cs.CL" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "14", "collection_queries": "llm-agent" } }, { "collection": "papers", "path": "papers/items/2026-2607-00334-managed-autonomy-at-runtime-gear-based-safety-and-governance-for-single-and-mult.md", "title": "\"Managed Autonomy at Runtime: Gear-Based Safety and Governance for Single- and Multi-Agent Cyber-Physical Systems\"", "type": "paper", "meta": { "type": "paper", "title": "\"Managed Autonomy at Runtime: Gear-Based Safety and Governance for Single- and Multi-Agent Cyber-Physical Systems\"", "authors": "Srini Ramaswamy, Wang Miaosheng", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2607.00334", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-07-01", "updated_at": "2026-07-01", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "embodied-agent", "multi-agent", "planning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "17", "collection_queries": "autonomous-agent-llm, multi-agent-llm" } }, { "collection": "papers", "path": "papers/items/2026-2607-00345-registry-governed-agent-lifecycle-completing-eddops-with-evaluation-drivenregist.md", "title": "\"Registry-Governed Agent Lifecycle:Completing EDDOps with Evaluation-DrivenRegistration, Promotion, and Retirement on AWS AgentCore\"", "type": "paper", "meta": { "type": "paper", "title": "\"Registry-Governed Agent Lifecycle:Completing EDDOps with Evaluation-DrivenRegistration, Promotion, and Retirement on AWS AgentCore\"", "authors": "Richard Kang, Vincent Wang", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2607.00345", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-07-01", "updated_at": "2026-07-01", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "workflow-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.SE" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "15", "collection_queries": "llm-agent" } }, { "collection": "papers", "path": "papers/items/2026-2607-00407-personalization-as-inverse-planning-learning-latent-design-intents-for-agentic-s.md", "title": "\"Personalization as Inverse Planning: Learning Latent Design Intents for Agentic Slide Generation via Structural Denoising\"", "type": "paper", "meta": { "type": "paper", "title": "\"Personalization as Inverse Planning: Learning Latent Design Intents for Agentic Slide Generation via Structural Denoising\"", "authors": "Tianci Liu, Zihan Dong, Linjun Zhang, Haoyu Wang, jing Gao, Emre Kiciman, Ranveer Chandra, Wei-Ting Chen", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2607.00407", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-07-01", "updated_at": "2026-07-01", "status": "queued", "relevance": "high", "topics": [ "multi-agent", "planning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "13", "collection_queries": "ai-agent" } }, { "collection": "papers", "path": "papers/items/2026-2607-00422-kidnaprag-a-black-box-attack-for-hijacking-reasoning-in-agentic-retrieval-augmen.md", "title": "\"KidnapRAG: A Black-Box Attack for Hijacking Reasoning in Agentic Retrieval-Augmented Generation Systems\"", "type": "paper", "meta": { "type": "paper", "title": "\"KidnapRAG: A Black-Box Attack for Hijacking Reasoning in Agentic Retrieval-Augmented Generation Systems\"", "authors": "Chanwoo Choi, Euntae Kim, Kyuho Lee, Youngsam Chun, Jinhee Jeong, Eunmi Kim, Myunggyo Oh, Junseo Jang, et al.", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2607.00422", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-07-01", "updated_at": "2026-07-01", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "rag", "reasoning" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CR" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "14", "collection_queries": "rag-agent" } }, { "collection": "papers", "path": "papers/items/2026-2607-00436-phreeqc-mcq-200-a-diagnostic-benchmark-for-tool-augmented-scientific-simulator-a.md", "title": "\"PHREEQC-MCQ-200: A Diagnostic Benchmark for Tool-Augmented Scientific Simulator Agents\"", "type": "paper", "meta": { "type": "paper", "title": "\"PHREEQC-MCQ-200: A Diagnostic Benchmark for Tool-Augmented Scientific Simulator Agents\"", "authors": "Ke Zhang, Sahchit Chundur, Mohammad Javad Qomi, Maziar Raissi", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2607.00436", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-07-01", "updated_at": "2026-07-01", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "rag", "tool-use", "world-model" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "16", "collection_queries": "tool-use" } }, { "collection": "papers", "path": "papers/items/2026-2607-00440-minos-a-multi-agent-collaborative-framework-for-provenance-based-backward-tracki.md", "title": "\"Minos: A Multi-Agent Collaborative Framework for Provenance-Based Backward Tracking\"", "type": "paper", "meta": { "type": "paper", "title": "\"Minos: A Multi-Agent Collaborative Framework for Provenance-Based Backward Tracking\"", "authors": "Jiahui Wang, Zhenyuan Li, Zhengkai Wang, Xiangmin Shen, Fan Zhang", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2607.00440", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-07-01", "updated_at": "2026-07-01", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "computer-use", "multi-agent", "rag", "reasoning" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CR" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "16", "collection_queries": "multi-agent-llm" } }, { "collection": "papers", "path": "papers/items/2026-2607-00454-agri-sage-simulation-grounded-multi-agent-llm-for-context-aware-agricultural-adv.md", "title": "\"Agri-SAGE: Simulation-Grounded Multi-Agent LLM for Context-Aware Agricultural Advisory Generation\"", "type": "paper", "meta": { "type": "paper", "title": "\"Agri-SAGE: Simulation-Grounded Multi-Agent LLM for Context-Aware Agricultural Advisory Generation\"", "authors": "Vedant Balasubramaniam, Geetha Charan, Manojkumar Patil, Rohit P Suresh, V Priyanka, Kodur Sai Vinay Sathvik, Y. Narahari", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2607.00454", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-07-01", "updated_at": "2026-07-01", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "computer-use", "memory", "multi-agent", "planning", "rag", "reasoning", "world-model" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI", "cs.MA" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "20", "collection_queries": "multi-agent-llm" } }, { "collection": "papers", "path": "papers/items/2026-2607-00502-a-task-state-representation-for-long-horizon-mobile-gui-agents.md", "title": "A Task-State Representation for Long-Horizon Mobile GUI Agents", "type": "paper", "meta": { "type": "paper", "title": "A Task-State Representation for Long-Horizon Mobile GUI Agents", "authors": "Yujie Zheng, Zikang Liu, Xin Zhao, Ji-Rong Wen", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2607.00502", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-07-01", "updated_at": "2026-07-01", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "computer-use", "memory", "planning", "reasoning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CL" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "16", "collection_queries": "web-gui-agent" } }, { "collection": "papers", "path": "papers/items/2026-2607-00555-rise-from-the-ashes-llm-based-static-analysis-for-deep-learning-framework-bugs.md", "title": "\"Rise From The Ashes: LLM-based Static Analysis for Deep Learning Framework Bugs\"", "type": "paper", "meta": { "type": "paper", "title": "\"Rise From The Ashes: LLM-based Static Analysis for Deep Learning Framework Bugs\"", "authors": "Shaoyu Yang, Haifeng Lin, Chunrong Fang, Xiang Chen, Wei Cheng, Jiawei Liu, Yiyu Zhang, Hongyu Liu, et al.", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2607.00555", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-07-01", "updated_at": "2026-07-01", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "coding-agent", "computer-use", "multi-agent", "rag", "tool-use", "workflow-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.SE" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "16", "collection_queries": "agentic-ai, multi-agent-llm" } }, { "collection": "papers", "path": "papers/items/2026-2607-00604-vehicle-routing-problem-meets-large-language-models-an-overview-and-perspectives.md", "title": "\"Vehicle Routing Problem Meets Large Language Models: An Overview and Perspectives\"", "type": "paper", "meta": { "type": "paper", "title": "\"Vehicle Routing Problem Meets Large Language Models: An Overview and Perspectives\"", "authors": "Xianchao Xiu, Chong Shen, Yanjiao Zhu, Wanquan Liu", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2607.00604", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-07-01", "updated_at": "2026-07-01", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "multi-agent", "planning", "reasoning", "tool-use", "workflow-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "math.OC" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "14", "collection_queries": "multi-agent-llm" } }, { "collection": "papers", "path": "papers/items/2026-2607-00627-agi-maze-as-a-benchmark-framework-for-world-modeling-agents.md", "title": "AGI Maze as a Benchmark Framework for World-Modeling Agents", "type": "paper", "meta": { "type": "paper", "title": "AGI Maze as a Benchmark Framework for World-Modeling Agents", "authors": "Alexey Potapov", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2607.00627", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-07-01", "updated_at": "2026-07-01", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "memory", "reasoning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "18", "collection_queries": "llm-agent" } }, { "collection": "papers", "path": "papers/items/2026-2607-00692-self-gc-self-governing-context-for-long-horizon-llm-agents.md", "title": "\"Self-GC: Self-Governing Context for Long-Horizon LLM Agents\"", "type": "paper", "meta": { "type": "paper", "title": "\"Self-GC: Self-Governing Context for Long-Horizon LLM Agents\"", "authors": "Xubin Hao, Hongjin Meng, Xin Yin, Jiawei Zhu, Chenpeng Cao", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2607.00692", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-07-01", "updated_at": "2026-07-01", "status": "queued", "relevance": "high", "topics": [ "planning", "rag", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "14", "collection_queries": "llm-agent, planning-agent" } }, { "collection": "papers", "path": "papers/items/2026-2607-00911-from-registry-to-repository-how-ai-agent-skills-are-written-adapted-and-maintain.md", "title": "\"From Registry to Repository: How AI Agent Skills Are Written, Adapted, and Maintained\"", "type": "paper", "meta": { "type": "paper", "title": "\"From Registry to Repository: How AI Agent Skills Are Written, Adapted, and Maintained\"", "authors": "Haoyu Gao, Jai Lal Lulla, Hong Yi Lin, Sebastian Baltes, Christoph Treude, Mansooreh Zahedi", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2607.00911", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-07-01", "updated_at": "2026-07-06", "status": "queued", "relevance": "high", "topics": [ "coding-agent", "tool-use", "workflow-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.SE" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "15", "collection_queries": "ai-agent, coding-agent" } }, { "collection": "papers", "path": "papers/items/2026-2607-00918-from-personas-to-plot-character-grounded-multi-agent-story-generation-for-long-f.md", "title": "\"From Personas to Plot: Character-Grounded Multi-Agent Story Generation for Long-Form Narratives\"", "type": "paper", "meta": { "type": "paper", "title": "\"From Personas to Plot: Character-Grounded Multi-Agent Story Generation for Long-Form Narratives\"", "authors": "Aayush Aluru, Chloe Ho, Muhammad Hammouri, Kerry Luo, Myra Malik, Ryan Lagasse, Arjun Bahuguna, Vasu Sharma", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2607.00918", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-07-01", "updated_at": "2026-07-01", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "multi-agent", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CL", "cs.AI", "cs.MA" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "14", "collection_queries": "multi-agent-llm" } }, { "collection": "papers", "path": "papers/items/2026-2607-00939-leveraging-llm-based-agentic-systems-to-generate-quantum-applications-for-test-o.md", "title": "Leveraging LLM-Based Agentic Systems to Generate Quantum Applications for Test Optimization", "type": "paper", "meta": { "type": "paper", "title": "Leveraging LLM-Based Agentic Systems to Generate Quantum Applications for Test Optimization", "authors": "Ming Tao, Yuechen Li, Tao Yue, Man Zhang, Aitor Arrieta Marcos", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2607.00939", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-07-01", "updated_at": "2026-07-01", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "coding-agent", "multi-agent", "rag", "workflow-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.SE", "quant-ph" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "17", "collection_queries": "multi-agent-llm" } }, { "collection": "papers", "path": "papers/items/2026-2607-00972-bayesian-uncertainty-propagation-for-agentic-rag-pipelines-a-proof-of-concept-st.md", "title": "\"Bayesian Uncertainty Propagation for Agentic RAG Pipelines: A Proof-of-Concept Study on Multi-Hop Question Answering\"", "type": "paper", "meta": { "type": "paper", "title": "\"Bayesian Uncertainty Propagation for Agentic RAG Pipelines: A Proof-of-Concept Study on Multi-Hop Question Answering\"", "authors": "Louis Donaldson, Connor Walker, Koorosh Aslansefat, Yiannis Papadopoulos", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2607.00972", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-07-01", "updated_at": "2026-07-01", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "planning", "rag", "reasoning", "workflow-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "14", "collection_queries": "rag-agent" } }, { "collection": "papers", "path": "papers/items/2026-2607-00990-swe-doctor-guiding-software-engineering-agents-with-runtime-diagnosis-from-multi.md", "title": "\"SWE-Doctor: Guiding Software Engineering Agents with Runtime Diagnosis from Multi-Faceted Bug Reproduction Tests\"", "type": "paper", "meta": { "type": "paper", "title": "\"SWE-Doctor: Guiding Software Engineering Agents with Runtime Diagnosis from Multi-Faceted Bug Reproduction Tests\"", "authors": "Yaoqi Guo, Yang Liu, Jie M. Zhang, Yun Ma, Yiling Lou, Zhenpeng Chen", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2607.00990", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-07-01", "updated_at": "2026-07-01", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "coding-agent", "computer-use", "rag" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.SE", "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "13", "collection_queries": "coding-agent" } }, { "collection": "papers", "path": "papers/items/2026-2607-01047-conversable-complexity-agentic-llm-collectives-as-interpretable-substrates.md", "title": "\"Conversable Complexity: Agentic LLM Collectives as Interpretable Substrates\"", "type": "paper", "meta": { "type": "paper", "title": "\"Conversable Complexity: Agentic LLM Collectives as Interpretable Substrates\"", "authors": "Elias Najarro, Ane Espeseth, Eleni Nisioti, Sebastian Risi, Stefano Nichele", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2607.01047", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-07-01", "updated_at": "2026-07-01", "status": "queued", "relevance": "high", "topics": [ "memory", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CL" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "13", "collection_queries": "llm-agent" } }, { "collection": "papers", "path": "papers/items/2026-2607-01061-agentic-generation-of-verifiable-rules-for-deterministic-self-expanding-reaction.md", "title": "Agentic generation of verifiable rules for deterministic, self-expanding reaction classification", "type": "paper", "meta": { "type": "paper", "title": "Agentic generation of verifiable rules for deterministic, self-expanding reaction classification", "authors": "Daniel Armstrong, Maarten Dobbelaere, Valentas Olikauskas, Helena Avila, Octavian Susanu, Jérôme Waser, Philippe Schwaller", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2607.01061", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-07-01", "updated_at": "2026-07-05", "status": "queued", "relevance": "high", "topics": [ "coding-agent", "multi-agent", "planning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI", "cs.CL" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "15", "collection_queries": "multi-agent-llm" } }, { "collection": "papers", "path": "papers/items/2026-2607-01071-memsyco-bench-benchmarking-sycophancy-in-agent-memory.md", "title": "\"MemSyco-Bench: Benchmarking Sycophancy in Agent Memory\"", "type": "paper", "meta": { "type": "paper", "title": "\"MemSyco-Bench: Benchmarking Sycophancy in Agent Memory\"", "authors": "Zhishang Xiang, Zerui Chen, Yunbo Tang, Zhimin Wei, Ruqin Ning, Yujie Lin, Qinggang Zhang, Jinsong Su", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2607.01071", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-07-01", "updated_at": "2026-07-02", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "memory", "reasoning" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.IR", "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "17", "collection_queries": "agent-memory" } }, { "collection": "papers", "path": "papers/items/2026-2607-01084-can-agents-generalize-to-the-open-world-unveiling-the-fragility-of-static-traini.md", "title": "Can Agents Generalize to the Open World? Unveiling the Fragility of Static Training in Tool Use", "type": "paper", "meta": { "type": "paper", "title": "Can Agents Generalize to the Open World? Unveiling the Fragility of Static Training in Tool Use", "authors": "Song-Lin Lv, Weiming Wu, Rui Zhu, Zi-Jian Cheng, Lan-Zhe Guo", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2607.01084", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-07-01", "updated_at": "2026-07-01", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "rag", "reasoning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "16", "collection_queries": "llm-agent, tool-use" } }, { "collection": "papers", "path": "papers/items/2026-2607-01120-next-generation-agentic-reinforcement-learning-systems-enable-self-evolving-agen.md", "title": "Next-Generation Agentic Reinforcement Learning Systems Enable Self-Evolving Agents", "type": "paper", "meta": { "type": "paper", "title": "Next-Generation Agentic Reinforcement Learning Systems Enable Self-Evolving Agents", "authors": "Ran Yan, Wei Fu, Jiale Li, Shusheng Xu, Zhiyu Mei, Jiaxuan Gao, Jiarui Zhang, Wentai Zhang, et al.", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2607.01120", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-07-01", "updated_at": "2026-07-02", "status": "queued", "relevance": "high", "topics": [ "coding-agent", "planning", "tool-use", "workflow-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.DC" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "15", "collection_queries": "llm-agent" } }, { "collection": "papers", "path": "papers/items/2026-2607-01211-are-performance-optimization-benchmarks-reliably-measuring-coding-agents.md", "title": "Are Performance-Optimization Benchmarks Reliably Measuring Coding Agents?", "type": "paper", "meta": { "type": "paper", "title": "Are Performance-Optimization Benchmarks Reliably Measuring Coding Agents?", "authors": "Zhi Chen, Zhensu Sun, Yuling Shi, David Lo, Lingxiao Jiang", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2607.01211", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-07-01", "updated_at": "2026-07-01", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "coding-agent", "rag" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.SE", "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "14", "collection_queries": "coding-agent" } }, { "collection": "papers", "path": "papers/items/2026-2607-01213-reporescue-an-empirical-study-of-llm-agents-on-whole-repository-compatibility-re.md", "title": "\"RepoRescue: An Empirical Study of LLM Agents on Whole-Repository Compatibility Rescue\"", "type": "paper", "meta": { "type": "paper", "title": "\"RepoRescue: An Empirical Study of LLM Agents on Whole-Repository Compatibility Rescue\"", "authors": "Zhihao Lin, Mingyi Zhou, Zhensu Sun, Yizhuo Yang, Renyu Yang, David Lo, Li Li", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2607.01213", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-07-01", "updated_at": "2026-07-01", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "coding-agent", "reasoning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.SE" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "15", "collection_queries": "llm-agent" } }, { "collection": "papers", "path": "papers/items/2026-2607-01366-auto-fl-research-agentic-search-for-federated-learning-algorithms.md", "title": "\"Auto-FL-Research: Agentic Search for Federated Learning Algorithms\"", "type": "paper", "meta": { "type": "paper", "title": "\"Auto-FL-Research: Agentic Search for Federated Learning Algorithms\"", "authors": "Holger R. Roth, Ziyue Xu, Chester Chen, Daguang Xu, Peter Cnudde, Andrew Feng", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2607.01366", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-07-01", "updated_at": "2026-07-01", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "coding-agent", "workflow-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "13", "collection_queries": "agentic-ai, coding-agent" } }, { "collection": "papers", "path": "papers/items/2026-2607-01421-risk-architecture-for-ai-native-engineering-teams-an-organizational-framework-fo.md", "title": "\"Risk Architecture for AI-Native Engineering Teams: An Organizational Framework for Agentic System Governance\"", "type": "paper", "meta": { "type": "paper", "title": "\"Risk Architecture for AI-Native Engineering Teams: An Organizational Framework for Agentic System Governance\"", "authors": "Laxmipriya Ganesh Iyer", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2607.01421", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-07-01", "updated_at": "2026-07-01", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "coding-agent", "computer-use", "rag", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.SE", "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "15", "collection_queries": "agentic-ai" } }, { "collection": "papers", "path": "papers/items/2026-2607-01510-janus-a-playground-for-user-involved-agentic-permission-management.md", "title": "\"Janus: a Playground for User-Involved Agentic Permission Management\"", "type": "paper", "meta": { "type": "paper", "title": "\"Janus: a Playground for User-Involved Agentic Permission Management\"", "authors": "Natalie Grace Brigham, Eugene Bagdasarian, Tadayoshi Kohno, Franziska Roesner", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2607.01510", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-07-01", "updated_at": "2026-07-01", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI", "cs.CR" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "13", "collection_queries": "ai-agent" } }, { "collection": "papers", "path": "papers/items/2026-2607-01523-multi-head-recurrent-memory-agents.md", "title": "Multi-Head Recurrent Memory Agents", "type": "paper", "meta": { "type": "paper", "title": "Multi-Head Recurrent Memory Agents", "authors": "Jiatong Li, Samuel Yeh, Sharon Li", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2607.01523", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-07-01", "updated_at": "2026-07-01", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "memory" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.LG", "cs.AI", "cs.CL" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "14", "collection_queries": "agent-memory" } }, { "collection": "papers", "path": "papers/items/2026-2607-01531-opine-world-programmatic-world-modeling-with-ontology-error-prioritized-interact.md", "title": "\"OPINE-World: Programmatic World Modeling with Ontology-error-Prioritized Interactive Exploration\"", "type": "paper", "meta": { "type": "paper", "title": "\"OPINE-World: Programmatic World Modeling with Ontology-error-Prioritized Interactive Exploration\"", "authors": "David Courtis, Wenhao Li, Scott Sanner", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2607.01531", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-07-01", "updated_at": "2026-07-01", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "computer-use", "planning", "tool-use", "world-model" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI", "cs.LG" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "13", "collection_queries": "llm-agent, planning-agent" } }, { "collection": "papers", "path": "papers/items/2026-2607-01600-boundary-sync-measuring-communication-induced-representational-coupling-in-multi.md", "title": "\"BOUNDARY_SYNC: Measuring Communication-Induced Representational Coupling in Multi-Agent LLM Systems\"", "type": "paper", "meta": { "type": "paper", "title": "\"BOUNDARY_SYNC: Measuring Communication-Induced Representational Coupling in Multi-Agent LLM Systems\"", "authors": "Zewen Liu", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2607.01600", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-07-02", "updated_at": "2026-07-02", "status": "queued", "relevance": "high", "topics": [ "multi-agent", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.LG", "cs.CL" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "14", "collection_queries": "llm-agent, multi-agent-llm" } }, { "collection": "papers", "path": "papers/items/2026-2607-01640-agentflow-building-agent-dependency-graphs-for-static-analysis-of-agent-programs.md", "title": "\"AgentFlow: Building Agent Dependency Graphs for Static Analysis of Agent Programs\"", "type": "paper", "meta": { "type": "paper", "title": "\"AgentFlow: Building Agent Dependency Graphs for Static Analysis of Agent Programs\"", "authors": "Shenao Wang, Xinyi Hou, Yanjie Zhao, Xiao Cheng, Haoyu Wang", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2607.01640", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-07-02", "updated_at": "2026-07-02", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "memory", "multi-agent", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.SE", "cs.CR" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "17", "collection_queries": "llm-agent, multi-agent-llm" } }, { "collection": "papers", "path": "papers/items/2026-2607-01641-when-agents-do-not-stop-uncovering-infinite-agentic-loops-in-llm-agents.md", "title": "\"When Agents Do Not Stop: Uncovering Infinite Agentic Loops in LLM Agents\"", "type": "paper", "meta": { "type": "paper", "title": "\"When Agents Do Not Stop: Uncovering Infinite Agentic Loops in LLM Agents\"", "authors": "Xinyi Hou, Shenao Wang, Yanjie Zhao, Haoyu Wang", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2607.01641", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-07-02", "updated_at": "2026-07-02", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "multi-agent", "planning", "tool-use", "workflow-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.SE" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "21", "collection_queries": "llm-agent, planning-agent, tool-use" } }, { "collection": "papers", "path": "papers/items/2026-2607-01661-diverse-evidence-better-forecasts-multi-agent-deliberation-under-information-asy.md", "title": "\"Diverse Evidence, Better Forecasts: Multi-Agent Deliberation Under Information Asymmetry\"", "type": "paper", "meta": { "type": "paper", "title": "\"Diverse Evidence, Better Forecasts: Multi-Agent Deliberation Under Information Asymmetry\"", "authors": "Yuante Li, Yicheng Tao, Kate Zhang, Taozhi Wang, Gefei Gu, Yaxin Zhou", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2607.01661", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-07-02", "updated_at": "2026-07-02", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "multi-agent", "reasoning" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "14", "collection_queries": "multi-agent-llm" } }, { "collection": "papers", "path": "papers/items/2026-2607-01668-verichat-an-agentic-conversational-ai-assistant-for-hardware-security-verificati.md", "title": "\"VeriChat: An Agentic Conversational AI Assistant for Hardware Security Verification\"", "type": "paper", "meta": { "type": "paper", "title": "\"VeriChat: An Agentic Conversational AI Assistant for Hardware Security Verification\"", "authors": "Dipayan Saha, Khan Thamid Hasan, Shams Tarek, Sujan Kumar Saha, Mark Tehranipoor, Farimah Farahmandi", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2607.01668", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-07-02", "updated_at": "2026-07-02", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "computer-use", "multi-agent", "rag", "tool-use", "workflow-agent", "world-model" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CR" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "17", "collection_queries": "agentic-ai" } }, { "collection": "papers", "path": "papers/items/2026-2607-01709-comfyclaw-self-evolving-skill-harnesses-for-image-generation-workflows.md", "title": "\"COMFYCLAW: Self-Evolving Skill Harnesses for Image Generation Workflows\"", "type": "paper", "meta": { "type": "paper", "title": "\"COMFYCLAW: Self-Evolving Skill Harnesses for Image Generation Workflows\"", "authors": "Zongxia Li, Dawei Liu, Fuxiao Liu, Yuhang Zhou, Xiyang Wu, Jingxi Chen, Jing Xie, Xiaomin Wu, et al.", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2607.01709", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-07-02", "updated_at": "2026-07-02", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "memory", "rag", "reasoning", "tool-use", "workflow-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI", "cs.LG" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "15", "collection_queries": "agent-memory" } }, { "collection": "papers", "path": "papers/items/2026-2607-01766-simworlds-a-multi-agent-system-for-dynamic-3d-scene-creation.md", "title": "\"SimWorlds: A Multi-Agent System for Dynamic 3D Scene Creation\"", "type": "paper", "meta": { "type": "paper", "title": "\"SimWorlds: A Multi-Agent System for Dynamic 3D Scene Creation\"", "authors": "Chunjiang Liu, Xiaoyuan Wang, Haoyu Chen, Yizhou Zhao, Ming-Hsuan Yang, László A. Jeni", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2607.01766", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-07-02", "updated_at": "2026-07-02", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "embodied-agent", "multi-agent", "planning", "reasoning", "tool-use", "workflow-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "20", "collection_queries": "llm-agent, multi-agent-llm" } }, { "collection": "papers", "path": "papers/items/2026-2607-01767-repair-the-amplifier-not-the-symptom-stable-world-model-correction-for-agent-rol.md", "title": "\"Repair the Amplifier, Not the Symptom: Stable World-Model Correction for Agent Rollouts\"", "type": "paper", "meta": { "type": "paper", "title": "\"Repair the Amplifier, Not the Symptom: Stable World-Model Correction for Agent Rollouts\"", "authors": "Xinyuan Song, Zekun Cai", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2607.01767", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-07-02", "updated_at": "2026-07-05", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "memory", "planning", "tool-use", "world-model" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "15", "collection_queries": "language-agent" } }, { "collection": "papers", "path": "papers/items/2026-2607-01788-krca-an-efficient-root-cause-analysis-system-in-hyper-scale-microservice-systems.md", "title": "\"KRCA: An Efficient Root Cause Analysis System in Hyper-Scale Microservice Systems via Agentic AI\"", "type": "paper", "meta": { "type": "paper", "title": "\"KRCA: An Efficient Root Cause Analysis System in Hyper-Scale Microservice Systems via Agentic AI\"", "authors": "Jiamin Jiang, Jingfei Feng, Yu Luo, Qingliang Zhang, Yongqian Su, Wenwei Gu, Shenglin Zhang, Tianyu Cui, et al.", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2607.01788", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-07-02", "updated_at": "2026-07-02", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "memory", "multi-agent", "rag", "reasoning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.SE" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "16", "collection_queries": "agentic-ai" } }, { "collection": "papers", "path": "papers/items/2026-2607-01793-safety-testing-llm-agents-at-scale-from-risk-discovery-to-evidence-grounded-veri.md", "title": "\"Safety Testing LLM Agents at Scale: From Risk Discovery to Evidence-Grounded Verification\"", "type": "paper", "meta": { "type": "paper", "title": "\"Safety Testing LLM Agents at Scale: From Risk Discovery to Evidence-Grounded Verification\"", "authors": "Yunhao Feng, Ruixiao Lin, Ming Wen, Qinqin He, Yanming Guo, Yifan Ding, Yutao Wu, Jialuo Chen, et al.", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2607.01793", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-07-02", "updated_at": "2026-07-04", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "coding-agent", "rag", "reasoning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "18", "collection_queries": "llm-agent" } }, { "collection": "papers", "path": "papers/items/2026-2607-01812-to-master-an-llm-agent-framework-for-automated-topology-optimization.md", "title": "\"TO-Master: an LLM-agent framework for automated topology optimization\"", "type": "paper", "meta": { "type": "paper", "title": "\"TO-Master: an LLM-agent framework for automated topology optimization\"", "authors": "Haoju Lin, Wenchang Zhang, Weipeng Xu, Xiang Li, Tian Xu, Tianju Xue", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2607.01812", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-07-02", "updated_at": "2026-07-02", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "computer-use", "reasoning", "tool-use", "workflow-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CE" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "13", "collection_queries": "llm-agent" } }, { "collection": "papers", "path": "papers/items/2026-2607-01874-skillcoach-self-evolving-rubrics-for-evaluating-and-enhancing-agentic-skill-use.md", "title": "\"SkillCoach: Self-Evolving Rubrics for Evaluating and Enhancing Agentic Skill-Use\"", "type": "paper", "meta": { "type": "paper", "title": "\"SkillCoach: Self-Evolving Rubrics for Evaluating and Enhancing Agentic Skill-Use\"", "authors": "Jiayin Zhu, Kelong Mao, Yudong Guo, Dengbo He, Sulong Xu, Simiu Gu, Yutao Yue", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2607.01874", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-07-02", "updated_at": "2026-07-02", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "coding-agent", "computer-use", "reasoning", "tool-use", "workflow-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI", "cs.CL" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "18", "collection_queries": "llm-agent" } }, { "collection": "papers", "path": "papers/items/2026-2607-01916-contextsniper-anttrail-s-token-efficient-code-memory-for-repository-level-progra.md", "title": "\"ContextSniper: AntTrail's Token-Efficient Code Memory for Repository-Level Program Repair\"", "type": "paper", "meta": { "type": "paper", "title": "\"ContextSniper: AntTrail's Token-Efficient Code Memory for Repository-Level Program Repair\"", "authors": "Chiwang Luk, Matin Mohammad Najafi, Zhifeng Jia, Wei Yang, Xiuchang Li, Jinwei Zhu, Yang Ren, Lei Chen, et al.", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2607.01916", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-07-02", "updated_at": "2026-07-06", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "coding-agent", "memory", "rag", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "17", "collection_queries": "agent-memory, coding-agent, rag-agent" } }, { "collection": "papers", "path": "papers/items/2026-2607-01929-beyond-textual-repository-exploration-dual-modal-structural-reasoning-for-agenti.md", "title": "\"Beyond Textual Repository Exploration: Dual-Modal Structural Reasoning for Agentic Issue Resolution\"", "type": "paper", "meta": { "type": "paper", "title": "\"Beyond Textual Repository Exploration: Dual-Modal Structural Reasoning for Agentic Issue Resolution\"", "authors": "Jiayi Zhang, Kai Huang, Yang Liu, Chunyang Chen", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2607.01929", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-07-02", "updated_at": "2026-07-02", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "coding-agent", "embodied-agent", "planning", "rag", "reasoning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.SE" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "17", "collection_queries": "coding-agent, function-calling" } }, { "collection": "papers", "path": "papers/items/2026-2607-01935-a-tma-decoupling-state-aware-memory-failures-in-long-term-agent-memory.md", "title": "\"A-TMA: Decoupling State-Aware Memory Failures in Long-Term Agent Memory\"", "type": "paper", "meta": { "type": "paper", "title": "\"A-TMA: Decoupling State-Aware Memory Failures in Long-Term Agent Memory\"", "authors": "Zitong Shi, Yixuan Tang, Anthony Kum Hoe Tung", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2607.01935", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-07-02", "updated_at": "2026-07-02", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "memory", "rag" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "17", "collection_queries": "agent-memory, llm-agent" } }, { "collection": "papers", "path": "papers/items/2026-2607-02032-pace-a-proxy-for-agentic-capability-evaluation.md", "title": "\"PACE: A Proxy for Agentic Capability Evaluation\"", "type": "paper", "meta": { "type": "paper", "title": "\"PACE: A Proxy for Agentic Capability Evaluation\"", "authors": "Yueqi Song, Lintang Sutawika, Jiarui Liu, Lindia Tjuatja, Jiayi Geng, Yunze Xiao, Daniel Lee, Aditya Bharat Soni, et al.", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2607.02032", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-07-02", "updated_at": "2026-07-06", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "coding-agent", "reasoning" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI", "cs.CL" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "21", "collection_queries": "agent-evaluation, coding-agent, llm-agent" } }, { "collection": "papers", "path": "papers/items/2026-2607-02186-ua-chatdev-uncertainty-aware-multi-agent-collaboration-for-reliable-software-dev.md", "title": "\"UA-ChatDev: Uncertainty-Aware Multi-Agent Collaboration for Reliable Software Development\"", "type": "paper", "meta": { "type": "paper", "title": "\"UA-ChatDev: Uncertainty-Aware Multi-Agent Collaboration for Reliable Software Development\"", "authors": "Temitayo Olamilekan Ogunsusi, Lijun Qian, Xishuang Dong", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2607.02186", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-07-02", "updated_at": "2026-07-02", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "coding-agent", "multi-agent", "rag", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "16", "collection_queries": "multi-agent-llm" } }, { "collection": "papers", "path": "papers/items/2026-2607-02210-criticality-based-guard-rail-validation-for-ai-agent-decisions-in-autonomous-tel.md", "title": "Criticality-Based Guard Rail Validation for AI Agent Decisions in Autonomous Telecom Networks", "type": "paper", "meta": { "type": "paper", "title": "Criticality-Based Guard Rail Validation for AI Agent Decisions in Autonomous Telecom Networks", "authors": "Ravi Kant Sharma", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2607.02210", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-07-02", "updated_at": "2026-07-02", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "multi-agent", "rag", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI", "cs.NI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "14", "collection_queries": "ai-agent" } }, { "collection": "papers", "path": "papers/items/2026-2607-02245-copewell-a-multi-agent-swarm-architecture-for-equitable-mental-wellness-support.md", "title": "\"Copewell: A Multi-Agent Swarm Architecture for Equitable Mental Wellness Support\"", "type": "paper", "meta": { "type": "paper", "title": "\"Copewell: A Multi-Agent Swarm Architecture for Equitable Mental Wellness Support\"", "authors": "Seren Yenikent, Jack Vinijtrongjit, Katherine Ng", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2607.02245", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-07-02", "updated_at": "2026-07-02", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "multi-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI", "cs.CY", "cs.HC" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "13", "collection_queries": "ai-agent" } }, { "collection": "papers", "path": "papers/items/2026-2607-02255-agenticsts-a-bounded-memory-testbed-for-long-horizon-llm-agents.md", "title": "\"AgenticSTS: A Bounded-Memory Testbed for Long-Horizon LLM Agents\"", "type": "paper", "meta": { "type": "paper", "title": "\"AgenticSTS: A Bounded-Memory Testbed for Long-Horizon LLM Agents\"", "authors": "Xiangchen Cheng, Yunwei Jiang, Jianwen Sun, Zizhen Li, Chuanhao Li, Xiangcheng Cao, Yihao Liu, Fanrui Zhang, et al.", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2607.02255", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-07-02", "updated_at": "2026-07-02", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "memory", "planning", "rag", "reasoning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI", "cs.CL" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "23", "collection_queries": "llm-agent" } }, { "collection": "papers", "path": "papers/items/2026-2607-02294-coding-agents-are-guessing-measuring-action-boundary-violations-in-underspecifie.md", "title": "\"Coding Agents Are Guessing: Measuring Action-Boundary Violations in Underspecified DevOps Instructions\"", "type": "paper", "meta": { "type": "paper", "title": "\"Coding Agents Are Guessing: Measuring Action-Boundary Violations in Underspecified DevOps Instructions\"", "authors": "Zimo Ji, Zekai Zhang, Congying Xu, Zongjie Li, Yudong Gao, Shuai Wang, Shing-Chi Cheung", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2607.02294", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-07-02", "updated_at": "2026-07-02", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "coding-agent", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.SE" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "17", "collection_queries": "agent-evaluation, coding-agent" } }, { "collection": "papers", "path": "papers/items/2026-2607-02381-hulat2-at-mer-trans-2026-governed-multi-agent-simplification-for-spanish-easy-to.md", "title": "\"HULAT2 at MER-TRANS 2026: Governed Multi-Agent Simplification for Spanish Easy-to-Read Generation\"", "type": "paper", "meta": { "type": "paper", "title": "\"HULAT2 at MER-TRANS 2026: Governed Multi-Agent Simplification for Spanish Easy-to-Read Generation\"", "authors": "Lourdes Moreno, Paloma Martínez, Marco Antonio Sanchez-Escudero, Miguel Domínguez-Gómez", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2607.02381", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-07-02", "updated_at": "2026-07-02", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "computer-use", "multi-agent", "tool-use", "workflow-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CL" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "13", "collection_queries": "agentic-ai" } }, { "collection": "papers", "path": "papers/items/2026-2607-02448-agentscad-automated-design-for-manufacturing-of-fdm-parts-via-multi-agent-llm-re.md", "title": "\"AgentsCAD: Automated Design for Manufacturing of FDM Parts via Multi-Agent LLM Reasoning and Geometric Feature Recognition\"", "type": "paper", "meta": { "type": "paper", "title": "\"AgentsCAD: Automated Design for Manufacturing of FDM Parts via Multi-Agent LLM Reasoning and Geometric Feature Recognition\"", "authors": "Emmanuel George, Christopher Keefe, Peter Pak, Amir Barati Farimani", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2607.02448", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-07-02", "updated_at": "2026-07-07", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "multi-agent", "reasoning", "workflow-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.MA" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "15", "collection_queries": "multi-agent-llm" } }, { "collection": "papers", "path": "papers/items/2026-2607-02453-adoption-and-ecosystem-health-a-longitudinal-analysis-of-open-source-multi-agent.md", "title": "\"Adoption and Ecosystem Health: A Longitudinal Analysis of Open-Source Multi-Agent Frameworks\"", "type": "paper", "meta": { "type": "paper", "title": "\"Adoption and Ecosystem Health: A Longitudinal Analysis of Open-Source Multi-Agent Frameworks\"", "authors": "Xi Zhang, Papi Menon, Vivian Chu, Koray Cosguner", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2607.02453", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-07-02", "updated_at": "2026-07-02", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "multi-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.MA" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "13", "collection_queries": "ai-agent" } }, { "collection": "papers", "path": "papers/items/2026-2607-02507-what-llm-agents-say-when-no-one-is-watching-social-structure-and-latent-objectiv.md", "title": "\"What LLM Agents Say When No One Is Watching: Social Structure and Latent Objective Emergence in Multi-Agent Debates\"", "type": "paper", "meta": { "type": "paper", "title": "\"What LLM Agents Say When No One Is Watching: Social Structure and Latent Objective Emergence in Multi-Agent Debates\"", "authors": "Arman Ghaffarizadeh, Danyal Mohaddes, Aliakbar Izadkhah, Shahriar Noroozizadeh", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2607.02507", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-07-02", "updated_at": "2026-07-02", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "multi-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI", "cs.CL", "cs.LG", "cs.MA" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "19", "collection_queries": "agent-evaluation, llm-agent, multi-agent-llm" } }, { "collection": "papers", "path": "papers/items/2026-2607-02577-benchmarking-the-benchmarks-a-validity-audit-of-tool-calling-evaluation.md", "title": "\"Benchmarking the Benchmarks: A Validity Audit of Tool-Calling Evaluation\"", "type": "paper", "meta": { "type": "paper", "title": "\"Benchmarking the Benchmarks: A Validity Audit of Tool-Calling Evaluation\"", "authors": "Vishvesh Bhat, Jay Vaghasiya, Muhammad Ahmed Mohsin, Asad Aali", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2607.02577", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-30", "updated_at": "2026-06-30", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.SE" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "14", "collection_queries": "tool-use" } }, { "collection": "papers", "path": "papers/items/2026-2607-02579-when-not-to-write-memory-governing-false-promotion-from-correlated-agent-traces.md", "title": "\"When Not to Write Memory: Governing False Promotion from Correlated Agent Traces\"", "type": "paper", "meta": { "type": "paper", "title": "\"When Not to Write Memory: Governing False Promotion from Correlated Agent Traces\"", "authors": "Yijiashun Qi, Xiang Xu, Yuxuan Li", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2607.02579", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-06-30", "updated_at": "2026-06-30", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "coding-agent", "memory", "rag", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.SE" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "16", "collection_queries": "agent-memory, language-agent" } }, { "collection": "papers", "path": "papers/items/2026-2607-02599-agentltl-a-trace-verification-framework-for-measuring-enforcing-and-training-pro.md", "title": "\"AgentLTL: A Trace-Verification Framework for Measuring, Enforcing, and Training Procedural Compliance in Tool-Using LLM Agents\"", "type": "paper", "meta": { "type": "paper", "title": "\"AgentLTL: A Trace-Verification Framework for Measuring, Enforcing, and Training Procedural Compliance in Tool-Using LLM Agents\"", "authors": "Laïla Elkoussy, Julien Perez", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2607.02599", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-07-01", "updated_at": "2026-07-01", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "rag", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.SE", "cs.AI", "cs.LO" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "16", "collection_queries": "llm-agent, tool-use" } }, { "collection": "papers", "path": "papers/items/2026-2607-02606-chainswe-benchmarking-coding-agents-on-multi-bug-software-maintenance.md", "title": "\"ChainSWE: Benchmarking Coding Agents on Multi-Bug Software Maintenance\"", "type": "paper", "meta": { "type": "paper", "title": "\"ChainSWE: Benchmarking Coding Agents on Multi-Bug Software Maintenance\"", "authors": "Qirui Jin, Lingching Tung, Kenan Li, Qiyang Shi, Yushi She, Huanzhong Jia, Harrison Zhao, Kejing Xia, et al.", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2607.02606", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-07-01", "updated_at": "2026-07-01", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "coding-agent", "workflow-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.SE" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "18", "collection_queries": "coding-agent" } }, { "collection": "papers", "path": "papers/items/2026-2607-02684-can-coding-agents-implement-missed-compiler-optimizations-evaluating-llm-agents-.md", "title": "Can Coding Agents Implement Missed Compiler Optimizations? Evaluating LLM Agents on LLVM Peephole Optimizations", "type": "paper", "meta": { "type": "paper", "title": "Can Coding Agents Implement Missed Compiler Optimizations? Evaluating LLM Agents on LLVM Peephole Optimizations", "authors": "Hongxu Xu, Chunhao Liao, Xintong Zhou, Chengnian Sun", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2607.02684", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-07-02", "updated_at": "2026-07-02", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "coding-agent", "reasoning" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.SE" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "19", "collection_queries": "coding-agent, llm-agent" } }, { "collection": "papers", "path": "papers/items/2026-2607-02689-s-ember-a-large-scale-benchmark-for-streaming-egocentric-memory-retrieval.md", "title": "\"S-EMBER: A Large-Scale Benchmark for Streaming Egocentric Memory Retrieval\"", "type": "paper", "meta": { "type": "paper", "title": "\"S-EMBER: A Large-Scale Benchmark for Streaming Egocentric Memory Retrieval\"", "authors": "Xiaodong Wang, Xuanyi Zhao, Pedro Rodriguez, Devendra Singh Sachan, Barlas Oguz, Seungwhan Moon, Shang-Wen Li, Gargi Ghosh, et al.", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2607.02689", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-07-02", "updated_at": "2026-07-02", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "memory", "rag", "reasoning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CV", "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "16", "collection_queries": "ai-agent" } }, { "collection": "papers", "path": "papers/items/2026-2607-02703-llmoxie-exploring-agentic-ai-for-scientific-software-development.md", "title": "\"LLMoxie: Exploring Agentic AI for Scientific Software Development\"", "type": "paper", "meta": { "type": "paper", "title": "\"LLMoxie: Exploring Agentic AI for Scientific Software Development\"", "authors": "Landung Setiawan, Anant Mittal, Cordero Core, Anshul Tambay, Carlos Garcia Jurado Suarez, David A. C. Beck, Andrew J. Connolly, Vani Mandava", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2607.02703", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-07-02", "updated_at": "2026-07-02", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "coding-agent", "planning", "reasoning", "workflow-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.SE", "cs.AI", "cs.DC", "cs.MA" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "18", "collection_queries": "agentic-ai, coding-agent" } }, { "collection": "papers", "path": "papers/items/2026-2607-02716-evaluating-large-language-models-for-decision-making-in-agent-based-urban-mobili.md", "title": "Evaluating Large Language Models for Decision-Making in Agent-Based Urban Mobility Simulations", "type": "paper", "meta": { "type": "paper", "title": "Evaluating Large Language Models for Decision-Making in Agent-Based Urban Mobility Simulations", "authors": "Bruno Cascaes Alves, Míriam Blank Born, Ulisses Gilioli Francescatto Júnior, Felipe Moura Goulart, Letícia Brandão Caldas, Marilton Sanchotene de Aguiar", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2607.02716", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-07-02", "updated_at": "2026-07-02", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "computer-use", "memory", "multi-agent", "planning", "tool-use", "world-model" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.MA" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "16", "collection_queries": "multi-agent-llm" } }, { "collection": "papers", "path": "papers/items/2026-2607-02807-swarmresearch-orchestrating-coding-agents-for-open-ended-discovery.md", "title": "\"SwarmResearch: Orchestrating Coding Agents for Open-Ended Discovery\"", "type": "paper", "meta": { "type": "paper", "title": "\"SwarmResearch: Orchestrating Coding Agents for Open-Ended Discovery\"", "authors": "Yuvraj Virk, Zack Edds, Chunqiu Steven Xia, Lingming Zhang", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2607.02807", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-07-02", "updated_at": "2026-07-02", "status": "queued", "relevance": "high", "topics": [ "coding-agent", "computer-use", "multi-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "15", "collection_queries": "coding-agent, multi-agent-llm" } }, { "collection": "papers", "path": "papers/items/2026-2607-02846-object-centric-environment-modeling-for-agentic-tasks.md", "title": "Object-Centric Environment Modeling for Agentic Tasks", "type": "paper", "meta": { "type": "paper", "title": "Object-Centric Environment Modeling for Agentic Tasks", "authors": "Yiyang Li, Tianyi Ma, Zehong Wang, Yijun Ma, Yanfang Ye", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2607.02846", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-07-03", "updated_at": "2026-07-03", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "rag", "tool-use", "world-model" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "15", "collection_queries": "llm-agent" } }, { "collection": "papers", "path": "papers/items/2026-2607-02857-mosaic-knowledge-guided-cli-command-composition-attack-in-llm-coding-agents.md", "title": "\"MOSAIC: Knowledge-Guided CLI Command Composition Attack in LLM Coding Agents\"", "type": "paper", "meta": { "type": "paper", "title": "\"MOSAIC: Knowledge-Guided CLI Command Composition Attack in LLM Coding Agents\"", "authors": "Jiangrong Wu, Huaijin Wang, Yihao Zhang, Yuhong Nan, Shuai Wang", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2607.02857", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-07-03", "updated_at": "2026-07-03", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "coding-agent", "computer-use", "workflow-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CR", "cs.SE" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "17", "collection_queries": "agent-evaluation, coding-agent" } }, { "collection": "papers", "path": "papers/items/2026-2607-02879-medcalc-pro-solving-complex-medical-calculations-with-llm-agents.md", "title": "\"MedCalc-Pro: Solving Complex Medical Calculations with LLM Agents\"", "type": "paper", "meta": { "type": "paper", "title": "\"MedCalc-Pro: Solving Complex Medical Calculations with LLM Agents\"", "authors": "Siran Zhao, Ruihui Hou, Ziyue Huai, Chennuo Zhang, Tong Ruan", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2607.02879", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-07-03", "updated_at": "2026-07-03", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "14", "collection_queries": "llm-agent" } }, { "collection": "papers", "path": "papers/items/2026-2607-02882-diagnosis-driven-automatic-repair-for-agentic-workflow-via-symbolic-inference.md", "title": "Diagnosis-Driven Automatic Repair for Agentic Workflow via Symbolic Inference", "type": "paper", "meta": { "type": "paper", "title": "Diagnosis-Driven Automatic Repair for Agentic Workflow via Symbolic Inference", "authors": "Xuyan Ma, Yawen Wang, Junjie Wang, Xiaofei Xie, Boyu Wu, Mingyang Li, Dandan Wang, Qing Wang", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2607.02882", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-07-03", "updated_at": "2026-07-03", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "planning", "tool-use", "workflow-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.SE" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "14", "collection_queries": "agentic-ai" } }, { "collection": "papers", "path": "papers/items/2026-2607-02911-coact-action-preserving-observation-compression-for-coding-agents.md", "title": "\"CoACT: Action-Preserving Observation Compression for Coding Agents\"", "type": "paper", "meta": { "type": "paper", "title": "\"CoACT: Action-Preserving Observation Compression for Coding Agents\"", "authors": "Haorui Chen, Yuancheng Zhu, Yitong Zhang, Jia Li", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2607.02911", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-07-03", "updated_at": "2026-07-03", "status": "queued", "relevance": "high", "topics": [ "coding-agent", "rag", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.SE", "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "15", "collection_queries": "coding-agent" } }, { "collection": "papers", "path": "papers/items/2026-2607-02927-videosearcher-empowering-video-deep-research-with-multi-tool-agentic-reasoning-v.md", "title": "\"VideoSearcher: Empowering Video Deep Research with Multi-Tool Agentic Reasoning via Reinforcement Learning\"", "type": "paper", "meta": { "type": "paper", "title": "\"VideoSearcher: Empowering Video Deep Research with Multi-Tool Agentic Reasoning via Reinforcement Learning\"", "authors": "Zhenkun Gao, Yicheng Bao, Jinlong Peng, Xueheng Li, Theo Huang, Bangwei Liu, Kunquan Li, Zhenye Gan, et al.", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2607.02927", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-07-03", "updated_at": "2026-07-03", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "rag", "reasoning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CV", "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "16", "collection_queries": "tool-use" } }, { "collection": "papers", "path": "papers/items/2026-2607-02942-a-workflow-aware-serving-layer-for-agentic-applications.md", "title": "A Workflow-Aware Serving Layer for Agentic Applications", "type": "paper", "meta": { "type": "paper", "title": "A Workflow-Aware Serving Layer for Agentic Applications", "authors": "Jiayi Qian, Zishen Wan, Hanchen Yang, Chun Tao, Souvik Kundu, Tushar Krishna", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2607.02942", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-07-03", "updated_at": "2026-07-03", "status": "queued", "relevance": "high", "topics": [ "reasoning", "tool-use", "workflow-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.DC", "cs.MA" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "13", "collection_queries": "agentic-ai" } }, { "collection": "papers", "path": "papers/items/2026-2607-03105-orbit-q-dual-axis-benchmarking-of-autonomous-agents-in-scientific-quantum-progra.md", "title": "\"ORBIT-Q: Dual-axis benchmarking of autonomous agents in scientific quantum programming\"", "type": "paper", "meta": { "type": "paper", "title": "\"ORBIT-Q: Dual-axis benchmarking of autonomous agents in scientific quantum programming\"", "authors": "Shi-Xin Zhang, Yu-Qin Chen", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2607.03105", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-07-03", "updated_at": "2026-07-03", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "coding-agent", "workflow-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "quant-ph" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "13", "collection_queries": "coding-agent" } }, { "collection": "papers", "path": "papers/items/2026-2607-03162-apeb-benchmarking-personalization-ability-of-large-language-model-agents.md", "title": "\"APeB: Benchmarking Personalization Ability of Large Language Model Agents\"", "type": "paper", "meta": { "type": "paper", "title": "\"APeB: Benchmarking Personalization Ability of Large Language Model Agents\"", "authors": "Garry Yang, Zizhe Chen, Xinru Chen, Yongqiang Chen, Jianxiang Wang, Deyu Zou, Linyi Ding, Jialiang Wu, et al.", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2607.03162", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-07-03", "updated_at": "2026-07-03", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "tool-use", "workflow-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI", "cs.HC" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "14", "collection_queries": "agentic-ai" } }, { "collection": "papers", "path": "papers/items/2026-2607-03220-contra-red-teaming-configurations-of-personalizable-agents.md", "title": "\"CONTRA: Red-Teaming Configurations of Personalizable Agents\"", "type": "paper", "meta": { "type": "paper", "title": "\"CONTRA: Red-Teaming Configurations of Personalizable Agents\"", "authors": "Jonathan Nöther, Adish Singla, Goran Radanovic", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2607.03220", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-07-03", "updated_at": "2026-07-03", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "coding-agent", "reasoning", "tool-use", "workflow-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CR", "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "13", "collection_queries": "autonomous-agent-llm" } }, { "collection": "papers", "path": "papers/items/2026-2607-03233-agentic-and-generative-ai-for-open-source-intelligence-and-cyber-investigations-.md", "title": "\"Agentic and Generative AI for Open-Source Intelligence and Cyber Investigations: Taxonomy, Evaluation, Challenges, and Future Directions\"", "type": "paper", "meta": { "type": "paper", "title": "\"Agentic and Generative AI for Open-Source Intelligence and Cyber Investigations: Taxonomy, Evaluation, Challenges, and Future Directions\"", "authors": "Eduardo Almeida Palmieri, Mohamed Chahine Ghanem, Dipo Dunsin, Zubair Baig, Ed de Quincey, Kim-Kwang Raymond Choo", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2607.03233", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-07-03", "updated_at": "2026-07-03", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "rag", "reasoning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CR", "cs.AI", "cs.IR", "cs.SI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "22", "collection_queries": "agentic-ai, rag-agent, tool-use" } }, { "collection": "papers", "path": "papers/items/2026-2607-03269-agentic-secpbft-agentic-ai-driven-proactive-security-framework-for-wireless-pbft.md", "title": "\"Agentic-SecPBFT: Agentic AI-Driven Proactive Security Framework for Wireless PBFT Consensus in Mobile Ad-Hoc Networks\"", "type": "paper", "meta": { "type": "paper", "title": "\"Agentic-SecPBFT: Agentic AI-Driven Proactive Security Framework for Wireless PBFT Consensus in Mobile Ad-Hoc Networks\"", "authors": "Haoxiang Luo, Yinqiu Liu, Ruichen Zhang, Guangyuan Liu, Gang Sun, Hongfang Yu, Zhu Han, Dong In Kim", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2607.03269", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-07-03", "updated_at": "2026-07-03", "status": "queued", "relevance": "high", "topics": [ "agent-safety", "computer-use", "multi-agent", "rag", "tool-use", "world-model" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.NI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "15", "collection_queries": "agentic-ai" } }, { "collection": "papers", "path": "papers/items/2026-2607-03316-is-agentic-code-review-helpful-mining-developers-feedback-to-coderabbit-reviews-.md", "title": "\"Is Agentic Code Review Helpful? Mining Developers' Feedback to CodeRabbit Reviews in the Wild\"", "type": "paper", "meta": { "type": "paper", "title": "\"Is Agentic Code Review Helpful? Mining Developers' Feedback to CodeRabbit Reviews in the Wild\"", "authors": "Hong Yi Lin, Mingzhao Liang, Kla Tantithamthavorn, Patanamon Thongtanunam", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2607.03316", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-07-03", "updated_at": "2026-07-03", "status": "queued", "relevance": "high", "topics": [ "agent-safety", "coding-agent", "workflow-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.SE", "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "13", "collection_queries": "autonomous-agent-llm" } }, { "collection": "papers", "path": "papers/items/2026-2607-03333-spork-self-speculative-forking-to-accelerate-agentic-llm-inference.md", "title": "\"SPORK: Self-Speculative Forking to Accelerate Agentic LLM Inference\"", "type": "paper", "meta": { "type": "paper", "title": "\"SPORK: Self-Speculative Forking to Accelerate Agentic LLM Inference\"", "authors": "Huajun Bai, Weiwei Lv, Huichuan Zheng, Youyou Lu, Jiwu Shu", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2607.03333", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-07-03", "updated_at": "2026-07-03", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "coding-agent", "reasoning", "tool-use", "workflow-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.DC", "cs.AI", "cs.LG" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "17", "collection_queries": "llm-agent" } }, { "collection": "papers", "path": "papers/items/2026-2607-03423-securing-multi-tool-ai-agent-chains-with-dynamic-real-time-compositional-policie.md", "title": "Securing Multi-Tool AI Agent Chains With Dynamic, Real-Time Compositional Policies", "type": "paper", "meta": { "type": "paper", "title": "Securing Multi-Tool AI Agent Chains With Dynamic, Real-Time Compositional Policies", "authors": "Chris Schneider, Kriti Faujdar, Philipp Schoenegger, Ben Bariach", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2607.03423", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-07-03", "updated_at": "2026-07-03", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "coding-agent", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CR", "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "14", "collection_queries": "ai-agent, coding-agent" } }, { "collection": "papers", "path": "papers/items/2026-2607-03441-no-time-like-the-present-agentic-test-time-training-for-llm-agents.md", "title": "\"No Time Like the Present: Agentic Test-Time Training for LLM Agents\"", "type": "paper", "meta": { "type": "paper", "title": "\"No Time Like the Present: Agentic Test-Time Training for LLM Agents\"", "authors": "Yanbo Wang, Jinhua Hao, Yuze Shi, Kun Yuan, Ming Sun", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2607.03441", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-07-03", "updated_at": "2026-07-03", "status": "queued", "relevance": "high", "topics": [ "coding-agent", "computer-use", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.LG", "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "18", "collection_queries": "coding-agent, llm-agent" } }, { "collection": "papers", "path": "papers/items/2026-2607-03510-cage-1-control-assurance-and-governance-evaluation-for-enterprise-agentic-ai.md", "title": "\"CAGE-1: Control, Assurance, and Governance Evaluation for Enterprise Agentic AI\"", "type": "paper", "meta": { "type": "paper", "title": "\"CAGE-1: Control, Assurance, and Governance Evaluation for Enterprise Agentic AI\"", "authors": "Roopam W. Sure", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2607.03510", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-07-03", "updated_at": "2026-07-03", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "memory", "planning", "rag", "tool-use", "workflow-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.SE", "cs.AI", "cs.CY" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "19", "collection_queries": "agentic-ai" } }, { "collection": "papers", "path": "papers/items/2026-2607-03525-gameenginebench-evaluating-coding-agents-on-real-c-runtime-environments.md", "title": "\"GameEngineBench: Evaluating Coding Agents on Real C++ Runtime Environments\"", "type": "paper", "meta": { "type": "paper", "title": "\"GameEngineBench: Evaluating Coding Agents on Real C++ Runtime Environments\"", "authors": "Brian La, Sejoon Chang, Ben Kim, Junyoung Bae, Aamish Ahmad Beg, Sei Chang, Gonzalo Gonzalez-Pumariega", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2607.03525", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-07-03", "updated_at": "2026-07-03", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "coding-agent", "embodied-agent", "memory", "tool-use", "world-model" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.SE", "cs.CL" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "16", "collection_queries": "coding-agent" } }, { "collection": "papers", "path": "papers/items/2026-2607-03601-archeval-measuring-ai-agents-as-computer-architects.md", "title": "\"ArchEval: Measuring AI Agents as Computer Architects\"", "type": "paper", "meta": { "type": "paper", "title": "\"ArchEval: Measuring AI Agents as Computer Architects\"", "authors": "Chenyu Wang, Zishen Wan, Jeffrey Ma, Shvetank Prakash, Zhenting Qi, Haebin Do, Andy Cheng, Arya Tschand, et al.", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2607.03601", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-07-03", "updated_at": "2026-07-03", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "memory", "reasoning", "tool-use", "workflow-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AR" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "21", "collection_queries": "ai-agent, llm-agent, tool-use" } }, { "collection": "papers", "path": "papers/items/2026-2607-03628-swarm-driven-multi-agent-reasoning-for-smart-city-security.md", "title": "Swarm-Driven Multi-Agent Reasoning for Smart City Security", "type": "paper", "meta": { "type": "paper", "title": "Swarm-Driven Multi-Agent Reasoning for Smart City Security", "authors": "Saeid Jamshidi, Kawser Wazed Nafi, Carol Fung, Foutse Khomh", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2607.03628", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-07-03", "updated_at": "2026-07-03", "status": "queued", "relevance": "high", "topics": [ "agent-safety", "multi-agent", "reasoning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CR", "cs.MA" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "14", "collection_queries": "multi-agent-llm" } }, { "collection": "papers", "path": "papers/items/2026-2607-03691-don-t-blame-the-large-language-model-how-scaffolding-evolution-shapes-coding-age.md", "title": "\"Don't Blame the Large Language Model: How Scaffolding Evolution Shapes Coding Agent Quality\"", "type": "paper", "meta": { "type": "paper", "title": "\"Don't Blame the Large Language Model: How Scaffolding Evolution Shapes Coding Agent Quality\"", "authors": "Oussama Ben Sghaier, Hao Li, Bram Adams, Ahmed E. Hassan", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2607.03691", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-07-04", "updated_at": "2026-07-04", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "coding-agent", "reasoning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.SE", "cs.AI", "cs.LG" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "16", "collection_queries": "coding-agent" } }, { "collection": "papers", "path": "papers/items/2026-2607-03695-social-networks-of-llm-agents.md", "title": "Social Networks of LLM Agents", "type": "paper", "meta": { "type": "paper", "title": "Social Networks of LLM Agents", "authors": "Kaixuan Liu, Guojun Xiong, Weinan Zhang, Shengpu Tang", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2607.03695", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-07-04", "updated_at": "2026-07-04", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "multi-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.LG" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "15", "collection_queries": "llm-agent, multi-agent-llm" } }, { "collection": "papers", "path": "papers/items/2026-2607-03702-agent-reinforcement-learning-via-pivotal-aware-self-feedback-retry.md", "title": "Agent Reinforcement Learning via Pivotal-Aware Self-Feedback Retry", "type": "paper", "meta": { "type": "paper", "title": "Agent Reinforcement Learning via Pivotal-Aware Self-Feedback Retry", "authors": "Weiyang Guo, Zesheng Shi, Longhui Zhang, Zeen Zhu, Min Zhang, Jing Li", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2607.03702", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-07-04", "updated_at": "2026-07-04", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "planning", "rag", "reasoning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "15", "collection_queries": "llm-agent" } }, { "collection": "papers", "path": "papers/items/2026-2607-03726-selfmem-self-optimizing-memory-for-ai-agents.md", "title": "\"SelfMem: Self-Optimizing Memory for AI Agents\"", "type": "paper", "meta": { "type": "paper", "title": "\"SelfMem: Self-Optimizing Memory for AI Agents\"", "authors": "Shu Yang, Junchao Wu, Derek F. Wong, Di Wang", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2607.03726", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-07-04", "updated_at": "2026-07-04", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "computer-use", "memory", "planning", "rag", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CL" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "20", "collection_queries": "agent-memory, ai-agent, tool-use" } }, { "collection": "papers", "path": "papers/items/2026-2607-03821-dualview-preventing-indirect-prompt-injection-in-personal-ai-agents.md", "title": "\"DualView: Preventing Indirect Prompt Injection in Personal AI Agents\"", "type": "paper", "meta": { "type": "paper", "title": "\"DualView: Preventing Indirect Prompt Injection in Personal AI Agents\"", "authors": "Juhee Kim, Woohyuk Choi, Taehyun Kang, Youngmin Kim, Byoungyoung Lee", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2607.03821", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-07-04", "updated_at": "2026-07-04", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CR", "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "15", "collection_queries": "ai-agent" } }, { "collection": "papers", "path": "papers/items/2026-2607-03853-cograd-a-cognitively-inspired-multi-agent-framework-for-radiology-report-generat.md", "title": "\"CogRad: A Cognitively-Inspired Multi-Agent Framework for Radiology Report Generation\"", "type": "paper", "meta": { "type": "paper", "title": "\"CogRad: A Cognitively-Inspired Multi-Agent Framework for Radiology Report Generation\"", "authors": "Saif Ur Rehman Khan, Hasaan Maqsood, Sebastian Vollmer, Andreas Dengel, Muhammad Nabeel Asim", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2607.03853", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-07-04", "updated_at": "2026-07-04", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "multi-agent", "rag", "reasoning" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CV" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "14", "collection_queries": "multi-agent-llm" } }, { "collection": "papers", "path": "papers/items/2026-2607-03953-the-remarkable-effectiveness-of-providing-ai-agents-with-natural-language-tools-.md", "title": "\"The Remarkable Effectiveness of Providing AI Agents with Natural Language Tools: A Replication Study Validating NLT Performance Across 14 Models\"", "type": "paper", "meta": { "type": "paper", "title": "\"The Remarkable Effectiveness of Providing AI Agents with Natural Language Tools: A Replication Study Validating NLT Performance Across 14 Models\"", "authors": "Alexander Somma, Isabelle Plante, Fred Premji", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2607.03953", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-07-04", "updated_at": "2026-07-04", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "rag", "reasoning", "tool-use", "workflow-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CL", "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "20", "collection_queries": "agentic-ai, ai-agent, llm-agent, tool-use" } }, { "collection": "papers", "path": "papers/items/2026-2607-03968-refused-in-chat-written-in-code-workflow-level-jailbreak-construction-in-ide-cod.md", "title": "\"Refused in Chat, Written in Code: Workflow-Level Jailbreak Construction in IDE Coding Agents\"", "type": "paper", "meta": { "type": "paper", "title": "\"Refused in Chat, Written in Code: Workflow-Level Jailbreak Construction in IDE Coding Agents\"", "authors": "Abhishek Kumar, Carsten Maple", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2607.03968", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-07-04", "updated_at": "2026-07-04", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "coding-agent", "workflow-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.SE", "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "15", "collection_queries": "coding-agent" } }, { "collection": "papers", "path": "papers/items/2026-2607-04009-physminer-an-agentic-ai-framework-for-discovering-turbulence-physics.md", "title": "\"PhysMiner: An Agentic AI Framework for Discovering Turbulence Physics\"", "type": "paper", "meta": { "type": "paper", "title": "\"PhysMiner: An Agentic AI Framework for Discovering Turbulence Physics\"", "authors": "Jiawei Chen, Han Gao, Ping He", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2607.04009", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-07-04", "updated_at": "2026-07-04", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "memory", "reasoning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "physics.flu-dyn" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "15", "collection_queries": "agentic-ai" } }, { "collection": "papers", "path": "papers/items/2026-2607-04034-the-i-don-t-know-filter-enhancing-agentic-reliability-in-function-calling.md", "title": "\"The \\\"I Don't Know\\\" Filter: Enhancing Agentic Reliability in Function Calling\"", "type": "paper", "meta": { "type": "paper", "title": "\"The \\\"I Don't Know\\\" Filter: Enhancing Agentic Reliability in Function Calling\"", "authors": "Stefan Broecker, Mason del Rosario, Boris Selitser, Thomas Strohmer", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2607.04034", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-07-04", "updated_at": "2026-07-04", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "rag", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.SE", "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "16", "collection_queries": "agent-evaluation, function-calling" } }, { "collection": "papers", "path": "papers/items/2026-2607-04089-placemem-toward-a-compute-aware-memory-plane-for-lifelong-agents.md", "title": "\"PLACEMEM: Toward a Compute-Aware Memory Plane for Lifelong Agents\"", "type": "paper", "meta": { "type": "paper", "title": "\"PLACEMEM: Toward a Compute-Aware Memory Plane for Lifelong Agents\"", "authors": "Sukanta Ganguly", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2607.04089", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-07-05", "updated_at": "2026-07-05", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "memory", "planning", "rag" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "15", "collection_queries": "agent-memory" } }, { "collection": "papers", "path": "papers/items/2026-2607-04149-beyond-scene-priors-fine-grained-traffic-scene-reasoning-with-benchmarking-and-q.md", "title": "\"Beyond Scene Priors: Fine-Grained Traffic Scene Reasoning with Benchmarking and Query-Guided Small-Object Focus\"", "type": "paper", "meta": { "type": "paper", "title": "\"Beyond Scene Priors: Fine-Grained Traffic Scene Reasoning with Benchmarking and Query-Guided Small-Object Focus\"", "authors": "Waikit Xiu, Qiang Lu, Zian Wang, Xinjie Yang, Zhiwei Chen, Chen Sun, Xiying Li", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2607.04149", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-07-05", "updated_at": "2026-07-05", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "coding-agent", "computer-use", "multi-agent", "reasoning" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CV" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "16", "collection_queries": "multi-agent-llm" } }, { "collection": "papers", "path": "papers/items/2026-2607-04162-ace-agentic-control-for-embodied-manipulation-via-zero-shot-workflow-reasoning.md", "title": "\"ACE: Agentic Control for Embodied Manipulation via Zero-shot Workflow Reasoning\"", "type": "paper", "meta": { "type": "paper", "title": "\"ACE: Agentic Control for Embodied Manipulation via Zero-shot Workflow Reasoning\"", "authors": "Iok Tong Lei, QianZhi Li, Ying Jie Yap, Yujie Zhang, Rui Zhong, Haichao Gui, Xiaolong Liu, Zhidong Deng", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2607.04162", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-07-05", "updated_at": "2026-07-05", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "embodied-agent", "memory", "planning", "rag", "reasoning", "tool-use", "workflow-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.RO", "cs.LG" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "17", "collection_queries": "agentic-ai" } }, { "collection": "papers", "path": "papers/items/2026-2607-04212-an-evaluation-of-role-based-multi-agent-code-generation-on-repository-scale-prob.md", "title": "An Evaluation of Role-Based Multi-Agent Code Generation on Repository-Scale Problems", "type": "paper", "meta": { "type": "paper", "title": "An Evaluation of Role-Based Multi-Agent Code Generation on Repository-Scale Problems", "authors": "Benedetta Donato, Noah Hagar-Dent, Aaron Worsnop, Leonardo Mariani, Valerio Terragni", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2607.04212", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-07-05", "updated_at": "2026-07-05", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "coding-agent", "multi-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.SE" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "16", "collection_queries": "multi-agent-llm" } }, { "collection": "papers", "path": "papers/items/2026-2607-04219-agentic-iot-architectures-applications-and-challenges-toward-the-internet-of-age.md", "title": "\"Agentic IoT: Architectures, Applications, and Challenges Toward the Internet of Agents\"", "type": "paper", "meta": { "type": "paper", "title": "\"Agentic IoT: Architectures, Applications, and Challenges Toward the Internet of Agents\"", "authors": "Rümeysa Hilal Sevinç, Bahaeddin Türkoğlu, İbrahim Kök", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2607.04219", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-07-05", "updated_at": "2026-07-05", "status": "queued", "relevance": "high", "topics": [ "multi-agent", "planning", "reasoning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI", "cs.MA", "cs.NI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "16", "collection_queries": "ai-agent, tool-use" } }, { "collection": "papers", "path": "papers/items/2026-2607-04240-biological-motifs-for-agentic-control.md", "title": "Biological Motifs for Agentic Control", "type": "paper", "meta": { "type": "paper", "title": "Biological Motifs for Agentic Control", "authors": "Bogdan Banu", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2607.04240", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-07-05", "updated_at": "2026-07-05", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "multi-agent", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI", "q-bio.CB" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "18", "collection_queries": "agent-evaluation, autonomous-agent-llm, multi-agent-llm" } }, { "collection": "papers", "path": "papers/items/2026-2607-04293-causalgame-benchmarking-causal-thinking-of-llm-agents-in-games.md", "title": "\"CausalGame: Benchmarking Causal Thinking of LLM Agents in Games\"", "type": "paper", "meta": { "type": "paper", "title": "\"CausalGame: Benchmarking Causal Thinking of LLM Agents in Games\"", "authors": "Zhenhao Chen, Yongqiang Chen, Chenxi Liu, Junchi Yu, Xiangchen Song, Zijian Li, Jialin Li, Philip Torr, et al.", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2607.04293", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-07-05", "updated_at": "2026-07-05", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "computer-use", "planning", "reasoning" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CL", "cs.AI", "cs.LG", "stat.ML" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "17", "collection_queries": "llm-agent" } }, { "collection": "papers", "path": "papers/items/2026-2607-04334-do-gui-agents-believe-their-eyes-diagnosing-state-belief-reliance-on-pixels-vers.md", "title": "Do GUI Agents Believe Their Eyes? Diagnosing State-Belief Reliance on Pixels versus Structure", "type": "paper", "meta": { "type": "paper", "title": "Do GUI Agents Believe Their Eyes? Diagnosing State-Belief Reliance on Pixels versus Structure", "authors": "Guijia Zhang, Harry Yang", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2607.04334", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-07-05", "updated_at": "2026-07-05", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "computer-use", "rag", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "13", "collection_queries": "web-gui-agent" } }, { "collection": "papers", "path": "papers/items/2026-2607-04391-memory-orchestrated-semantic-system-moss-an-auditable-agentic-memory-architectur.md", "title": "\"Memory-Orchestrated Semantic System (MOSS): An Auditable Agentic Memory Architecture\"", "type": "paper", "meta": { "type": "paper", "title": "\"Memory-Orchestrated Semantic System (MOSS): An Auditable Agentic Memory Architecture\"", "authors": "Serge Lacasse, Jérémie Hatier, Alex Baker", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2607.04391", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-07-05", "updated_at": "2026-07-05", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "memory", "rag", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CL" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "20", "collection_queries": "agent-memory, ai-agent, rag-agent" } }, { "collection": "papers", "path": "papers/items/2026-2607-04394-mechmath-agent-team-llm-driven-agents-for-mathematical-research.md", "title": "\"MechMath Agent Team: LLM Driven Agents for Mathematical Research\"", "type": "paper", "meta": { "type": "paper", "title": "\"MechMath Agent Team: LLM Driven Agents for Mathematical Research\"", "authors": "Yichuan Cao, Ruichen Qiu, Junqi Liu, Jiaqi Wang, Dakai Guo, Ruyong Feng, Lihong Zhi, Xiao-Shan Gao", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2607.04394", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-07-05", "updated_at": "2026-07-05", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "multi-agent", "planning", "rag", "reasoning" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI", "cs.SC" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "13", "collection_queries": "multi-agent-llm" } }, { "collection": "papers", "path": "papers/items/2026-2607-04395-nki-agent-domain-specific-fine-tuning-and-agentic-tool-use-for-neuron-kernel-gen.md", "title": "\"NKI-Agent: Domain-Specific Fine-Tuning and Agentic Tool Use for Neuron Kernel Generation\"", "type": "paper", "meta": { "type": "paper", "title": "\"NKI-Agent: Domain-Specific Fine-Tuning and Agentic Tool Use for Neuron Kernel Generation\"", "authors": "Junjie Tang, Jun Huan, Hao Zhou, Yuhao Zhang, Lin Wang", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2607.04395", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-07-05", "updated_at": "2026-07-05", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "computer-use", "memory", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.LG" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "19", "collection_queries": "tool-use" } }, { "collection": "papers", "path": "papers/items/2026-2607-04426-ace-brain-0-5-a-unified-embodied-foundational-model-for-physical-agentic-ai.md", "title": "\"ACE-Brain-0.5: A Unified Embodied Foundational Model for Physical Agentic AI\"", "type": "paper", "meta": { "type": "paper", "title": "\"ACE-Brain-0.5: A Unified Embodied Foundational Model for Physical Agentic AI\"", "authors": "\"ACE-Brain Team, :, Ziyang Gong, Haoming Gu, Zehang Luo, Tianyi Zhang, Tao Tao, Yixiao Chi, et al.\"", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2607.04426", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-07-05", "updated_at": "2026-07-05", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "embodied-agent", "memory", "planning", "rag", "reasoning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.RO" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "17", "collection_queries": "agentic-ai" } }, { "collection": "papers", "path": "papers/items/2026-2607-04433-autonomous-information-seeking-a-roadmap-for-agentic-recommender-systems.md", "title": "\"Autonomous Information Seeking: A Roadmap for Agentic Recommender Systems\"", "type": "paper", "meta": { "type": "paper", "title": "\"Autonomous Information Seeking: A Roadmap for Agentic Recommender Systems\"", "authors": "Xinyu Lin, Yashar Deldjoo, Sunhao Dai, Honghui Bao, Xiaopeng Ye, Fatemeh Nazary, Wenjie Wang, Tommaso Di Noia, et al.", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2607.04433", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-07-05", "updated_at": "2026-07-05", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "memory", "planning", "reasoning", "tool-use", "workflow-agent", "world-model" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.IR", "cs.CL" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "21", "collection_queries": "tool-use" } }, { "collection": "papers", "path": "papers/items/2026-2607-04470-regime-conditional-stabilisation-of-llm-augmented-cooperative-multi-agent-reinfo.md", "title": "Regime-Conditional Stabilisation of LLM-Augmented Cooperative Multi-Agent Reinforcement Learning", "type": "paper", "meta": { "type": "paper", "title": "Regime-Conditional Stabilisation of LLM-Augmented Cooperative Multi-Agent Reinforcement Learning", "authors": "Faid Keddouri, Sohaib Houhou, Aissa Boulmerka, Nadir Farhi", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2607.04470", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-07-05", "updated_at": "2026-07-05", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "multi-agent", "rag", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.LG", "cs.AI", "math.OC" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "14", "collection_queries": "multi-agent-llm" } }, { "collection": "papers", "path": "papers/items/2026-2607-04528-measuring-harness-induced-belief-divergence-in-multi-step-llm-agents.md", "title": "Measuring Harness-Induced Belief Divergence in Multi-Step LLM Agents", "type": "paper", "meta": { "type": "paper", "title": "Measuring Harness-Induced Belief Divergence in Multi-Step LLM Agents", "authors": "Haiwen Yi, Xinyuan Song", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2607.04528", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-07-05", "updated_at": "2026-07-05", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "coding-agent", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "17", "collection_queries": "agent-evaluation, llm-agent" } }, { "collection": "papers", "path": "papers/items/2026-2607-04569-llms-for-agentic-home-energy-management.md", "title": "LLMs for Agentic Home Energy Management", "type": "paper", "meta": { "type": "paper", "title": "LLMs for Agentic Home Energy Management", "authors": "Sokipriala Jonah", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2607.04569", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-07-06", "updated_at": "2026-07-06", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "rag", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "eess.SY" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "16", "collection_queries": "function-calling, llm-agent" } }, { "collection": "papers", "path": "papers/items/2026-2607-04617-mrms-a-multi-resolution-memory-substrate-for-long-lived-ai-agents.md", "title": "\"MRMS: A Multi-Resolution Memory Substrate for Long-Lived AI Agents\"", "type": "paper", "meta": { "type": "paper", "title": "\"MRMS: A Multi-Resolution Memory Substrate for Long-Lived AI Agents\"", "authors": "Jizhizi Li, Amy Shi-Nash", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2607.04617", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-07-06", "updated_at": "2026-07-06", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "computer-use", "memory", "rag", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "16", "collection_queries": "ai-agent" } }, { "collection": "papers", "path": "papers/items/2026-2607-04623-can-llms-really-recover-microservice-failures-a-recovery-aware-evaluation-of-dia.md", "title": "Can LLMs Really Recover Microservice Failures? A Recovery-Aware Evaluation of Diagnosis-to-Action Reasoning", "type": "paper", "meta": { "type": "paper", "title": "Can LLMs Really Recover Microservice Failures? A Recovery-Aware Evaluation of Diagnosis-to-Action Reasoning", "authors": "Jiaxing Qi, Zhongzhi Luan, Hongyu Zhang, Shaohan Huang, Carol Fung, Yongxin Tong, Hailong Yang, Depei Qian", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2607.04623", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-07-06", "updated_at": "2026-07-06", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "planning", "rag", "reasoning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.SE", "cs.DC" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "14", "collection_queries": "rag-agent" } }, { "collection": "papers", "path": "papers/items/2026-2607-04686-toolfailbench-diagnosing-tool-use-failures-in-llm-agents.md", "title": "\"ToolFailBench: Diagnosing Tool-Use Failures in LLM Agents\"", "type": "paper", "meta": { "type": "paper", "title": "\"ToolFailBench: Diagnosing Tool-Use Failures in LLM Agents\"", "authors": "Harsh Soni", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2607.04686", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-07-06", "updated_at": "2026-07-06", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CL", "cs.AI", "cs.SE" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "18", "collection_queries": "agentic-ai, llm-agent, tool-use" } }, { "collection": "papers", "path": "papers/items/2026-2607-04697-ai-agent-pull-requests-on-github-frequency-structure-and-merge-conflict-rates.md", "title": "\"AI Agent Pull Requests on GitHub: Frequency, Structure, and Merge Conflict Rates\"", "type": "paper", "meta": { "type": "paper", "title": "\"AI Agent Pull Requests on GitHub: Frequency, Structure, and Merge Conflict Rates\"", "authors": "George Xu, Arjun Subramanian, Nithilan Karthik", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2607.04697", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-07-06", "updated_at": "2026-07-07", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "coding-agent", "multi-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.SE" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "13", "collection_queries": "ai-agent, coding-agent" } }, { "collection": "papers", "path": "papers/items/2026-2607-04713-rspo-reward-swap-policy-optimization-for-multi-turn-llm-agents.md", "title": "\"RSPO: Reward-Swap Policy Optimization for Multi-Turn LLM Agents\"", "type": "paper", "meta": { "type": "paper", "title": "\"RSPO: Reward-Swap Policy Optimization for Multi-Turn LLM Agents\"", "authors": "Qiang Liu, Taian Guo, Ruizhi Qiao, Xing Sun", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2607.04713", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-07-06", "updated_at": "2026-07-06", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "planning", "rag" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.LG", "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "16", "collection_queries": "agent-evaluation, llm-agent" } }, { "collection": "papers", "path": "papers/items/2026-2607-04963-stapo-selective-trajectory-aware-policy-optimization-for-llm-agent-training.md", "title": "\"STAPO: Selective Trajectory-Aware Policy Optimization for LLM Agent Training\"", "type": "paper", "meta": { "type": "paper", "title": "\"STAPO: Selective Trajectory-Aware Policy Optimization for LLM Agent Training\"", "authors": "Qiuyi Qi, Tian Liang, Mutian Bao, Jinjian Zhang, Dongnan Liu, Wei Zhou, Linjian Mo, Ming Kong, et al.", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2607.04963", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-07-06", "updated_at": "2026-07-06", "status": "queued", "relevance": "high", "topics": [ "planning", "rag", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "14", "collection_queries": "llm-agent" } }, { "collection": "papers", "path": "papers/items/2026-2607-05001-tactic-kg-toward-small-agent-teams-for-cyber-threat-intelligence-knowledge-graph.md", "title": "\"TACTIC-KG: Toward Small Agent Teams for Cyber Threat Intelligence Knowledge Graph Construction\"", "type": "paper", "meta": { "type": "paper", "title": "\"TACTIC-KG: Toward Small Agent Teams for Cyber Threat Intelligence Knowledge Graph Construction\"", "authors": "Mouhamed Amine Bouchiha, Gregory Blanc", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2607.05001", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-07-06", "updated_at": "2026-07-07", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "reasoning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CR", "cs.AI", "cs.LG", "cs.MA" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "13", "collection_queries": "llm-agent" } }, { "collection": "papers", "path": "papers/items/2026-2607-05029-your-agent-s-memories-are-not-its-own-forged-reasoning-attacks-on-llm-agent-memo.md", "title": "\"Your Agent's Memories Are Not Its Own: Forged Reasoning Attacks on LLM Agent Memory and Defenses\"", "type": "paper", "meta": { "type": "paper", "title": "\"Your Agent's Memories Are Not Its Own: Forged Reasoning Attacks on LLM Agent Memory and Defenses\"", "authors": "Neeraj Karamchandani, Piyush Nagasubramaniam, Sencun Zhu, Dinghao Wu", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2607.05029", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-07-06", "updated_at": "2026-07-06", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "memory", "reasoning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CR", "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "19", "collection_queries": "agent-memory, llm-agent" } }, { "collection": "papers", "path": "papers/items/2026-2607-05055-toward-trustworthy-large-language-model-agents-in-healthcare.md", "title": "Toward Trustworthy Large Language Model Agents in Healthcare", "type": "paper", "meta": { "type": "paper", "title": "Toward Trustworthy Large Language Model Agents in Healthcare", "authors": "Hadi Hasan, Safaa Salman, Adam Tai Abou Dargham, Ammar Mohanna, Ali Chehab", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2607.05055", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-07-06", "updated_at": "2026-07-06", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "rag", "tool-use", "workflow-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "15", "collection_queries": "function-calling, rag-agent" } }, { "collection": "papers", "path": "papers/items/2026-2607-05120-agent-data-injection-attacks-are-realistic-threats-to-ai-agents.md", "title": "Agent Data Injection Attacks are Realistic Threats to AI Agents", "type": "paper", "meta": { "type": "paper", "title": "Agent Data Injection Attacks are Realistic Threats to AI Agents", "authors": "Woohyuk Choi, Juhee Kim, Taehyun Kang, Jihyeon Jeong, Luyi Xing, Byoungyoung Lee", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2607.05120", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-07-06", "updated_at": "2026-07-06", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "coding-agent", "computer-use", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CR", "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "19", "collection_queries": "agent-safety, ai-agent, coding-agent, web-gui-agent" } }, { "collection": "papers", "path": "papers/items/2026-2607-05132-when-agents-lie-premeditation-persistence-and-exploitation-in-repeated-games.md", "title": "\"When Agents Lie: Premeditation, Persistence, and Exploitation in Repeated Games\"", "type": "paper", "meta": { "type": "paper", "title": "\"When Agents Lie: Premeditation, Persistence, and Exploitation in Repeated Games\"", "authors": "Jerick Shi, Terry Jingcheng Zhang, Bernhard Schölkopf, Vincent Conitzer, Zhijing Jin", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2607.05132", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-07-06", "updated_at": "2026-07-07", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "planning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CY", "cs.CL" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "14", "collection_queries": "autonomous-agent-llm, llm-agent, planning-agent" } }, { "collection": "papers", "path": "papers/items/2026-2607-05174-agentgym2-benchmarking-large-language-model-agents-in-de-idealized-real-world-en.md", "title": "\"AgentGym2: Benchmarking Large Language Model Agents in De-Idealized Real-World Environments\"", "type": "paper", "meta": { "type": "paper", "title": "\"AgentGym2: Benchmarking Large Language Model Agents in De-Idealized Real-World Environments\"", "authors": "Zhiheng Xi, Dingwen Yang, Jiaqi Liu, Jixuan Huang, Honglin Guo, Baodai Huang, Tinggang Chen, Qi Zhang, et al.", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2607.05174", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-07-06", "updated_at": "2026-07-06", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "planning", "reasoning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "19", "collection_queries": "language-agent, llm-agent, planning-agent" } }, { "collection": "papers", "path": "papers/items/2026-2607-05188-latent-programming-horizons-in-coding-agents.md", "title": "Latent Programming Horizons in Coding Agents", "type": "paper", "meta": { "type": "paper", "title": "Latent Programming Horizons in Coding Agents", "authors": "André Silva, Han Tu, Martin Monperrus", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2607.05188", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-07-06", "updated_at": "2026-07-06", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "coding-agent", "reasoning" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.LG", "cs.SE" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "14", "collection_queries": "coding-agent" } }, { "collection": "papers", "path": "papers/items/2026-2607-05202-evoagentbench-benchmarking-agent-self-evolution-via-ability-transfer.md", "title": "\"EvoAgentBench: Benchmarking Agent Self-Evolution via Ability Transfer\"", "type": "paper", "meta": { "type": "paper", "title": "\"EvoAgentBench: Benchmarking Agent Self-Evolution via Ability Transfer\"", "authors": "Xingze Gao, Chuanrui Hu, Hongda Chen, Pengfei Yao, Zhao Wang, Yi Bai, Zhengwei Wu, Yunyun Han, et al.", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2607.05202", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-07-06", "updated_at": "2026-07-06", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "coding-agent", "computer-use", "memory", "planning", "reasoning" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "19", "collection_queries": "agent-evaluation" } }, { "collection": "papers", "path": "papers/items/2026-2607-05297-metaskill-evolve-recursive-self-improvement-of-llm-agents-via-two-timescale-meta.md", "title": "\"MetaSkill-Evolve: Recursive Self-Improvement of LLM Agents via Two-Timescale Meta-Skill Evolution\"", "type": "paper", "meta": { "type": "paper", "title": "\"MetaSkill-Evolve: Recursive Self-Improvement of LLM Agents via Two-Timescale Meta-Skill Evolution\"", "authors": "Zefeng Wang, Minxi Yan, Jinhe Bi, Sikuan Yan, Volker Tresp, Yunpu Ma", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2607.05297", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-07-06", "updated_at": "2026-07-06", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "planning", "reasoning", "workflow-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "17", "collection_queries": "agent-evaluation, llm-agent" } }, { "collection": "papers", "path": "papers/items/2026-2607-05318-pisas-benchmarking-contextual-integrity-in-multi-user-agentic-systems.md", "title": "\"PiSAs: Benchmarking Contextual Integrity in Multi-User Agentic Systems\"", "type": "paper", "meta": { "type": "paper", "title": "\"PiSAs: Benchmarking Contextual Integrity in Multi-User Agentic Systems\"", "authors": "Shubham Gupta, Nazanin Mohammadi Sepahvand, Abhinav Kumar, Cem Subakan, Spandana Gella, Pierre-André Noël, Perouz Taslakian, Eugene Bagdasarian, et al.", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2607.05318", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-07-06", "updated_at": "2026-07-06", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "memory", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.MA", "cs.CR" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "20", "collection_queries": "llm-agent" } }, { "collection": "papers", "path": "papers/items/2026-2607-05363-sovereignpa-bench-evaluating-user-owned-personal-agents-under-evolving-intent-pl.md", "title": "\"SovereignPA-Bench: Evaluating User-Owned Personal Agents under Evolving Intent, Platform Mediation, and Consent Constraints\"", "type": "paper", "meta": { "type": "paper", "title": "\"SovereignPA-Bench: Evaluating User-Owned Personal Agents under Evolving Intent, Platform Mediation, and Consent Constraints\"", "authors": "Dylan Zongmin Liu", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2607.05363", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-07-06", "updated_at": "2026-07-06", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "computer-use", "embodied-agent", "memory", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "16", "collection_queries": "agent-evaluation, tool-use" } }, { "collection": "papers", "path": "papers/items/2026-2607-05378-compactionrl-reinforcement-learning-with-context-compaction-for-long-horizon-age.md", "title": "\"CompactionRL: Reinforcement Learning with Context Compaction for Long-Horizon Agents\"", "type": "paper", "meta": { "type": "paper", "title": "\"CompactionRL: Reinforcement Learning with Context Compaction for Long-Horizon Agents\"", "authors": "Yujiang Li, Zhenyu Hou, Yi Jing, Jie Tang, Yuxiao Dong", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2607.05378", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-07-06", "updated_at": "2026-07-06", "status": "queued", "relevance": "high", "topics": [ "coding-agent", "planning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.LG" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "14", "collection_queries": "coding-agent, llm-agent" } }, { "collection": "papers", "path": "papers/items/2026-2607-05391-llm-as-a-verifier-a-general-purpose-verification-framework.md", "title": "\"LLM-as-a-Verifier: A General-Purpose Verification Framework\"", "type": "paper", "meta": { "type": "paper", "title": "\"LLM-as-a-Verifier: A General-Purpose Verification Framework\"", "authors": "Jacky Kwok, Shulu Li, Pranav Atreya, Yuejiang Liu, Yixing Jiang, Chelsea Finn, Marco Pavone, Ion Stoica, et al.", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2607.05391", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-07-06", "updated_at": "2026-07-07", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "coding-agent", "embodied-agent", "reasoning" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI", "cs.CL", "cs.LG", "cs.MA", "cs.RO" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "13", "collection_queries": "coding-agent" } }, { "collection": "papers", "path": "papers/items/2026-2607-05428-charlie-an-on-premise-multi-agent-retrieval-augmented-generation-system-for-evid.md", "title": "\"CHARLIE: An On-Premise Multi-Agent Retrieval-Augmented Generation System for Evidential Reasoning in Forensic Science\"", "type": "paper", "meta": { "type": "paper", "title": "\"CHARLIE: An On-Premise Multi-Agent Retrieval-Augmented Generation System for Evidential Reasoning in Forensic Science\"", "authors": "Leandro D. Carneiro, Andre L. S. Meirelles, Juliano de A. Gomes, Rafael C. A. Cabral", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2607.05428", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-07-01", "updated_at": "2026-07-01", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "memory", "multi-agent", "planning", "rag", "reasoning", "tool-use", "workflow-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.DL", "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "17", "collection_queries": "rag-agent" } }, { "collection": "papers", "path": "papers/items/2026-2607-05456-prompt-to-paper-agentic-ai-system-for-bioinformatics.md", "title": "\"Prompt-to-Paper: Agentic AI System for Bioinformatics\"", "type": "paper", "meta": { "type": "paper", "title": "\"Prompt-to-Paper: Agentic AI System for Bioinformatics\"", "authors": "Ramsha Kamran, Maheera Amjad, Zartasha Mustansar, Arsalan Shaukat, Salma Sherbaz, Muhammad U. S. Khan", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2607.05456", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-07-05", "updated_at": "2026-07-05", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "coding-agent", "multi-agent", "rag", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI", "cs.CL", "q-bio.QM" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "21", "collection_queries": "agentic-ai, coding-agent, multi-agent-llm" } }, { "collection": "papers", "path": "papers/items/2026-2607-05458-learning-to-control-llm-agent-harnesses-with-offline-reinforcement-learning.md", "title": "Learning to Control LLM Agent Harnesses with Offline Reinforcement Learning", "type": "paper", "meta": { "type": "paper", "title": "Learning to Control LLM Agent Harnesses with Offline Reinforcement Learning", "authors": "Haiwen Yi, Xinyuan Song", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2607.05458", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-07-05", "updated_at": "2026-07-05", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "coding-agent", "reasoning", "tool-use", "workflow-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.LG", "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "16", "collection_queries": "llm-agent" } }, { "collection": "papers", "path": "papers/items/2026-2607-05518-aiauthz-off-host-identity-bound-authorization-for-ai-agents.md", "title": "\"aiAuthZ: Off-Host, Identity-Bound Authorization for AI Agents\"", "type": "paper", "meta": { "type": "paper", "title": "\"aiAuthZ: Off-Host, Identity-Bound Authorization for AI Agents\"", "authors": "Sai Varun Kodathala", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2607.05518", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-07-06", "updated_at": "2026-07-06", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CR", "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "13", "collection_queries": "ai-agent" } }, { "collection": "papers", "path": "papers/items/2026-2607-05659-agents-with-feelings-personality-and-emotion-in-multi-agent-software-teams.md", "title": "Agents with Feelings? Personality and Emotion in Multi-Agent Software Teams", "type": "paper", "meta": { "type": "paper", "title": "Agents with Feelings? Personality and Emotion in Multi-Agent Software Teams", "authors": "Yunyan Ding, Thomas Zimmermann, Iftekhar Ahmed", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2607.05659", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-07-06", "updated_at": "2026-07-06", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "coding-agent", "multi-agent", "workflow-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.SE" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "16", "collection_queries": "llm-agent, multi-agent-llm" } }, { "collection": "papers", "path": "papers/items/2026-2607-05666-what-do-ai-agents-actually-change-an-empirical-taxonomy-of-mutation-patterns-in-.md", "title": "What Do AI Agents Actually Change? An Empirical Taxonomy of Mutation Patterns in Performance-Improving Pull Requests", "type": "paper", "meta": { "type": "paper", "title": "What Do AI Agents Actually Change? An Empirical Taxonomy of Mutation Patterns in Performance-Improving Pull Requests", "authors": "Illia Dovhoshliubnyi, Nima Soroush, Ashkan Sami, Alexander Brownlee", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2607.05666", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-07-06", "updated_at": "2026-07-06", "status": "queued", "relevance": "high", "topics": [ "coding-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.SE", "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "13", "collection_queries": "ai-agent, coding-agent" } }, { "collection": "papers", "path": "papers/items/2026-2607-05677-from-conversation-to-contribution-characterizing-coding-agent-in-open-source-sof.md", "title": "\"From Conversation to Contribution: Characterizing Coding Agent in Open-Source Software\"", "type": "paper", "meta": { "type": "paper", "title": "\"From Conversation to Contribution: Characterizing Coding Agent in Open-Source Software\"", "authors": "Zihan Fang, Yueke Zhang, Ningzhi Tang, Collin McMillan, Toby Jia-Jun Li, Yu Huang", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2607.05677", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-07-06", "updated_at": "2026-07-06", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "coding-agent", "computer-use", "multi-agent", "tool-use", "workflow-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.SE", "cs.HC" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "14", "collection_queries": "coding-agent" } }, { "collection": "papers", "path": "papers/items/2026-2607-05690-memory-in-the-loop-in-process-retrieval-as-extendedworking-memory-for-language-a.md", "title": "\"Memory in the Loop: In-Process Retrieval as ExtendedWorking Memory for Language Agents\"", "type": "paper", "meta": { "type": "paper", "title": "\"Memory in the Loop: In-Process Retrieval as ExtendedWorking Memory for Language Agents\"", "authors": "Yusuf Khan, Carlo Lipizzi", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2607.05690", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-07-06", "updated_at": "2026-07-06", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "memory", "rag", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI", "cs.CL" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "15", "collection_queries": "language-agent" } }, { "collection": "papers", "path": "papers/items/2026-2607-05743-the-balkanization-of-execution-security-research-for-ai-coding-agents-isolation-.md", "title": "\"The Balkanization of Execution-Security Research for AI Coding Agents: Isolation, Access Control, and Time-of-Check-to-Time-of-Use Vulnerabilities\"", "type": "paper", "meta": { "type": "paper", "title": "\"The Balkanization of Execution-Security Research for AI Coding Agents: Isolation, Access Control, and Time-of-Check-to-Time-of-Use Vulnerabilities\"", "authors": "Mohammadreza Rashidi", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2607.05743", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-07-07", "updated_at": "2026-07-07", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "coding-agent", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CR", "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "15", "collection_queries": "agentic-ai, coding-agent" } }, { "collection": "papers", "path": "papers/items/2026-2607-05772-detecting-vulnerability-inducing-commits-via-multi-stage-reasoning-with-llm-base.md", "title": "Detecting Vulnerability-Inducing Commits via Multi-Stage Reasoning with LLM-Based Agents", "type": "paper", "meta": { "type": "paper", "title": "Detecting Vulnerability-Inducing Commits via Multi-Stage Reasoning with LLM-Based Agents", "authors": "Liyou Chen, Hailong Sun, Xiang Gao, Yue Pan", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2607.05772", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-07-07", "updated_at": "2026-07-07", "status": "queued", "relevance": "high", "topics": [ "agent-safety", "multi-agent", "rag", "reasoning", "tool-use", "workflow-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.SE" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "14", "collection_queries": "multi-agent-llm" } }, { "collection": "papers", "path": "papers/items/2026-2607-05773-beyond-static-evaluation-building-simulation-environments-for-scalable-agentic-r.md", "title": "\"Beyond Static Evaluation: Building Simulation Environments for Scalable Agentic Reinforcement Learning\"", "type": "paper", "meta": { "type": "paper", "title": "\"Beyond Static Evaluation: Building Simulation Environments for Scalable Agentic Reinforcement Learning\"", "authors": "Akshay Arora, Ishan Nigam, Ashutosh Aggarwal, Shefali Bansal, Krishna Singh, Sweta Kumari, Nikhil Mittal, Shariq Farhan, et al.", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2607.05773", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-07-07", "updated_at": "2026-07-07", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "computer-use", "tool-use", "world-model" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "21", "collection_queries": "autonomous-agent-llm, tool-use, web-gui-agent" } }, { "collection": "papers", "path": "papers/items/2026-2607-05775-beyond-the-leaderboard-a-synthesis-of-tool-use-planning-and-reasoning-failures-i.md", "title": "\"Beyond the Leaderboard: A Synthesis of Tool-Use, Planning, and Reasoning Failures in Large Language Model Agents\"", "type": "paper", "meta": { "type": "paper", "title": "\"Beyond the Leaderboard: A Synthesis of Tool-Use, Planning, and Reasoning Failures in Large Language Model Agents\"", "authors": "Wael Albayaydh, Rui Zhao, Ivan Flechais", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2607.05775", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-07-07", "updated_at": "2026-07-07", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "coding-agent", "embodied-agent", "multi-agent", "planning", "reasoning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "23", "collection_queries": "llm-agent, multi-agent-llm, planning-agent, tool-use" } }, { "collection": "papers", "path": "papers/items/2026-2607-05794-from-passive-retrieval-to-active-memory-navigation-learning-to-use-memory-as-a-s.md", "title": "\"From Passive Retrieval to Active Memory Navigation: Learning to Use Memory as a Structured Action Space\"", "type": "paper", "meta": { "type": "paper", "title": "\"From Passive Retrieval to Active Memory Navigation: Learning to Use Memory as a Structured Action Space\"", "authors": "Yue Xu, Yutao Sun, Yihao Liu, Mengyu Zhou, Jiayi Qiao, Lu Ma, Kai Tang, Wenjie Wang, et al.", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2607.05794", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-07-07", "updated_at": "2026-07-07", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "embodied-agent", "memory", "rag", "reasoning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "13", "collection_queries": "tool-use" } }, { "collection": "papers", "path": "papers/items/2026-2607-05805-onnes-a-physics-grounded-multi-agent-llm-simulator-for-cryogenic-fault-diagnosis.md", "title": "\"Onnes: A Physics-Grounded Multi-Agent LLM Simulator for Cryogenic Fault Diagnosis in Quantum Computing Infrastructure\"", "type": "paper", "meta": { "type": "paper", "title": "\"Onnes: A Physics-Grounded Multi-Agent LLM Simulator for Cryogenic Fault Diagnosis in Quantum Computing Infrastructure\"", "authors": "Praneeth Narisetty, Uday Kumar Reddy Kattamanchi, Shiva Nagendra Babu Kore", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2607.05805", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-07-07", "updated_at": "2026-07-07", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "multi-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI", "cs.LG", "quant-ph" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "15", "collection_queries": "llm-agent, multi-agent-llm" } }, { "collection": "papers", "path": "papers/items/2026-2607-05915-pcbworld-a-benchmark-environment-for-engine-grounded-pcb-design-automation.md", "title": "\"PCBWorld: A Benchmark Environment for Engine-Grounded PCB Design Automation\"", "type": "paper", "meta": { "type": "paper", "title": "\"PCBWorld: A Benchmark Environment for Engine-Grounded PCB Design Automation\"", "authors": "Hyungseok Song, Junseok Park, Won-Seok Choi, Seohui Bae, Han-Seul Jeong, Youngjoon Park, Soonyoung Lee", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2607.05915", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-07-07", "updated_at": "2026-07-07", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "tool-use", "workflow-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "15", "collection_queries": "agentic-ai, llm-agent, tool-use" } }, { "collection": "papers", "path": "papers/items/2026-2607-06000-context-to-execution-integrity-for-llm-agents.md", "title": "Context-to-Execution Integrity for LLM Agents", "type": "paper", "meta": { "type": "paper", "title": "Context-to-Execution Integrity for LLM Agents", "authors": "Igor Santos-Grueiro", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2607.06000", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-07-07", "updated_at": "2026-07-07", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "coding-agent", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CR" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "16", "collection_queries": "agent-evaluation, coding-agent, llm-agent" } }, { "collection": "papers", "path": "papers/items/2026-2607-06001-information-limits-and-attractor-dynamics-in-economies-of-frontier-llm-agents-a-.md", "title": "\"Information Limits and Attractor Dynamics in Economies of Frontier LLM Agents: A Pre-Registered Test\"", "type": "paper", "meta": { "type": "paper", "title": "\"Information Limits and Attractor Dynamics in Economies of Frontier LLM Agents: A Pre-Registered Test\"", "authors": "Cheng Qian", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2607.06001", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-07-07", "updated_at": "2026-07-07", "status": "queued", "relevance": "high", "topics": [ "agent-safety", "multi-agent", "reasoning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI", "cs.MA" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "16", "collection_queries": "llm-agent, multi-agent-llm" } }, { "collection": "papers", "path": "papers/items/2026-2607-06008-polyworkbench-benchmarking-multilingual-long-horizon-llm-agents.md", "title": "\"PolyWorkBench: Benchmarking Multilingual Long-Horizon LLM Agents\"", "type": "paper", "meta": { "type": "paper", "title": "\"PolyWorkBench: Benchmarking Multilingual Long-Horizon LLM Agents\"", "authors": "Hongliang Li, Yijin Liu, Zhiwei Zhang, Zihe Liu, Xinyue Lou, Jinan Xu, Fandong Meng, Kaiyu Huang", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2607.06008", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-07-07", "updated_at": "2026-07-07", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "computer-use", "planning", "reasoning", "tool-use", "workflow-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI", "cs.CL" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "25", "collection_queries": "agent-evaluation, llm-agent, planning-agent, tool-use" } }, { "collection": "papers", "path": "papers/items/2026-2607-06080-from-blueprint-to-reality-modeling-and-applying-putnam-s-social-capital-theory-w.md", "title": "\"From Blueprint to Reality: Modeling and Applying Putnam's Social Capital Theory with LLM-based Multi-agent Simulations\"", "type": "paper", "meta": { "type": "paper", "title": "\"From Blueprint to Reality: Modeling and Applying Putnam's Social Capital Theory with LLM-based Multi-agent Simulations\"", "authors": "Shiyi Ling, Zhi Zheng, Hui Zheng, Wenjun Xue, Feng Ye, Tong Xu", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2607.06080", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-07-07", "updated_at": "2026-07-07", "status": "queued", "relevance": "high", "topics": [ "agent-safety", "multi-agent", "rag", "tool-use", "world-model" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CL", "cs.AI", "cs.SI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "17", "collection_queries": "llm-agent, multi-agent-llm" } }, { "collection": "papers", "path": "papers/items/2026-2607-06101-agents-that-teach-towards-designing-incidental-learning-back-into-ai-assisted-so.md", "title": "\"Agents That Teach: Towards Designing Incidental Learning Back into AI-Assisted Software Development\"", "type": "paper", "meta": { "type": "paper", "title": "\"Agents That Teach: Towards Designing Incidental Learning Back into AI-Assisted Software Development\"", "authors": "Rohit Mehra, Samdyuti Suri, Prithviraj K Tagadinamani, Kapil Singi, Vikrant Kaulgud, Adam P. Burden", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2607.06101", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-07-07", "updated_at": "2026-07-07", "status": "queued", "relevance": "high", "topics": [ "agent-safety", "coding-agent", "computer-use", "multi-agent", "rag", "reasoning", "tool-use", "workflow-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.SE", "cs.AI", "cs.CY", "cs.HC" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "16", "collection_queries": "coding-agent" } }, { "collection": "papers", "path": "papers/items/2026-2607-06118-webretriever-a-large-scale-comprehensive-benchmark-for-efficient-web-agent-evalu.md", "title": "\"WebRetriever: A Large-Scale Comprehensive Benchmark for Efficient Web Agent Evaluation\"", "type": "paper", "meta": { "type": "paper", "title": "\"WebRetriever: A Large-Scale Comprehensive Benchmark for Efficient Web Agent Evaluation\"", "authors": "Wei Dong, Tianyu Fu, Zhe Yu, Hanning Wang, Anyang Su, Zhizhou Fang, Yuyang Chen, Shuo Wang, et al.", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2607.06118", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-07-07", "updated_at": "2026-07-07", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "computer-use", "embodied-agent", "rag", "tool-use", "workflow-agent" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CV", "cs.MM" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "21", "collection_queries": "agent-evaluation, web-gui-agent" } }, { "collection": "papers", "path": "papers/items/2026-2607-06140-curateevo-data-curation-evolving-for-agentic-post-training.md", "title": "\"CurateEvo: Data-Curation Evolving for Agentic Post-Training\"", "type": "paper", "meta": { "type": "paper", "title": "\"CurateEvo: Data-Curation Evolving for Agentic Post-Training\"", "authors": "Dingzirui Wang, Xuanliang Zhang, Keyan Xu, Qingfu Zhu, Wanxiang Che", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2607.06140", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-07-07", "updated_at": "2026-07-07", "status": "queued", "relevance": "high", "topics": [ "memory", "planning", "rag" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CL" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "14", "collection_queries": "llm-agent" } }, { "collection": "papers", "path": "papers/items/2026-2607-06157-llm-agents-for-deliberative-collaboration-a-study-on-joint-decision-making-under.md", "title": "\"LLM Agents for Deliberative Collaboration: A Study on Joint Decision Making Under Partial Observability\"", "type": "paper", "meta": { "type": "paper", "title": "\"LLM Agents for Deliberative Collaboration: A Study on Joint Decision Making Under Partial Observability\"", "authors": "Chenxu Wang, Yongkun Yang, Boyuan Du, Shiwei Lin, Huaping Liu", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2607.06157", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-07-07", "updated_at": "2026-07-07", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "multi-agent", "reasoning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CL", "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "18", "collection_queries": "llm-agent, multi-agent-llm" } }, { "collection": "papers", "path": "papers/items/2026-2607-06195-logichunter-testing-llm-agent-frameworks-with-an-agentic-oracle.md", "title": "\"LogicHunter: Testing LLM Agent Frameworks with an Agentic Oracle\"", "type": "paper", "meta": { "type": "paper", "title": "\"LogicHunter: Testing LLM Agent Frameworks with an Agentic Oracle\"", "authors": "Minghui Long, Yanjie Zhao, Haoyu Wang", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2607.06195", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-07-07", "updated_at": "2026-07-07", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "computer-use", "memory" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.SE" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "17", "collection_queries": "llm-agent" } }, { "collection": "papers", "path": "papers/items/2026-2607-06223-information-gain-based-rollout-policy-optimization-an-adaptive-tree-structured-r.md", "title": "\"Information Gain-based Rollout Policy Optimization: An Adaptive Tree-Structured Rollout Approach for Multi-Turn LLM Agents\"", "type": "paper", "meta": { "type": "paper", "title": "\"Information Gain-based Rollout Policy Optimization: An Adaptive Tree-Structured Rollout Approach for Multi-Turn LLM Agents\"", "authors": "Yijun Zhang, Fan Xu, Jiaxin Ding, Yule Xie, Shiqing Gao, Xin Ding, Haoxiang Zhang, Luoyi Fu, et al.", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2607.06223", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-07-07", "updated_at": "2026-07-07", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "computer-use", "planning", "rag" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "15", "collection_queries": "llm-agent" } }, { "collection": "papers", "path": "papers/items/2026-2607-06273-agenttether-graph-guided-diagnosis-and-runtime-intervention-for-reliable-llm-age.md", "title": "\"AgentTether: Graph-Guided Diagnosis and Runtime Intervention for Reliable LLM Agent Operation\"", "type": "paper", "meta": { "type": "paper", "title": "\"AgentTether: Graph-Guided Diagnosis and Runtime Intervention for Reliable LLM Agent Operation\"", "authors": "Chenyu Zhao, Shenglin Zhang, Wenwei Gu, Yongqian Sun, Dan Pei, Chetan Bansal, Saravan Rajmohan, Minghua Ma", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2607.06273", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-07-07", "updated_at": "2026-07-07", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "computer-use", "memory", "reasoning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.SE" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "17", "collection_queries": "llm-agent, tool-use" } }, { "collection": "papers", "path": "papers/items/2026-2607-06341-harnessing-code-agents-for-automatic-software-verification.md", "title": "Harnessing Code Agents for Automatic Software Verification", "type": "paper", "meta": { "type": "paper", "title": "Harnessing Code Agents for Automatic Software Verification", "authors": "Shuangxiang Kan, Shuanglong Kan, Sebastian Ertel", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2607.06341", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-07-07", "updated_at": "2026-07-07", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "coding-agent", "memory", "rag", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.FL", "cs.AI", "cs.SE" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "13", "collection_queries": "coding-agent" } }, { "collection": "papers", "path": "papers/items/2026-2607-06411-rubench-a-repository-level-agentic-coding-benchmark-with-natively-authored-russi.md", "title": "\"RuBench: A Repository-Level Agentic Coding Benchmark with Natively Authored Russian Task Specifications\"", "type": "paper", "meta": { "type": "paper", "title": "\"RuBench: A Repository-Level Agentic Coding Benchmark with Natively Authored Russian Task Specifications\"", "authors": "Evgeny Shilov", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2607.06411", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-07-07", "updated_at": "2026-07-07", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "coding-agent", "reasoning" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.SE", "cs.AI", "cs.CL" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "16", "collection_queries": "agent-evaluation, coding-agent" } }, { "collection": "papers", "path": "papers/items/2026-2607-06413-an-experimental-design-approach-to-evaluating-agentic-ai-s-autonomous-model-disc.md", "title": "\"An Experimental Design Approach to Evaluating Agentic AI's Autonomous Model Discovery\"", "type": "paper", "meta": { "type": "paper", "title": "\"An Experimental Design Approach to Evaluating Agentic AI's Autonomous Model Discovery\"", "authors": "Hao He, Xueying Liu, Chris J. Kuhlman, Xinwei Deng", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2607.06413", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-07-07", "updated_at": "2026-07-07", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "coding-agent", "reasoning" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "stat.ME", "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "16", "collection_queries": "agentic-ai, coding-agent" } }, { "collection": "papers", "path": "papers/items/2026-2607-06452-from-voting-to-agent-collaboration-answer-type-aware-llm-pipelines-for-bioasq-14.md", "title": "\"From Voting to Agent Collaboration: Answer-Type-Aware LLM Pipelines for BioASQ 14b\"", "type": "paper", "meta": { "type": "paper", "title": "\"From Voting to Agent Collaboration: Answer-Type-Aware LLM Pipelines for BioASQ 14b\"", "authors": "Taeyun Roh, Eunha Lee, Wonjune Jang, Sohyun Chung, Junha Jung, Jaewoo Kang", "year": "2026", "venue": "arXiv", "url": "https://arxiv.org/abs/2607.06452", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "published_at": "2026-07-07", "updated_at": "2026-07-07", "status": "queued", "relevance": "high", "topics": [ "agent-evaluation", "multi-agent", "rag", "reasoning", "tool-use" ], "methods": [], "benchmarks": [], "models": [], "datasets": [ "cs.CL", "cs.AI" ], "related_concepts": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "collection_score": "14", "collection_queries": "multi-agent-llm" } }, { "collection": "papers", "path": "papers/items/2026-agent-safety-benchmark-taxonomy.md", "title": "\"Taxonomy and Consistency Analysis of Safety Benchmarks for AI Agents\"", "type": "paper", "meta": { "type": "paper", "title": "\"Taxonomy and Consistency Analysis of Safety Benchmarks for AI Agents\"", "authors": [], "year": "2026", "venue": [], "url": "https://arxiv.org/html/2605.16282v1", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "status": "queued", "relevance": "medium", "topics": [ "agent-safety", "agent-evaluation" ], "methods": [ "benchmark-taxonomy", "coverage-matrix" ], "benchmarks": [ "agent-safety-benchmarks" ], "models": [], "datasets": [], "related_concepts": [ "guardrail", "red-teaming" ], "related_jobs": [ "2026-07-08-tencent-cloud-ai-agent-test-engineer" ], "related_experiments": [], "related_projects": [] } }, { "collection": "papers", "path": "papers/items/2026-king-dialogue-swebench.md", "title": "\"Dialogue-SWEBench: A Benchmark for Dialogue-Driven Coding Agents\"", "type": "paper", "meta": { "type": "paper", "title": "\"Dialogue-SWEBench: A Benchmark for Dialogue-Driven Coding Agents\"", "authors": "Brendan King, Jeffrey Flanigan", "year": "2026", "venue": [], "url": "https://arxiv.org/html/2606.13995v1", "code_url": "https://jlab-nlp.github.io/dialogue-swe-bench/", "source": "arxiv", "collected_at": "2026-07-08", "status": "skimmed", "relevance": "high", "topics": [ "coding-agent", "human-in-the-loop", "agent-evaluation" ], "methods": [ "user-simulator", "dialogue-benchmark", "schema-guided-agent" ], "benchmarks": [ "Dialogue-SWEBench" ], "models": [], "datasets": [ "SWE-Bench Verified" ], "related_concepts": [ "human-in-the-loop", "coding-agent" ], "related_jobs": [ "2026-07-08-baidu-aidu-agent-fullstack-engineer-beijing" ], "related_experiments": [], "related_projects": [] } }, { "collection": "papers", "path": "papers/items/2026-memory-agent-survey.md", "title": "\"Memory for Autonomous LLM Agents: Mechanisms, Evaluation, and Emerging Frontiers\"", "type": "paper", "meta": { "type": "paper", "title": "\"Memory for Autonomous LLM Agents: Mechanisms, Evaluation, and Emerging Frontiers\"", "authors": [], "year": "2026", "venue": [], "url": "https://arxiv.org/html/2603.07670v1", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "status": "skimmed", "relevance": "high", "topics": [ "memory", "agent-architecture", "agent-evaluation" ], "methods": [ "write-manage-read-loop", "retrieval-augmented-memory", "reflective-memory", "hierarchical-context" ], "benchmarks": [], "models": [], "datasets": [], "related_concepts": [ "memory", "context-engineering" ], "related_jobs": [ "2026-07-08-baidu-aidu-agent-algorithm-engineer-beijing", "2026-07-08-bytedance-seed-llm-agent-research-engineer" ], "related_experiments": [], "related_projects": [] } }, { "collection": "papers", "path": "papers/items/2026-pham-swe-evo.md", "title": "\"SWE-EVO: Benchmarking Coding Agents in Long-Horizon Software Evolution Scenarios\"", "type": "paper", "meta": { "type": "paper", "title": "\"SWE-EVO: Benchmarking Coding Agents in Long-Horizon Software Evolution Scenarios\"", "authors": "Minh Vu Thai Pham, Tue Le, Dung Nguyen Manh, Huy Nhat Phan, Nghi D. Q. Bui", "year": "2026", "venue": [], "url": "https://arxiv.org/html/2512.18470v5", "code_url": "https://github.com/SWE-EVO/SWE-EVO", "source": "arxiv", "collected_at": "2026-07-08", "status": "skimmed", "relevance": "high", "topics": [ "coding-agent", "agent-evaluation" ], "methods": [ "long-horizon-software-evolution", "benchmark" ], "benchmarks": [ "SWE-EVO" ], "models": [ "OpenAI", "DeepSeek", "Zhipu", "Qwen", "Moonshot" ], "datasets": [ "release-notes" ], "related_concepts": [ "coding-agent", "long-horizon-task" ], "related_jobs": [ "2026-07-08-deepseek-agent-hiring-wave-beijing-hangzhou" ], "related_experiments": [], "related_projects": [] } }, { "collection": "papers", "path": "papers/items/2026-wang-evomembench.md", "title": "\"EvoMemBench: Benchmarking Agent Memory from a Self-Evolving Perspective\"", "type": "paper", "meta": { "type": "paper", "title": "\"EvoMemBench: Benchmarking Agent Memory from a Self-Evolving Perspective\"", "authors": "Yuyao Wang, Zhongjian Zhang, Mo Chi, Kaichi Yu, Yuhan Li, Miao Peng, Bing Tong, Chen Zhang, Yan Zhou, Jia Li", "year": "2026", "venue": [], "url": "https://arxiv.org/html/2605.18421", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "status": "skimmed", "relevance": "high", "topics": [ "memory", "agent-evaluation" ], "methods": [ "memory-benchmark", "self-evolving-agent" ], "benchmarks": [ "EvoMemBench" ], "models": [], "datasets": [], "related_concepts": [ "memory", "procedural-memory" ], "related_jobs": [ "2026-07-08-baidu-aidu-agent-algorithm-engineer-beijing", "2026-07-08-bytedance-seed-llm-agent-research-engineer" ], "related_experiments": [], "related_projects": [] } }, { "collection": "industry", "path": "industry/items/2026-07-08-bytedance-seed2-1-agent-productivity.md", "title": "\"Seed2.1 Officially Released: Advancing AI Productivity\"", "type": "industry", "meta": { "type": "industry", "company": "ByteDance Seed", "team": "Seed", "title": "\"Seed2.1 Officially Released: Advancing AI Productivity\"", "url": "https://seed.bytedance.com/en/blog/seed2-1-officially-released-advancing-ai-productivity", "source_name": "ByteDance Seed", "source_type": "research-blog", "source_quality": "official", "published_at": [], "collected_at": "2026-07-08", "status": "analyzed", "topics": [ "agent", "coding-agent", "gui-agent" ], "implementation_signals": [ "reinforcement-learning", "gui-tool-use", "mcp", "crowdsourced-evaluation", "internal-benchmark" ], "product_area": [ "productivity", "coding-agent", "office-agent" ], "models": [ "Seed2.1" ], "tools": [ "MCP" ], "benchmarks": [ "CreativeWork", "ProgramBench" ], "related_papers": [ "2026-pham-swe-evo" ], "related_jobs": [ "2026-07-08-bytedance-seed-llm-agent-research-engineer" ], "related_experiments": [], "related_projects": [], "evidence_level": "high", "relevance": "high" } }, { "collection": "industry", "path": "industry/items/2026-07-08-google-agent-security-roadmap.md", "title": "Securing the future of AI agents", "type": "industry", "meta": { "type": "industry", "company": "Google DeepMind", "team": [], "title": "Securing the future of AI agents", "url": "https://deepmind.google/blog/securing-the-future-of-ai-agents/", "source_name": "Google DeepMind", "source_type": "technical-report", "source_quality": "official", "published_at": [], "collected_at": "2026-07-08", "status": "analyzed", "topics": [ "agent-safety", "agent-security", "governance" ], "implementation_signals": [ "defense-in-depth", "sandboxing", "permission-control", "prompt-injection-resistance" ], "product_area": [ "internal-agents", "enterprise-agent" ], "models": [], "tools": [], "benchmarks": [], "related_papers": [ "2025-vijayvargiya-openagentsafety", "2026-agent-safety-benchmark-taxonomy" ], "related_jobs": [ "2026-07-08-tencent-cloud-ai-agent-test-engineer" ], "related_experiments": [], "related_projects": [], "evidence_level": "high", "relevance": "high" } }, { "collection": "industry", "path": "industry/items/2026-07-08-google-gemini-computer-use.md", "title": "Introducing computer use in Gemini 3.5 Flash", "type": "industry", "meta": { "type": "industry", "company": "Google DeepMind", "team": "Gemini", "title": "Introducing computer use in Gemini 3.5 Flash", "url": "https://deepmind.google/blog/introducing-computer-use-in-gemini-3-5-flash/", "source_name": "Google DeepMind / Google Blog", "source_type": "product-blog", "source_quality": "official", "published_at": "2026-06-24", "collected_at": "2026-07-08", "status": "analyzed", "topics": [ "computer-use", "agent", "tool-use" ], "implementation_signals": [ "browser-control", "gui-control", "safety", "human-in-the-loop" ], "product_area": [ "enterprise-agent", "automation" ], "models": [ "Gemini 3.5 Flash" ], "tools": [ "Gemini API", "Gemini Enterprise Agent Platform" ], "benchmarks": [], "related_papers": [ "2025-vijayvargiya-openagentsafety" ], "related_jobs": [ "2026-07-08-baidu-aidu-agent-fullstack-engineer-beijing" ], "related_experiments": [], "related_projects": [], "evidence_level": "high", "relevance": "high" } }, { "collection": "industry", "path": "industry/items/2026-07-08-microsoft-agentrx.md", "title": "\"Systematic debugging for AI agents: Introducing the AgentRx framework\"", "type": "industry", "meta": { "type": "industry", "company": "Microsoft", "team": "Microsoft Research", "title": "\"Systematic debugging for AI agents: Introducing the AgentRx framework\"", "url": "https://www.microsoft.com/en-us/research/blog/systematic-debugging-for-ai-agents-introducing-the-agentrx-framework/", "source_name": "Microsoft Research", "source_type": "research-blog", "source_quality": "official", "published_at": [], "collected_at": "2026-07-08", "status": "analyzed", "topics": [ "agent-evaluation", "observability", "debugging" ], "implementation_signals": [ "trajectory-normalization", "constraint-synthesis", "constraint-checking", "failure-taxonomy" ], "product_area": [ "agent-debugging" ], "models": [], "tools": [ "AgentRx" ], "benchmarks": [ "AgentRx Benchmark" ], "related_papers": [ "2025-vijayvargiya-openagentsafety" ], "related_jobs": [ "2026-07-08-tencent-cloud-ai-agent-test-engineer" ], "related_experiments": [], "related_projects": [], "evidence_level": "high", "relevance": "high" } }, { "collection": "industry", "path": "industry/items/2026-07-08-microsoft-skillopt.md", "title": "\"SkillOpt: Agent skills as trainable parameters\"", "type": "industry", "meta": { "type": "industry", "company": "Microsoft", "team": "Microsoft Research", "title": "\"SkillOpt: Agent skills as trainable parameters\"", "url": "https://www.microsoft.com/en-us/research/blog/skillopt-agent-skills-as-trainable-parameters/", "source_name": "Microsoft Research", "source_type": "research-blog", "source_quality": "official", "published_at": "2026-06-30", "collected_at": "2026-07-08", "status": "analyzed", "topics": [ "agent", "prompt-optimization", "skill-learning" ], "implementation_signals": [ "skill-files", "validation-gating", "bounded-edits", "eval-loop" ], "product_area": [ "agent-framework" ], "models": [], "tools": [], "benchmarks": [ "six-benchmark-evaluation" ], "related_papers": [], "related_jobs": [ "2026-07-08-deepseek-agent-hiring-wave-beijing-hangzhou" ], "related_experiments": [], "related_projects": [], "evidence_level": "high", "relevance": "high" } }, { "collection": "industry", "path": "industry/items/2026-07-08-microsoft-state-bench.md", "title": "\"Introducing STATE-Bench: a benchmark for AI agent memory\"", "type": "industry", "meta": { "type": "industry", "company": "Microsoft", "team": "Microsoft Open Source", "title": "\"Introducing STATE-Bench: a benchmark for AI agent memory\"", "url": "https://opensource.microsoft.com/blog/2026/05/19/introducing-state-bench-a-benchmark-for-ai-agent-memory/", "source_name": "Microsoft Open Source Blog", "source_type": "benchmark", "source_quality": "official", "published_at": "2026-05-19", "collected_at": "2026-07-08", "status": "analyzed", "topics": [ "memory", "agent-evaluation", "enterprise-ai" ], "implementation_signals": [ "stateful-environment", "user-simulator", "deterministic-assertions", "bring-your-own-memory" ], "product_area": [ "customer-support", "travel", "shopping" ], "models": [], "tools": [ "STATE-Bench" ], "benchmarks": [ "STATE-Bench" ], "related_papers": [ "2026-memory-agent-survey", "2026-wang-evomembench" ], "related_jobs": [ "2026-07-08-baidu-aidu-agent-algorithm-engineer-beijing" ], "related_experiments": [], "related_projects": [], "evidence_level": "high", "relevance": "high" } }, { "collection": "industry", "path": "industry/items/2026-07-08-openai-agents-transforming-work.md", "title": "How agents are transforming work", "type": "industry", "meta": { "type": "industry", "company": "OpenAI", "team": [], "title": "How agents are transforming work", "url": "https://openai.com/index/how-agents-are-transforming-work/", "source_name": "OpenAI", "source_type": "research-blog", "source_quality": "official", "published_at": [], "collected_at": "2026-07-08", "status": "analyzed", "topics": [ "agent", "enterprise-ai", "coding-agent" ], "implementation_signals": [ "adoption-metrics", "workflow-change" ], "product_area": [ "coding-agent", "enterprise-work" ], "models": [ "Codex" ], "tools": [], "benchmarks": [], "related_papers": [], "related_jobs": [ "2026-07-08-deepseek-agent-hiring-wave-beijing-hangzhou" ], "related_experiments": [], "related_projects": [], "evidence_level": "medium", "relevance": "high" } }, { "collection": "industry", "path": "industry/items/2026-07-08-openai-in-house-data-agent.md", "title": "Inside OpenAI's in-house data agent", "type": "industry", "meta": { "type": "industry", "company": "OpenAI", "team": "data / engineering", "title": "Inside OpenAI's in-house data agent", "url": "https://openai.com/index/inside-our-in-house-data-agent/", "source_name": "OpenAI", "source_type": "engineering-blog", "source_quality": "official", "published_at": [], "collected_at": "2026-07-08", "status": "analyzed", "topics": [ "agent", "data-agent", "enterprise-ai" ], "implementation_signals": [ "memory", "rag", "mcp", "eval", "permissions" ], "product_area": [ "data-analysis", "internal-tools" ], "models": [ "GPT-5.2", "Codex" ], "tools": [ "Evals API", "Embeddings API", "MCP" ], "benchmarks": [], "related_papers": [ "2026-memory-agent-survey" ], "related_jobs": [ "2026-07-08-baidu-aidu-agent-fullstack-engineer-beijing" ], "related_experiments": [], "related_projects": [], "evidence_level": "high", "relevance": "high" } }, { "collection": "industry", "path": "industry/items/2026-07-08-qwen-agentworld.md", "title": "\"Qwen-AgentWorld: Language World Models for General Agents\"", "type": "industry", "meta": { "type": "industry", "company": "Alibaba Qwen", "team": "Qwen", "title": "\"Qwen-AgentWorld: Language World Models for General Agents\"", "url": "https://qwen.ai/blog?id=qwen-agentworld", "source_name": "Qwen Blog", "source_type": "research-blog", "source_quality": "official", "published_at": [], "collected_at": "2026-07-08", "status": "queued", "topics": [ "agent", "world-model", "evaluation" ], "implementation_signals": [ "simulated-environment", "world-model", "multi-domain-agent" ], "product_area": [ "general-agent" ], "models": [ "Qwen" ], "tools": [], "benchmarks": [], "related_papers": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "evidence_level": "medium", "relevance": "high" } } ]